From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 06:58:42
cdef5a1: PCI: clean up rescan_bus_bridge_resize
7b6deb4: PCI: make pci_rescan_bus_bridge_resize use pci_scan_bridge instead
876bd4a: PCI: Add pci_bus_add_single_device()
6e5f346: PCI, sysfs: create rescan_bridge under /sys/.../pci/devices/... for pci bridges
3c92113: PCI, sys: Use device_type and attr_groups with pci dev
e6d3c54: PCI, pciehp: Remove not needed bus number range checking
2306e31: PCI: Double checking setting for bus register and bus struct.
daddc90: pcmcia: remove workaround for fixing pci parent bus subordinate
6a88939: PCI: Seperate child bus scanning to two passes overall
60c7d94: PCI: kill pci_fixup_parent_subordinate_busnr()
e9e72be: PCI: Allocate bus range instead of use max blindly
f22af7f: PCI: Strict checking of valid range for bridge
8656ea6: PCI: Probe safe range that we can use for unassigned bridge.
4023fc3: PCI: Add pci_bus_extend/shrink_top()
a066c95: PCI, parisc: Register busn_res for root buses
d069fbb: PCI, powerpc: Register busn_res for root buses
f00e514: PCI, ia64: Register busn_res for root buses
1a203f1: PCI, x86: Register busn_res for root buses
8b1b01b: PCI: Add busn_res tracking in core
a9ccca7: PCI: add /proc/iobusn
ec1cdb3: PCI: Add busn_res operation functions
2a09fbd: Make %pR could handle bus resource with domain
9c83a59: PCI: add busn inline helper
b18cd6a: PCI: Add iobusn_resource
Set up iobusn_resource tree, and register bus number range to it.
Later when need to find bus range, will try to allocate from the tree
Need to test on arches other than x86. esp for ia64 and powerpc that support
more than on peer root buses.
last five patches are rescan cleanup that make use busn alloc.
could be found at:
git://git.kernel.org/pub/scm/linux/kernel/git/yinghai/linux-yinghai.git for-pci-busn-alloc
-v2: according to Jesse, split to more small patches.
-v3: address some request from Bjorn. like make use %pR for busn_res debug print
out, and move the comment change with code change.
-v4: fixes the problem about rescan that Bjorn found.
-v5: add /proc/iobusn that is requested by Bjorn.
remove old workaround from pciehp.
add rescan_bridge for pci bridge in /sys
Thanks
Yinghai
Documentation/ABI/testing/sysfs-bus-pci | 10 +
arch/ia64/pci/pci.c | 2 +
arch/powerpc/kernel/pci-common.c | 7 +-
arch/x86/include/asm/topology.h | 3 +-
arch/x86/pci/acpi.c | 8 +-
arch/x86/pci/bus_numa.c | 8 +-
arch/x86/pci/common.c | 11 +-
drivers/parisc/dino.c | 2 +
drivers/parisc/lba_pci.c | 3 +
drivers/pci/bus.c | 40 +++
drivers/pci/hotplug/pciehp_pci.c | 12 +-
drivers/pci/pci-sysfs.c | 59 ++++-
drivers/pci/pci.h | 1 +
drivers/pci/probe.c | 460 +++++++++++++++++++++++++------
drivers/pci/remove.c | 1 +
drivers/pcmcia/yenta_socket.c | 75 -----
include/linux/ioport.h | 18 ++
include/linux/pci.h | 9 +
kernel/resource.c | 50 ++++
lib/vsprintf.c | 29 ++-
20 files changed, 617 insertions(+), 191 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 06:58:48
also add busn_res into struct pci_bus.
will use them to have bus number resource tree.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
Cc: Andrew Morton <akpm@linux-foundation.org>
---
include/linux/ioport.h | 1 +
include/linux/pci.h | 1 +
kernel/resource.c | 8 ++++++++
3 files changed, 10 insertions(+), 0 deletions(-)
@@ -136,6 +136,7 @@ struct resource {/* PC/ISA/whatever - the normal PC address spaces: IO and memory */externstructresourceioport_resource;externstructresourceiomem_resource;+externstructresourceiobusn_resource;externstructresource*request_resource_conflict(structresource*root,structresource*new);externintrequest_resource(structresource*root,structresource*new);
@@ -419,6 +419,7 @@ struct pci_bus {structlist_headslots;/* list of slots on this bus */structresource*resource[PCI_BRIDGE_RESOURCE_NUM];structlist_headresources;/* address space routed to this bus */+structresourcebusn_res;/* track registered bus num range */structpci_ops*ops;/* configuration access functions */void*sysdata;/* hook for sys-specific extension */
@@ -38,6 +38,14 @@ struct resource iomem_resource = {};EXPORT_SYMBOL(iomem_resource);+structresourceiobusn_resource={+.name="PCI busn",+.start=0,+.end=0xffffff,+.flags=IORESOURCE_BUS,+};+EXPORT_SYMBOL(iobusn_resource);+/* constraints to be met while allocating resources */structresource_constraint{resource_size_tmin,max,align;
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 06:59:04
convert back and forth with busn and domain_nr/bus_nr
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
Cc: Andrew Morton <akpm@linux-foundation.org>
---
include/linux/ioport.h | 17 +++++++++++++++++
1 files changed, 17 insertions(+), 0 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 06:59:12
will use them insert/update busn res in pci_bus
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 42 ++++++++++++++++++++++++++++++++++++++++++
include/linux/pci.h | 3 +++
2 files changed, 45 insertions(+), 0 deletions(-)
@@ -1622,6 +1622,48 @@ err_out:returnNULL;}+voidpci_bus_insert_busn_res(structpci_bus*b,intbus,intbus_max)+{+structresource*res=&b->busn_res;+structresource*parent_res=&iobusn_resource;+intret;++res->start=busn(pci_domain_nr(b),bus);+res->end=busn(pci_domain_nr(b),bus_max);+res->flags=IORESOURCE_BUS;++if(!pci_is_root_bus(b))+parent_res=&b->parent->busn_res;++ret=insert_resource(parent_res,res);++dev_printk(KERN_DEBUG,&b->dev,+"busn_res: %pR %s inserted under %pR\n",+res,ret?"can not be":"is",parent_res);+}++voidpci_bus_update_busn_res_end(structpci_bus*b,intbus_max)+{+structresource*res=&b->busn_res;+structresourceold_res=*res;++res->end=busn_update_bus_nr(res->end,bus_max);+dev_printk(KERN_DEBUG,&b->dev,+"busn_res: %pR end updated to %pR\n",+&old_res,res);+}++voidpci_bus_release_busn_res(structpci_bus*b)+{+structresource*res=&b->busn_res;+intret;++ret=release_resource(res);+dev_printk(KERN_DEBUG,&b->dev,+"busn_res: %pR %s released\n",+res,ret?"can not be":"is");+}+structpci_bus*__devinitpci_scan_root_bus(structdevice*parent,intbus,structpci_ops*ops,void*sysdata,structlist_head*resources){
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 06:59:23
May help debug purpose.
need to fill root buses name.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 3 +++
kernel/resource.c | 42 ++++++++++++++++++++++++++++++++++++++++++
2 files changed, 45 insertions(+), 0 deletions(-)
@@ -1594,6 +1594,8 @@ struct pci_bus *pci_create_root_bus(struct device *parent, int bus,b->number=b->secondary=bus;+sprintf(b->name,"PCI Bus %04x:%02x",pci_domain_nr(b),b->number);+/* Add initial resources to the bus */list_for_each_entry_safe(bus_res,n,resources,list)list_move_tail(&bus_res->list,&b->resources);
@@ -1631,6 +1633,7 @@ void pci_bus_insert_busn_res(struct pci_bus *b, int bus, int bus_max)res->start=busn(pci_domain_nr(b),bus);res->end=busn(pci_domain_nr(b),bus_max);res->flags=IORESOURCE_BUS;+res->name=b->name;if(!pci_is_root_bus(b))parent_res=&b->parent->busn_res;
@@ -1732,6 +1732,8 @@ void __devinit pcibios_scan_phb(struct pci_controller *hose)bus->secondary=hose->first_busno;hose->bus=bus;+pci_bus_insert_busn_res(bus,hose->first_busno,hose->last_busno);+/* Get probe mode and perform scan */mode=PCI_PROBE_NORMAL;if(node&&ppc_md.pci_probe_mode)
@@ -1742,8 +1744,11 @@ void __devinit pcibios_scan_phb(struct pci_controller *hose)of_scan_bus(node,bus);}-if(mode==PCI_PROBE_NORMAL)+if(mode==PCI_PROBE_NORMAL){+pci_bus_update_busn_res_end(bus,255);hose->last_busno=bus->subordinate=pci_scan_child_bus(bus);+pci_bus_update_busn_res_end(bus,bus->subordinate);+}/* Platform gets a chance to do some global fixups before*weproceedtoresourceallocation
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:06
extend or shrink bus and parent buses top (subordinate)
extended range is verified safe range, and stop at recorded parent_res.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 36 ++++++++++++++++++++++++++++++++++++
1 files changed, 36 insertions(+), 0 deletions(-)
@@ -1014,7 +1014,9 @@ static int __init dino_probe(struct parisc_device *dev)return0;}+pci_bus_insert_busn_res(bus,dino_current_bus,255);bus->subordinate=pci_scan_child_bus(bus);+pci_bus_update_busn_res_end(bus,bus->subordinate);/* This code *depends* on scanning being single threaded*ifitisn't,thisglobalbusnumbercountwillfail
@@ -1531,6 +1531,9 @@ lba_driver_probe(struct parisc_device *dev)return0;}+pci_bus_insert_busn_res(lba_bus,lba_dev->hba.bus_num.start,+lba_dev->hba.bus_num.end);+lba_bus->subordinate=pci_scan_child_bus(lba_bus);/* This is in lieu of calling pci_assign_unassigned_resources() */
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:16
children bridges busn range should be able to be allocated from parent bus range.
to avoid overlapping between sibling bridges on same bus.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 25 +++++++++++++++++++++++++
1 files changed, 25 insertions(+), 0 deletions(-)
@@ -770,6 +770,31 @@ reduce_needed_size:returnret;}+staticint__devinitpci_bridge_check_busn_broken(structpci_bus*bus,+structpci_dev*dev,+intsecondary,intsubordinate)+{+intbroken=0;++structresourcebusn_res;+intret;++memset(&busn_res,0,sizeof(structresource));+dev_printk(KERN_DEBUG,&dev->dev,+"check if busn %02x-%02x is in busn_res: %pR\n",+secondary,subordinate,&bus->busn_res);+ret=allocate_resource(&bus->busn_res,&busn_res,+(subordinate-secondary+1),+busn(pci_domain_nr(bus),secondary),+busn(pci_domain_nr(bus),subordinate),+1,NULL,NULL);+if(ret)+broken=1;+else+release_resource(&busn_res);++returnbroken;+}/**Ifit'sabridge,configureitandscanthebusbehindit.*ForCardBusbridges,wedon'tscanbehindasthedeviceswill
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:27
every bus have extra busn_res, and linked them toghter to iobusn_resource.
when need to find usable bus number range, try probe from iobusn_resource tree.
To avoid falling to small hole in the middle, we try from 8 spare bus.
if can not find 8 or more in the middle, will try to append 8 on top later.
then if can not append, will try to find 7 from the middle, then will try to append 7 on top.
then if can not append, will try to find 6 from the middle...
for cardbus will only find 4 spare.
if extend from top, at last will shrink back to really needed range...
-v4: fix checking with pci rescan. Found by Bjorn.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 101 ++++++++++++++++++++++++++++++---------------------
1 files changed, 60 insertions(+), 41 deletions(-)
@@ -809,10 +809,11 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,{structpci_bus*child;intis_cardbus=(dev->hdr_type==PCI_HEADER_TYPE_CARDBUS);-u32buses,i,j=0;+u32buses;u16bctl;u8primary,secondary,subordinate;intbroken=0;+structresource*parent_res=NULL;pci_read_config_dword(dev,PCI_PRIMARY_BUS,&buses);primary=buses&0xFF;
@@ -824,10 +825,16 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,/* Check if setup is sensible at all */if(!pass&&-(primary!=bus->number||secondary<=bus->number)){-dev_dbg(&dev->dev,"bus configuration invalid, reconfiguring\n");+(primary!=bus->number||secondary<=bus->number))broken=1;-}++/* more strict checking */+if(!pass&&!broken&&!dev->subordinate)+broken=pci_bridge_check_busn_broken(bus,dev,+secondary,subordinate);++if(broken)+dev_dbg(&dev->dev,"bus configuration invalid, reconfiguring\n");/* Disable MasterAbortMode during probing to avoid reportingofbuserrors(insomearchitectures)*/
@@ -860,6 +867,8 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,child->primary=primary;child->subordinate=subordinate;child->bridge_ctl=bctl;++pci_bus_insert_busn_res(child,secondary,subordinate);}cmax=pci_scan_child_bus(child);
@@ -872,6 +881,11 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,*Weneedtoassignanumbertothisbuswhichwealways*dointhesecondpass.*/+longshrink_size;+structresourcebusn_res;+intret=-ENOMEM;+intold_max=max;+if(!pass){if(pcibios_assign_all_busses()||broken)/* Temporarily disable forwarding of the
@@ -888,20 +902,44 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,/* Clear errors */pci_write_config_word(dev,PCI_STATUS,0xffff);-/* Prevent assigning a bus number that already exists.-*Thiscanhappenwhenabridgeishot-plugged,soin-*thiscaseweonlyre-scanthisbus.*/-child=pci_find_bus(pci_domain_nr(bus),max+1);-if(!child){-child=pci_add_new_bus(bus,dev,++max);-if(!child)-gotoout;+if(dev->subordinate){+/* We get here only for cardbus */+child=dev->subordinate;+if(!is_cardbus)+dev_warn(&dev->dev,+"rescan scaned bridge as broken one again ?");++gotoout;}+/*+*ForCardBusbridges,weleave4busnumbers+*ascardswithaPCI-to-PCIbridgecanbe+*insertedlater.+*otherjustallocate8bustoavoidwefallinto+*smallholeinthemiddle.+*/+ret=pci_bridge_probe_busn_res(bus,dev,&busn_res,+is_cardbus?(CARDBUS_RESERVE_BUSNR+1):8,+&parent_res);++if(ret!=0)+gotoout;++child=pci_add_new_bus(bus,dev,busn_bus_nr(busn_res.start));+if(!child)+gotoout;++child->subordinate=busn_bus_nr(busn_res.end);+pci_bus_insert_busn_res(child,busn_bus_nr(busn_res.start),+busn_bus_nr(busn_res.end));+buses=(buses&0xff000000)|((unsignedint)(child->primary)<<0)|((unsignedint)(child->secondary)<<8)|((unsignedint)(child->subordinate)<<16);+max=child->subordinate;+/**yenta.cforcesasecondarylatencytimerof176.*Copythatbehaviourhere.
@@ -932,43 +970,24 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,*therealvalueofmax.*/pci_fixup_parent_subordinate_busnr(child,max);+}else{-/*-*ForCardBusbridges,weleave4busnumbers-*ascardswithaPCI-to-PCIbridgecanbe-*insertedlater.-*/-for(i=0;i<CARDBUS_RESERVE_BUSNR;i++){-structpci_bus*parent=bus;-if(pci_find_bus(pci_domain_nr(bus),-max+i+1))-break;-while(parent->parent){-if((!pcibios_assign_all_busses())&&-(parent->subordinate>max)&&-(parent->subordinate<=max+i)){-j=1;-}-parent=parent->parent;-}-if(j){-/*-*Often,therearetwocardbusbridges-*--trytoleaveonevalidbusnumber-*foreachone.-*/-i/=2;-break;-}-}-max+=i;pci_fixup_parent_subordinate_busnr(child,max);}/**Setthesubordinatebusnumbertoitsrealvalue.*/+shrink_size=child->subordinate-max;child->subordinate=max;pci_write_config_byte(dev,PCI_SUBORDINATE_BUS,max);+pci_bus_update_busn_res_end(child,max);++/* shrink some back, if we extend top before */+if(!is_cardbus&&(shrink_size>0)&&parent_res)+pci_bus_shrink_top(bus,shrink_size,parent_res);++if(old_max>max)+max=old_max;}sprintf(child->name,
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:31
Now we can safely extend parent top and shrink them according iobusn_resource tree.
Don't need that any more.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 32 --------------------------------
1 files changed, 0 insertions(+), 32 deletions(-)
@@ -608,22 +608,6 @@ struct pci_bus *__ref pci_add_new_bus(struct pci_bus *parent, struct pci_dev *dereturnchild;}-staticvoidpci_fixup_parent_subordinate_busnr(structpci_bus*child,intmax)-{-structpci_bus*parent=child->parent;--/* Attempts to fix that up are really dangerous unless-we'regoingtore-assignallbusnumbers.*/-if(!pcibios_assign_all_busses())-return;--while(parent->parent&&parent->subordinate<max){-parent->subordinate=max;-pci_write_config_byte(parent->self,PCI_SUBORDINATE_BUS,max);-parent=parent->parent;-}-}-staticvoid__devinitpci_bus_update_top(structpci_bus*parent,longsize,structresource*parent_res){
@@ -956,23 +940,7 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,if(!is_cardbus){child->bridge_ctl=bctl;-/*-*Adjustsubordinatebusnrinparentbuses.-*Wedothisbeforescanningforchildrenbecause-*somedevicesmaynotbedetectedifthebios-*waslazy.-*/-pci_fixup_parent_subordinate_busnr(child,max);-/* Now we can scan all subordinate buses... */max=pci_scan_child_bus(child);-/*-*nowfixitupagainsincewehavefound-*therealvalueofmax.-*/-pci_fixup_parent_subordinate_busnr(child,max);--}else{-pci_fixup_parent_subordinate_busnr(child,max);}/**Setthesubordinatebusnumbertoitsrealvalue.
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:34
In extreme case: Two bridges are properly setup.
bridge A
bridge AA
bridge AB
bridge B
bridge BA
bridge BB
but AA, AB are not setup properly.
bridge A has small range, and bridge AB could need more, when do the
first pass0 for bridge A, it will do pass0 and pass1 for AA and AB,
during that process, it will extend range of A for AB blindly.
because bridge B is not registered yet.
that could overlap range that is used by bridge B.
Right way should be:
do pass0 for all good bridges at first.
So we could do pass0 for bridge B before pass1 for bridge AB.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 50 ++++++++++++++++++++++++++++++++++----------------
1 files changed, 34 insertions(+), 16 deletions(-)
@@ -779,6 +779,9 @@ static int __devinit pci_bridge_check_busn_broken(struct pci_bus *bus,returnbroken;}++staticunsignedint__devinit__pci_scan_child_bus(structpci_bus*bus,+intpass);/**Ifit'sabridge,configureitandscanthebusbehindit.*ForCardBusbridges,wedon'tscanbehindasthedeviceswill
@@ -830,11 +833,10 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,!is_cardbus&&!broken){unsignedintcmax;/*-*Busalreadyconfiguredbyfirmware,processitinthefirst-*passandjustnotetheconfiguration.+*Busalreadyconfiguredbyfirmware,stillprocessitintwo+*passesinextremecaseliketwoadjacedbridgeshavechildren+*bridgesthatarenotsetupproperly.*/-if(pass)-gotoout;/**Ifwealreadygottothisbusthroughadifferentbridge,
@@ -855,7 +857,7 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,pci_bus_insert_busn_res(child,secondary,subordinate);}-cmax=pci_scan_child_bus(child);+cmax=__pci_scan_child_bus(child,pass);if(cmax>max)max=cmax;if(child->subordinate>max)
@@ -1651,12 +1653,13 @@ void pcie_bus_configure_settings(struct pci_bus *bus, u8 mpss)}EXPORT_SYMBOL_GPL(pcie_bus_configure_settings);-unsignedint__devinitpci_scan_child_bus(structpci_bus*bus)+staticunsignedint__devinit__pci_scan_child_bus(structpci_bus*bus,+intpass){-unsignedintdevfn,pass,max=bus->secondary;+unsignedintdevfn,max=bus->secondary;structpci_dev*dev;-dev_dbg(&bus->dev,"scanning bus\n");+dev_dbg(&bus->dev,"scanning bus pass %d\n",pass);/* Go find them, Rover! */for(devfn=0;devfn<0x100;devfn+=8)
@@ -1670,18 +1673,16 @@ unsigned int __devinit pci_scan_child_bus(struct pci_bus *bus)*allPCI-to-PCIbridgesonthisbus.*/if(!bus->is_added){-dev_dbg(&bus->dev,"fixups for bus\n");+dev_dbg(&bus->dev,"fixups for bus pass %d\n",pass);pcibios_fixup_bus(bus);if(pci_is_root_bus(bus))bus->is_added=1;}-for(pass=0;pass<2;pass++)-list_for_each_entry(dev,&bus->devices,bus_list){-if(dev->hdr_type==PCI_HEADER_TYPE_BRIDGE||-dev->hdr_type==PCI_HEADER_TYPE_CARDBUS)-max=pci_scan_bridge(bus,dev,max,pass);-}+list_for_each_entry(dev,&bus->devices,bus_list)+if(dev->hdr_type==PCI_HEADER_TYPE_BRIDGE||+dev->hdr_type==PCI_HEADER_TYPE_CARDBUS)+max=pci_scan_bridge(bus,dev,max,pass);/**We'vescannedthebusandsoweknowallaboutwhat'son
@@ -1690,7 +1691,24 @@ unsigned int __devinit pci_scan_child_bus(struct pci_bus *bus)**Returnhowfarwe'vegotfindingsub-buses.*/-dev_dbg(&bus->dev,"bus scan returning with max=%02x\n",max);+dev_dbg(&bus->dev,"bus scan returning with max=%02x pass %d\n",+max,pass);++returnmax;+}++unsignedint__devinitpci_scan_child_bus(structpci_bus*bus)+{+intpass;+unsignedintmax=0,tmp;++for(pass=0;pass<2;pass++){+tmp=__pci_scan_child_bus(bus,pass);++if(tmp>max)+max=tmp;+}+returnmax;}
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:38
User could use setpci change bridge's bus register. that could make value of register
and struct is out of sync.
User later will use rescan to see the devices is moving.
In the rescaning, we need to double check the range and remove the old struct at first.
to make thing working user may need have script to remove children devices under bridge
at first, and then use setpci update bus register and then rescan.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 40 ++++++++++++++++++++++++++++++++++++++++
1 files changed, 40 insertions(+), 0 deletions(-)
@@ -815,6 +815,46 @@ int __devinit pci_scan_bridge(struct pci_bus *bus, struct pci_dev *dev, int max,(primary!=bus->number||secondary<=bus->number))broken=1;+if(!pass&&dev->subordinate){+child=dev->subordinate;+/*+*Usercouldchangebusregisterinbridgemanuallywith+*setpciandrescan.Sodoublecheckthesetting,andremove+*oldstructs.Don'tsetbrokenyet,letfollowingcheck+*toseeifthenewsettinggood.+*/+if(primary!=child->primary||+secondary!=child->secondary||+subordinate!=child->subordinate){+dev_info(&dev->dev,+"someone changed bus register from pri:%02x, sec:%02x, sub:%02x to pri:%02x, sec:%02x, sub:%02x\n",+child->primary,child->secondary,child->subordinate,+primary,secondary,subordinate);+if(!list_empty(&dev->subordinate->devices)){+u32old_buses;++dev_info(&dev->dev,+"but children devices are not removed manually before that.\n");+/*+*Trybesttoremoveleftchildrendevices+*butweneedtosetbusregisterback,otherwise+*Wecannotaccesschildrendeviceandstopthem+*/+old_buses=(buses&0xff000000)+|((unsignedint)(child->primary)<<0)+|((unsignedint)(child->secondary)<<8)+|((unsignedint)(child->subordinate)<<16);+pci_write_config_dword(dev,PCI_PRIMARY_BUS,+old_buses);+pci_remove_behind_bridge(dev);+pci_write_config_dword(dev,PCI_PRIMARY_BUS,+buses);+}+pci_remove_bus(dev->subordinate);+dev->subordinate=NULL;+}+}+/* more strict checking */if(!pass&&!broken&&!dev->subordinate)broken=pci_bridge_check_busn_broken(bus,dev,
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:44
We want to create rescan in sys only for pci bridge instead of all pci dev.
We could use attribute_groups/is_visible method to do that.
Now pci dev does not use device type yet. So add pci_dev_type to take
attr_groups with it.
Add pci_dev_bridge_attrs_are_visible() to control attr_bridge_group only create
attr for bridge.
This is the framework related change, later could attr to bridge_attr_group,
to make those attr only show up on pci bridge device.
Also We could add more attr groups with is_visible to reduce messness in
pci_create_sysfs_dev_files. ( at least for boot_vga one )
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/pci-sysfs.c | 30 ++++++++++++++++++++++++++++++
drivers/pci/pci.h | 1 +
drivers/pci/probe.c | 1 +
3 files changed, 32 insertions(+), 0 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:49
Should only use it with bridge instead of bus.
it could get new subordinate. so can not use it with reguar bus.
in that case, use may need to make all children devices get removed already.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/pci-sysfs.c | 7 ++-----
1 files changed, 2 insertions(+), 5 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:00:51
Now pci busn allocation code is there, and it will preallocate bus number and it
will make sure parent buses subordinate is right.
So remove workaround here.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
Cc: Rusty Russell <redacted>
Cc: Mauro Carvalho Chehab <redacted>
Cc: Dominik Brodowski <linux@dominikbrodowski.net>
Cc: linux-pcmcia@lists.infradead.org
---
drivers/pcmcia/yenta_socket.c | 75 -----------------------------------------
1 files changed, 0 insertions(+), 75 deletions(-)
@@ -1064,79 +1064,6 @@ static void yenta_config_init(struct yenta_socket *socket)config_writew(socket,CB_BRIDGE_CONTROL,bridge);}-/**-*yenta_fixup_parent_bridge-Fixsubordinatebus#oftheparentbridge-*@cardbus_bridge:ThePCIbuswhichtheCardBusbridgebridgesto-*-*ChecksifdevicesonthebuswhichtheCardBusbridgebridgestowouldbe-*invisibleduringPCIscansbecauseofamisconfiguredsubordinatenumber-*oftheparentbrige-someBIOSesseemtobetoolazytosetitright.-*Doesthefixupcarefullybycheckinghowfaritcangowithoutconflicts.-*Seehttp://bugzilla.kernel.org/show_bug.cgi?id=2944 for more information.-*/-staticvoidyenta_fixup_parent_bridge(structpci_bus*cardbus_bridge)-{-structlist_head*tmp;-unsignedcharupper_limit;-/*-*Weonlycheckandfixtheparentbridge:Allsystemswhichneed-*thisfixupthathavebeenreviewedarelaptopsandtheonlybridge-*whichneededfixingwastheparentbridgeoftheCardBusbridge:-*/-structpci_bus*bridge_to_fix=cardbus_bridge->parent;--/* Check bus numbers are already set up correctly: */-if(bridge_to_fix->subordinate>=cardbus_bridge->subordinate)-return;/* The subordinate number is ok, nothing to do */--if(!bridge_to_fix->parent)-return;/* Root bridges are ok */--/* stay within the limits of the bus range of the parent: */-upper_limit=bridge_to_fix->parent->subordinate;--/* check the bus ranges of all silbling bridges to prevent overlap */-list_for_each(tmp,&bridge_to_fix->parent->children){-structpci_bus*silbling=pci_bus_b(tmp);-/*-*Ifthesilblinghasahighersecondarybusnumber-*andit'ssecondaryisequalorsmallerthanour-*currentupperlimit,setthenewupperlimitto-*thebusnumberbelowthesilbling'srange:-*/-if(silbling->secondary>bridge_to_fix->subordinate-&&silbling->secondary<=upper_limit)-upper_limit=silbling->secondary-1;-}--/* Show that the wanted subordinate number is not possible: */-if(cardbus_bridge->subordinate>upper_limit)-dev_printk(KERN_WARNING,&cardbus_bridge->dev,-"Upper limit for fixing this "-"bridge's parent bridge: #%02x\n",upper_limit);--/* If we have room to increase the bridge's subordinate number, */-if(bridge_to_fix->subordinate<upper_limit){--/* use the highest number of the hidden bus, within limits */-unsignedcharsubordinate_to_assign=-min(cardbus_bridge->subordinate,upper_limit);--dev_printk(KERN_INFO,&bridge_to_fix->dev,-"Raising subordinate bus# of parent "-"bus (#%02x) from #%02x to #%02x\n",-bridge_to_fix->number,-bridge_to_fix->subordinate,subordinate_to_assign);--/* Save the new subordinate in the bus struct of the bridge */-bridge_to_fix->subordinate=subordinate_to_assign;--/* and update the PCI config space with the new subordinate */-pci_write_config_byte(bridge_to_fix->self,-PCI_SUBORDINATE_BUS,bridge_to_fix->subordinate);-}-}-/**Initializeacardbuscontroller.Makesurewehaveausable*interrupt,andthatwecanmapthecardbusarea.Fillinthe
@@ -1257,8 +1184,6 @@ static int __devinit yenta_probe(struct pci_dev *dev, const struct pci_device_iddev_printk(KERN_INFO,&dev->dev,"Socket status: %08x\n",cb_readl(socket,CB_SOCKET_STATE));-yenta_fixup_parent_bridge(dev->subordinate);-/* Register it with the pcmcia layer.. */ret=pcmcia_register_socket(&socket->socket);if(ret==0){
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:01:33
So after remove all children and then using setpci change bus register and rescan
bridge could use new set bus number.
Otherwise need to rescan parent bus, it would have too much overhead.
also need to use pci_bus_add_single_device to make sure new change bus have directory
/sys/../.../pci_bus.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 10 ++++++----
1 files changed, 6 insertions(+), 4 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:02:00
will use it to pci_bus_bridge_scan_resize(0 to make bridge will
have pci_bus directory created.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/bus.c | 40 ++++++++++++++++++++++++++++++++++++++++
include/linux/pci.h | 1 +
2 files changed, 41 insertions(+), 0 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:02:21
Current code will create rescan for every pci device under parent bus.
that is not right. the device is already there, there is no reason to rescan it.
We could have rescan for pci bridges. less confusing.
Need to move rescan attr to pci dev bridge attribute group.
And We should rescan bridge's secondary bus instead of primary bus.
-v3: Use device_type for pci dev.
-v4: Seperate pci device type change out
-v5: add rescan_bridge for bridge type, and still keep the old rescan.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
Documentation/ABI/testing/sysfs-bus-pci | 10 ++++++++++
drivers/pci/pci-sysfs.c | 24 ++++++++++++++++++++++++
2 files changed, 34 insertions(+), 0 deletions(-)
@@ -111,6 +111,16 @@ Description: from this part of the device tree. Depends on CONFIG_HOTPLUG.+What: /sys/bus/pci/devices/.../rescan_bridge+Date: February 2012+Contact: Linux PCI developers <linux-pci@vger.kernel.org>+Description:+ Writing a non-zero value to this attribute will+ force a rescan of the bridge and all child buses, and+ re-discover devices removed earlier from this part of+ the device tree.+ Depends on CONFIG_HOTPLUG.+ What: /sys/bus/pci/devices/.../reset Date: July 2009 Contact: Michael S. Tsirkin <mst@redhat.com>
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:02:43
Found hotplug adding one EM with bridge fail, bios only leave one bus range
for slot.
[ 1169.621444] pciehp: No bus number available for hot-added bridge 0000:55:00.0
[ 1169.633277] pcieport 0000:40:03.0: PCI bridge to [bus 55-55]
With busn_res tracking and allocating, we don't need that checking anymore.
Parent bridges' bus number will be extended safely.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/hotplug/pciehp_pci.c | 12 +-----------
1 files changed, 1 insertions(+), 11 deletions(-)
@@ -37,18 +37,8 @@staticint__refpciehp_add_bridge(structpci_dev*dev){structpci_bus*parent=dev->bus;-intpass,busnr,start=parent->secondary;-intend=parent->subordinate;+intpass,busnr=parent->secondary;-for(busnr=start;busnr<=end;busnr++){-if(!pci_find_bus(pci_domain_nr(parent),busnr))-break;-}-if(busnr-->end){-err("No bus number available for hot-added bridge %s\n",-pci_name(dev));-return-1;-}for(pass=0;pass<2;pass++)busnr=pci_scan_bridge(parent,dev,busnr,pass);if(!dev->subordinate)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:04:40
So could use %pR for busn_res with domain nr in start/end
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
Cc: Andrew Morton <akpm@linux-foundation.org>
---
lib/vsprintf.c | 29 +++++++++++++++++++++++++----
1 files changed, 25 insertions(+), 4 deletions(-)
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-05 07:16:36
Try to allocate from parent bus busn_res. if can not find any big enough, will try
to extend parent bus top. even the extending is through allocating, after allocating
will pad the range to parent buses top.
When extending happens, We will record the parent_res, so could use it as stopper
for really extend/shrink top later.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 110 +++++++++++++++++++++++++++++++++++++++++++++++++++
1 files changed, 110 insertions(+), 0 deletions(-)
@@ -660,6 +660,116 @@ static void __devinit pci_bus_shrink_top(struct pci_bus *parent,pci_bus_update_top(parent,-size,parent_res);}+staticresource_size_t__devinitfind_res_top_free_size(structresource*res)+{+resource_size_tn_size;+structresourcetmp_res;++/*+*findoutnumberbelowres->end,thatwecanuseatfirst+*res->startcannotbeused.+*/+n_size=resource_size(res)-1;+memset(&tmp_res,0,sizeof(structresource));+while(n_size>0){+intret;++ret=allocate_resource(res,&tmp_res,n_size,+res->end-n_size+1,res->end,+1,NULL,NULL);+if(ret==0){+release_resource(&tmp_res);+break;+}+n_size--;+}++returnn_size;+}++staticint__devinitpci_bridge_probe_busn_res(structpci_bus*bus,+structpci_dev*dev,structresource*busn_res,+resource_size_tneeded_size,structresource**p)+{+intret=-ENOMEM;+resource_size_tn_size;+structpci_bus*parent;+structresource*parent_res=NULL;+resource_size_ttmp=bus->busn_res.end+1;+intfree_sz=-1;++again:+/*+*findbigestrangeinbus->busn_resthatwecanuseinthemiddle+*andwecannotusebus->busn_res.start.+*/+n_size=resource_size(&bus->busn_res)-1;+memset(busn_res,0,sizeof(structresource));+dev_printk(KERN_DEBUG,&dev->dev,+"find free busn in busn_res: %pR\n",&bus->busn_res);+while(n_size>=needed_size){+ret=allocate_resource(&bus->busn_res,busn_res,n_size,+bus->busn_res.start+1,bus->busn_res.end,+1,NULL,NULL);+if(ret==0){+/* found one, prepare to return */+release_resource(busn_res);++returnret;+}+n_size--;+}++/* try extend the top of parent bus, find free under top af first */+if(free_sz<0){+free_sz=find_res_top_free_size(&bus->busn_res);+dev_printk(KERN_DEBUG,&dev->dev,+"found free busn %d in busn_res: %pR top\n",+free_sz,&bus->busn_res);+}+n_size=free_sz;++/* can not extend cross domain boundary */+if((0xff-busn_bus_nr(bus->busn_res.end))<(needed_size-n_size))+gotoreduce_needed_size;++/* find exteded range */+memset(busn_res,0,sizeof(structresource));+parent=bus->parent;+while(parent){+ret=allocate_resource(&parent->busn_res,busn_res,+needed_size-n_size,+tmp,tmp+needed_size-n_size-1,+1,NULL,NULL);+if(ret==0)+break;+parent=parent->parent;+}++reduce_needed_size:+if(ret!=0){+needed_size--;+if(!needed_size)+returnret;++gotoagain;+}++/* save parent_res, we need it as stopper later */+parent_res=busn_res->parent;++/* prepare busn_res for return */+release_resource(busn_res);+busn_res->start-=n_size;++/* extend parent bus top*/+pci_bus_extend_top(bus,needed_size-n_size,parent_res);++*p=parent_res;++returnret;+}+/**Ifit'sabridge,configureitandscanthebusbehindit.*ForCardBusbridges,wedon'tscanbehindasthedeviceswill
On Sat, Feb 4, 2012 at 10:57 PM, Yinghai Lu [off-list ref] wrote:
quoted hunk
also add busn_res into struct pci_bus.
will use them to have bus number resource tree.
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
Cc: Andrew Morton <akpm@linux-foundation.org>
---
include/linux/ioport.h | 1 +
include/linux/pci.h | 1 +
kernel/resource.c | 8 ++++++++
3 files changed, 10 insertions(+), 0 deletions(-)
struct list_head slots; /* list of slots on this bus */
struct resource *resource[PCI_BRIDGE_RESOURCE_NUM];
struct list_head resources; /* address space routed to this bus */
+ struct resource busn_res; /* track registered bus num range */
struct pci_ops *ops; /* configuration access functions */
void *sysdata; /* hook for sys-specific extension */
I'm not sure this should be global. iomem_resource and
ioport_resource *are* really global, because they refer to processor
address space that is the same for everybody. But PCI bus numbers are
specific to PCI. Some machines don't have PCI at all, and there are
different bus architectures to which this doesn't apply.
The 0-0xffffff range is misleading because it includes both the domain
and the bus number, and it's meaningless to allocate ranges that cross
domain boundaries. For example, [bus 0x0000f0-0x000120] includes bus
numbers from domain 0000 and domain 0001, which doesn't make any sense
because a bus can only be in one domain.
I think it would make more sense to keep this bus number resource in a
per-host bridge structure. Then we wouldn't need to include the
domain number at all because the host bridge determines the domain.
+EXPORT_SYMBOL(iobusn_resource);
+
/* constraints to be met while allocating resources */
struct resource_constraint {
resource_size_t min, max, align;
--
1.7.7
I'm not sure this should be global. iomem_resource and
ioport_resource *are* really global, because they refer to processor
address space that is the same for everybody. But PCI bus numbers are
specific to PCI. Some machines don't have PCI at all, and there are
different bus architectures to which this doesn't apply.
that does not hurt them.
The 0-0xffffff range is misleading because it includes both the domain
and the bus number, and it's meaningless to allocate ranges that cross
domain boundaries. For example, [bus 0x0000f0-0x000120] includes bus
numbers from domain 0000 and domain 0001, which doesn't make any sense
because a bus can only be in one domain.
allocation code will make sure it will be cross the boundary for domain.
I think it would make more sense to keep this bus number resource in a
per-host bridge structure. Then we wouldn't need to include the
domain number at all because the host bridge determines the domain.
not sure. insert the all busn_res of all peer root buses into one
global iobusn_resource
looks more simple.
Yinghai
On Sat, Feb 4, 2012 at 10:57 PM, Yinghai Lu [off-list ref] wrote:
quoted hunk
will use them insert/update busn res in pci_bus
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 42 ++++++++++++++++++++++++++++++++++++++++++
include/linux/pci.h | 3 +++
2 files changed, 45 insertions(+), 0 deletions(-)
I think this design is a mistake. Here's what you're doing:
- initialize struct resource (keys are "start" and "end")
- insert into tree (placed in tree by kernel/resource.c based on
"start" and "end")
- update "end"
You "know" in this case that the update is safe because the caller has
validated "bus_max." But that still breaks the kernel/resource.c
encapsulation. If we change the kernel/resource.c implementation,
this code might break.
I think it would be better to remove the bus resource from the tree,
change its "end," then re-insert it.
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-06 20:45:18
On Mon, Feb 6, 2012 at 10:59 AM, Bjorn Helgaas [off-list ref] wrote:
On Sat, Feb 4, 2012 at 10:57 PM, Yinghai Lu [off-list ref] wrote:
quoted
will use them insert/update busn res in pci_bus
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 42 ++++++++++++++++++++++++++++++++++++++++++
include/linux/pci.h | 3 +++
2 files changed, 45 insertions(+), 0 deletions(-)
I think this design is a mistake. Here's what you're doing:
- initialize struct resource (keys are "start" and "end")
- insert into tree (placed in tree by kernel/resource.c based on
"start" and "end")
- update "end"
You "know" in this case that the update is safe because the caller has
validated "bus_max." But that still breaks the kernel/resource.c
encapsulation. If we change the kernel/resource.c implementation,
this code might break.
the point is: I only want to reuse allocate_resource() to get right position.
and the code does not depends to kernel/resource.c much.
I think it would be better to remove the bus resource from the tree,
change its "end," then re-insert it.
how about parent buses that have extended top?
Yinghai
The only architecture-specific thing here is discovering the range of
bus numbers below a host bridge. The architecture should not have to
mess around with pci_bus_update_busn_res_end() like this. It should
be able to say "here's my bus number range" (and of course the PCI
core can default to 0-255 if the arch doesn't supply a range) and the
core should take care of the rest.
Bjorn
void pci_bus_insert_busn_res(struct pci_bus *b, int bus, int busmax);
void pci_bus_update_busn_res_end(struct pci_bus *b, int busmax);
void pci_bus_release_busn_res(struct pci_bus *b);
+struct pci_bus * __devinit pci_scan_root_bus_max(struct device *parent, int bus,
+ int busmax, struct pci_ops *ops,
+ void *sysdata,
+ struct list_head *resources);
struct pci_bus * __devinit pci_scan_root_bus(struct device *parent, int bus,
struct pci_ops *ops, void *sysdata,
struct list_head *resources);
Now we have both pci_bus_insert_busn_res() and
pci_scan_root_bus_max(). Why do we need both? I think it's too much
of a burden on architectures to expect them to understand both.
Maybe some sample pseudo-code using both interfaces would help me
understand what you expect a typical architecture to do.
Bjorn
void pci_bus_insert_busn_res(struct pci_bus *b, int bus, int busmax);
void pci_bus_update_busn_res_end(struct pci_bus *b, int busmax);
void pci_bus_release_busn_res(struct pci_bus *b);
+struct pci_bus * __devinit pci_scan_root_bus_max(struct device *parent, int bus,
+ int busmax, struct pci_ops *ops,
+ void *sysdata,
+ struct list_head *resources);
struct pci_bus * __devinit pci_scan_root_bus(struct device *parent, int bus,
struct pci_ops *ops, void *sysdata,
struct list_head *resources);
Now we have both pci_bus_insert_busn_res() and
pci_scan_root_bus_max(). Why do we need both? I think it's too much
of a burden on architectures to expect them to understand both.
Maybe some sample pseudo-code using both interfaces would help me
understand what you expect a typical architecture to do.
We need to call pci_bus_insert_busn_res for root bus:
1. after that bus is allocated (...) : that busn_res is in the bus struct.
2. before pci_scan_child_bus
Yinghai
The only architecture-specific thing here is discovering the range of
bus numbers below a host bridge. The architecture should not have to
mess around with pci_bus_update_busn_res_end() like this. It should
be able to say "here's my bus number range" (and of course the PCI
core can default to 0-255 if the arch doesn't supply a range) and the
core should take care of the rest.
during the pci_scan_child_bus, child bus busn_res will be inserted
under parent bus busn_res.
So need to make sure parent busn_res.end is bigger enough.
Yinghai
From: Benjamin Herrenschmidt <benh@kernel.crashing.org> Date: 2012-02-08 22:03:15
On Wed, 2012-02-08 at 07:58 -0800, Bjorn Helgaas wrote:
The only architecture-specific thing here is discovering the range of
bus numbers below a host bridge. The architecture should not have to
mess around with pci_bus_update_busn_res_end() like this. It should
be able to say "here's my bus number range" (and of course the PCI
core can default to 0-255 if the arch doesn't supply a range) and the
core should take care of the rest.
So it's a bit messy in here because we deal with several things.
What the firmware gives us is the range it assigned, but that isn't
necessarily the HW limits (almost never is in fact).
In some cases we honor it, for example when in "probe only" mode where
we prevent any reassigning, and in some case, we ignore it and let the
PCI core renumber things (typically because the FW "forgot" to set aside
bus numbers for a cardbus slot for example, that sort of things).
So it's a bit of a tricky situation.
Off the top of my head, I'm pretty sure that most if not all of our PCI
host bridges simply support a full 0...255 range and there is no sharing
between bridges like on x86, they are just different domains.
But I can't vouch 100% for some of the oddball cases like Pegasos or
some freescale gear.
Cheers,
Ben.
On Wed, Feb 8, 2012 at 2:02 PM, Benjamin Herrenschmidt
[off-list ref] wrote:
On Wed, 2012-02-08 at 07:58 -0800, Bjorn Helgaas wrote:
quoted
The only architecture-specific thing here is discovering the range of
bus numbers below a host bridge. The architecture should not have to
mess around with pci_bus_update_busn_res_end() like this. It should
be able to say "here's my bus number range" (and of course the PCI
core can default to 0-255 if the arch doesn't supply a range) and the
core should take care of the rest.
So it's a bit messy in here because we deal with several things.
What the firmware gives us is the range it assigned, but that isn't
necessarily the HW limits (almost never is in fact).
In some cases we honor it, for example when in "probe only" mode where
we prevent any reassigning, and in some case, we ignore it and let the
PCI core renumber things (typically because the FW "forgot" to set aside
bus numbers for a cardbus slot for example, that sort of things).
So it's a bit of a tricky situation.
Off the top of my head, I'm pretty sure that most if not all of our PCI
host bridges simply support a full 0...255 range and there is no sharing
between bridges like on x86, they are just different domains.
My point is that the interface between the arch and the PCI core
should be simply the arch telling the core "this is the range of bus
numbers you can use." If the firmware doesn't give you the HW limits,
that's the arch's problem. If you want to assume 0..255 are
available, again, that's the arch's decision.
But the answer to the question "what bus numbers are available to me"
depends only on the host bridge HW configuration. It does not depend
on what pci_scan_child_bus() found. Therefore, I think we can come up
with a design where pci_bus_update_busn_res_end() is unnecessary.
Bjorn
From: Benjamin Herrenschmidt <benh@kernel.crashing.org> Date: 2012-02-09 21:36:23
On Thu, 2012-02-09 at 11:24 -0800, Bjorn Helgaas wrote:
My point is that the interface between the arch and the PCI core
should be simply the arch telling the core "this is the range of bus
numbers you can use." If the firmware doesn't give you the HW limits,
that's the arch's problem. If you want to assume 0..255 are
available, again, that's the arch's decision.
But the answer to the question "what bus numbers are available to me"
depends only on the host bridge HW configuration. It does not depend
on what pci_scan_child_bus() found. Therefore, I think we can come up
with a design where pci_bus_update_busn_res_end() is unnecessary.
In an ideal world yes. In a world where there are reverse engineered
platforms on which we aren't 100% sure how thing actually work under the
hood and have the code just adapt on "what's there" (and try to fix it
up -sometimes-), thinks can get a bit murky :-)
But yes, I see your point. As for what is the "correct" setting that
needs to be done so that the patch doesn't end up a regression for us,
I'll have to dig into some ancient HW to dbl check a few things. I hope
0...255 will just work but I can't guarantee it.
What I'll probably do is constraint the core to the values in
hose->min/max, and update selected platforms to put 0..255 in there when
I know for sure they can cope.
Cheers,
Ben.
I'm not sure this should be global. iomem_resource and
ioport_resource *are* really global, because they refer to processor
address space that is the same for everybody. But PCI bus numbers are
specific to PCI. Some machines don't have PCI at all, and there are
different bus architectures to which this doesn't apply.
that does not hurt them.
Yes but it's superfluous and confusing if you're porting to a new arch
or looking at changes in generic code that may affect you.
quoted
The 0-0xffffff range is misleading because it includes both the domain
and the bus number, and it's meaningless to allocate ranges that cross
domain boundaries. For example, [bus 0x0000f0-0x000120] includes bus
numbers from domain 0000 and domain 0001, which doesn't make any sense
because a bus can only be in one domain.
allocation code will make sure it will be cross the boundary for domain.
But that means everyone reading it will do a double take, have to dig
into the implementation, and only then say "ah yeah ok it looks
correct" rather than it being obvious from the fact that the resource
is tracked on a per-domain basis.
quoted
I think it would make more sense to keep this bus number resource in a
per-host bridge structure. Then we wouldn't need to include the
domain number at all because the host bridge determines the domain.
not sure. insert the all busn_res of all peer root buses into one
global iobusn_resource
looks more simple.
In what sense? Simpler in the sense of your current implementation,
but not simpler at all to someone just reading the code...
--
Jesse Barnes, Intel Open Source Technology Center
I'm not sure this should be global. iomem_resource and
ioport_resource *are* really global, because they refer to processor
address space that is the same for everybody. But PCI bus numbers are
specific to PCI. Some machines don't have PCI at all, and there are
different bus architectures to which this doesn't apply.
that does not hurt them.
Yes but it's superfluous and confusing if you're porting to a new arch
or looking at changes in generic code that may affect you.
quoted
quoted
The 0-0xffffff range is misleading because it includes both the domain
and the bus number, and it's meaningless to allocate ranges that cross
domain boundaries. For example, [bus 0x0000f0-0x000120] includes bus
numbers from domain 0000 and domain 0001, which doesn't make any sense
because a bus can only be in one domain.
allocation code will make sure it will be cross the boundary for domain.
But that means everyone reading it will do a double take, have to dig
into the implementation, and only then say "ah yeah ok it looks
correct" rather than it being obvious from the fact that the resource
is tracked on a per-domain basis.
quoted
quoted
I think it would make more sense to keep this bus number resource in a
per-host bridge structure. Then we wouldn't need to include the
domain number at all because the host bridge determines the domain.
not sure. insert the all busn_res of all peer root buses into one
global iobusn_resource
looks more simple.
In what sense? Simpler in the sense of your current implementation,
but not simpler at all to someone just reading the code...
ok, will check if i could drop iobusn_resource.
Yinghai
On Mon, Feb 6, 2012 at 12:45 PM, Yinghai Lu [off-list ref] wrote:
On Mon, Feb 6, 2012 at 10:59 AM, Bjorn Helgaas [off-list ref] wrote:
quoted
On Sat, Feb 4, 2012 at 10:57 PM, Yinghai Lu [off-list ref] wrote:
quoted
will use them insert/update busn res in pci_bus
Signed-off-by: Yinghai Lu <yinghai@kernel.org>
---
drivers/pci/probe.c | 42 ++++++++++++++++++++++++++++++++++++++++++
include/linux/pci.h | 3 +++
2 files changed, 45 insertions(+), 0 deletions(-)
I think this design is a mistake. Here's what you're doing:
- initialize struct resource (keys are "start" and "end")
- insert into tree (placed in tree by kernel/resource.c based on
"start" and "end")
- update "end"
You "know" in this case that the update is safe because the caller has
validated "bus_max." But that still breaks the kernel/resource.c
encapsulation. If we change the kernel/resource.c implementation,
this code might break.
the point is: I only want to reuse allocate_resource() to get right position.
and the code does not depends to kernel/resource.c much.
quoted
I think it would be better to remove the bus resource from the tree,
change its "end," then re-insert it.
how about parent buses that have extended top?
I don't understand your question. I assume you mean there's a case
where remove/update/reinsert doesn't work, but I don't see why that
would be a problem. Can you show an example?
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-13 00:03:09
On Sun, Feb 12, 2012 at 3:51 PM, Bjorn Helgaas [off-list ref] wrote:
quoted
quoted
I think it would be better to remove the bus resource from the tree,
change its "end," then re-insert it.
how about parent buses that have extended top?
I don't understand your question. I assume you mean there's a case
where remove/update/reinsert doesn't work, but I don't see why that
would be a problem. Can you show an example?
I mean parent busn_res already had several level's children busn_res.
and every level may have some siblings.
before remove will need to record those resources, to later to put them back.
that just increase not necessary complexity. because we already know
those resource could be extended safely.
Yinghai
On Sun, Feb 12, 2012 at 4:03 PM, Yinghai Lu [off-list ref] wrote:
On Sun, Feb 12, 2012 at 3:51 PM, Bjorn Helgaas [off-list ref] wrote:
quoted
quoted
quoted
I think it would be better to remove the bus resource from the tree,
change its "end," then re-insert it.
how about parent buses that have extended top?
I don't understand your question. I assume you mean there's a case
where remove/update/reinsert doesn't work, but I don't see why that
would be a problem. Can you show an example?
I mean parent busn_res already had several level's children busn_res.
and every level may have some siblings.
before remove will need to record those resources, to later to put them back.
that just increase not necessary complexity. because we already know
those resource could be extended safely.
You're doing surgery on the middle of a relatively complicated data
structure. Now readers of the code have to trust that not only does
kernel/resource.c work correctly, but they also have to examine this
PCI code to make sure that these alterations are safe. I know this is
all crystal-clear in your mind, and no doubt it is correct right now,
but I don't think it is a reader-friendly approach.
But I don't expect to convince you, so I'll stop trying :)
Bjorn
void pci_bus_insert_busn_res(struct pci_bus *b, int bus, int busmax);
void pci_bus_update_busn_res_end(struct pci_bus *b, int busmax);
void pci_bus_release_busn_res(struct pci_bus *b);
+struct pci_bus * __devinit pci_scan_root_bus_max(struct device *parent, int bus,
+ int busmax, struct pci_ops *ops,
+ void *sysdata,
+ struct list_head *resources);
struct pci_bus * __devinit pci_scan_root_bus(struct device *parent, int bus,
struct pci_ops *ops, void *sysdata,
struct list_head *resources);
Now we have both pci_bus_insert_busn_res() and
pci_scan_root_bus_max(). Why do we need both? I think it's too much
of a burden on architectures to expect them to understand both.
Maybe some sample pseudo-code using both interfaces would help me
understand what you expect a typical architecture to do.
We need to call pci_bus_insert_busn_res for root bus:
1. after that bus is allocated (...) : that busn_res is in the bus struct.
2. before pci_scan_child_bus
I agree with Bjorn; at the very least this has a really bad name. How
are people supposed to know which function to call? If it's just for
the primary scan at init, then it should be named as such. But ideally
it would be inside the existing scan_root_bus...
--
Jesse Barnes, Intel Open Source Technology Center
On Fri, 10 Feb 2012 08:35:58 +1100
Benjamin Herrenschmidt [off-list ref] wrote:
On Thu, 2012-02-09 at 11:24 -0800, Bjorn Helgaas wrote:
quoted
My point is that the interface between the arch and the PCI core
should be simply the arch telling the core "this is the range of bus
numbers you can use." If the firmware doesn't give you the HW limits,
that's the arch's problem. If you want to assume 0..255 are
available, again, that's the arch's decision.
But the answer to the question "what bus numbers are available to me"
depends only on the host bridge HW configuration. It does not depend
on what pci_scan_child_bus() found. Therefore, I think we can come up
with a design where pci_bus_update_busn_res_end() is unnecessary.
In an ideal world yes. In a world where there are reverse engineered
platforms on which we aren't 100% sure how thing actually work under the
hood and have the code just adapt on "what's there" (and try to fix it
up -sometimes-), thinks can get a bit murky :-)
But yes, I see your point. As for what is the "correct" setting that
needs to be done so that the patch doesn't end up a regression for us,
I'll have to dig into some ancient HW to dbl check a few things. I hope
0...255 will just work but I can't guarantee it.
What I'll probably do is constraint the core to the values in
hose->min/max, and update selected platforms to put 0..255 in there when
I know for sure they can cope.
But I think the point is, can't we intiialize the busn resource after
the first & last bus numbers have been determined? E.g. rather than
Yinghai's current:
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
+
/* Get probe mode and perform scan */
mode = PCI_PROBE_NORMAL;
if (node && ppc_md.pci_probe_mode)
if (mode == PCI_PROBE_NORMAL)
hose->last_busno = bus->subordinate = pci_scan_child_bus(bus);
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
since we should have the final bus range by then? Setting the end to
255 and then changing it again doesn't make sense; and definitely makes
the code hard to follow.
--
Jesse Barnes, Intel Open Source Technology Center
On Thu, Feb 23, 2012 at 12:25 PM, Jesse Barnes [off-list ref] wrote:
quoted hunk
On Fri, 10 Feb 2012 08:35:58 +1100
Benjamin Herrenschmidt [off-list ref] wrote:
quoted
On Thu, 2012-02-09 at 11:24 -0800, Bjorn Helgaas wrote:
quoted
My point is that the interface between the arch and the PCI core
should be simply the arch telling the core "this is the range of bus
numbers you can use." If the firmware doesn't give you the HW limits,
that's the arch's problem. If you want to assume 0..255 are
available, again, that's the arch's decision.
But the answer to the question "what bus numbers are available to me"
depends only on the host bridge HW configuration. It does not depend
on what pci_scan_child_bus() found. Therefore, I think we can come up
with a design where pci_bus_update_busn_res_end() is unnecessary.
In an ideal world yes. In a world where there are reverse engineered
platforms on which we aren't 100% sure how thing actually work under the
hood and have the code just adapt on "what's there" (and try to fix it
up -sometimes-), thinks can get a bit murky :-)
But yes, I see your point. As for what is the "correct" setting that
needs to be done so that the patch doesn't end up a regression for us,
I'll have to dig into some ancient HW to dbl check a few things. I hope
0...255 will just work but I can't guarantee it.
What I'll probably do is constraint the core to the values in
hose->min/max, and update selected platforms to put 0..255 in there when
I know for sure they can cope.
But I think the point is, can't we intiialize the busn resource after
the first & last bus numbers have been determined? E.g. rather than
Yinghai's current:
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
+
/* Get probe mode and perform scan */
mode = PCI_PROBE_NORMAL;
if (node && ppc_md.pci_probe_mode)
of_scan_bus(node, bus);
}
if (mode == PCI_PROBE_NORMAL)
hose->last_busno = bus->subordinate = pci_scan_child_bus(bus);
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
since we should have the final bus range by then? Setting the end to
255 and then changing it again doesn't make sense; and definitely makes
the code hard to follow.
I have two issues here:
1) hose->last_busno is currently the highest bus number found by
pci_scan_child_bus(). If I understand correctly,
pci_bus_insert_busn_res() is supposed to update the core's idea of the
host bridge's bus number aperture. (Actually, I guess it just updates
the *end* of the aperture, since we supply the start directly to
pci_scan_root_bus()). The aperture and the highest bus number we
found are not related, except that we should have:
hose->first_busno <= bus->subordinate <= hose->last_busno
If we set the aperture to [first_busno - last_busno], we artificially
prevent some hotplug.
2) We already have a way to add resources to a root bus: the
pci_add_resource() used to add I/O port and MMIO apertures. I think
it'd be a lot simpler to just use that same interface for the bus
number aperture, e.g.,
pci_add_resource(&resources, hose->io_space);
pci_add_resource(&resources, hose->mem_space);
pci_add_resource(&resources, hose->busnr_space);
bus = pci_scan_root_bus(dev, next_busno, pci_ops, sysdata, &resources);
This is actually a bit redundant, since "next_busno" should be the
same as hose->busnr_space->start. So if we adopted this approach, we
might want to eventually drop the "next_busno" argument.
Bjorn
On Thu, 23 Feb 2012 12:51:30 -0800
Bjorn Helgaas [off-list ref] wrote:
On Thu, Feb 23, 2012 at 12:25 PM, Jesse Barnes [off-list ref] wrote:
quoted
On Fri, 10 Feb 2012 08:35:58 +1100
Benjamin Herrenschmidt [off-list ref] wrote:
quoted
On Thu, 2012-02-09 at 11:24 -0800, Bjorn Helgaas wrote:
quoted
My point is that the interface between the arch and the PCI core
should be simply the arch telling the core "this is the range of bus
numbers you can use." If the firmware doesn't give you the HW limits,
that's the arch's problem. If you want to assume 0..255 are
available, again, that's the arch's decision.
But the answer to the question "what bus numbers are available to me"
depends only on the host bridge HW configuration. It does not depend
on what pci_scan_child_bus() found. Therefore, I think we can come up
with a design where pci_bus_update_busn_res_end() is unnecessary.
In an ideal world yes. In a world where there are reverse engineered
platforms on which we aren't 100% sure how thing actually work under the
hood and have the code just adapt on "what's there" (and try to fix it
up -sometimes-), thinks can get a bit murky :-)
But yes, I see your point. As for what is the "correct" setting that
needs to be done so that the patch doesn't end up a regression for us,
I'll have to dig into some ancient HW to dbl check a few things. I hope
0...255 will just work but I can't guarantee it.
What I'll probably do is constraint the core to the values in
hose->min/max, and update selected platforms to put 0..255 in there when
I know for sure they can cope.
But I think the point is, can't we intiialize the busn resource after
the first & last bus numbers have been determined? E.g. rather than
Yinghai's current:
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
+
/* Get probe mode and perform scan */
mode = PCI_PROBE_NORMAL;
if (node && ppc_md.pci_probe_mode)
of_scan_bus(node, bus);
}
if (mode == PCI_PROBE_NORMAL)
hose->last_busno = bus->subordinate = pci_scan_child_bus(bus);
+ pci_bus_insert_busn_res(bus, hose->first_busno, hose->last_busno);
since we should have the final bus range by then? Setting the end to
255 and then changing it again doesn't make sense; and definitely makes
the code hard to follow.
I have two issues here:
1) hose->last_busno is currently the highest bus number found by
pci_scan_child_bus(). If I understand correctly,
pci_bus_insert_busn_res() is supposed to update the core's idea of the
host bridge's bus number aperture. (Actually, I guess it just updates
the *end* of the aperture, since we supply the start directly to
pci_scan_root_bus()). The aperture and the highest bus number we
found are not related, except that we should have:
hose->first_busno <= bus->subordinate <= hose->last_busno
If we set the aperture to [first_busno - last_busno], we artificially
prevent some hotplug.
Oh true, we'll need to allocate any extra bus number space somehow so
that hot plug of bridges is possible in the future w/o renumbering
(until our glorious future when we can move resources on the fly by
stopping drivers).
2) We already have a way to add resources to a root bus: the
pci_add_resource() used to add I/O port and MMIO apertures. I think
it'd be a lot simpler to just use that same interface for the bus
number aperture, e.g.,
pci_add_resource(&resources, hose->io_space);
pci_add_resource(&resources, hose->mem_space);
pci_add_resource(&resources, hose->busnr_space);
bus = pci_scan_root_bus(dev, next_busno, pci_ops, sysdata, &resources);
This is actually a bit redundant, since "next_busno" should be the
same as hose->busnr_space->start. So if we adopted this approach, we
might want to eventually drop the "next_busno" argument.
Yeah that would be nice, the call would certainly make more sense that
way.
--
Jesse Barnes, Intel Open Source Technology Center
From: Yinghai Lu <yinghai@kernel.org> Date: 2012-02-25 07:47:29
On Fri, Feb 24, 2012 at 2:24 PM, Jesse Barnes [off-list ref] wrote:
On Thu, 23 Feb 2012 12:51:30 -0800
Bjorn Helgaas [off-list ref] wrote:
quoted
2) We already have a way to add resources to a root bus: the
pci_add_resource() used to add I/O port and MMIO apertures. I think
it'd be a lot simpler to just use that same interface for the bus
number aperture, e.g.,
pci_add_resource(&resources, hose->io_space);
pci_add_resource(&resources, hose->mem_space);
pci_add_resource(&resources, hose->busnr_space);
bus = pci_scan_root_bus(dev, next_busno, pci_ops, sysdata, &resources);
This is actually a bit redundant, since "next_busno" should be the
same as hose->busnr_space->start. So if we adopted this approach, we
might want to eventually drop the "next_busno" argument.
Yeah that would be nice, the call would certainly make more sense that
way.
no, I don't think so.
using pci_add_resource will need to create dummy resource abut bus range.
there is lots of pci_scan_root_bus(), and those user does not bus end
yet before scan.
so could just hide pci_insert_busn_res in pci_scan_root_bus, and
update busn_res end there.
other arch like x86, ia64, powerpc, sparc, will insert exact bus range
between pci_create_root_bus and
pci_scan_child_bus, will not need to update busn_res end.
please check v7 of this patchset.
git://git.kernel.org/pub/scm/linux/kernel/git/yinghai/linux-yinghai.git
for-pci-busn-alloc
It should be clean and have minimum lines of change.
Thanks
Yinghai
On Sat, Feb 25, 2012 at 12:47 AM, Yinghai Lu [off-list ref] wrote:
On Fri, Feb 24, 2012 at 2:24 PM, Jesse Barnes [off-list ref] wrote:
quoted
On Thu, 23 Feb 2012 12:51:30 -0800
Bjorn Helgaas [off-list ref] wrote:
quoted
2) We already have a way to add resources to a root bus: the
pci_add_resource() used to add I/O port and MMIO apertures. I think
it'd be a lot simpler to just use that same interface for the bus
number aperture, e.g.,
pci_add_resource(&resources, hose->io_space);
pci_add_resource(&resources, hose->mem_space);
pci_add_resource(&resources, hose->busnr_space);
bus = pci_scan_root_bus(dev, next_busno, pci_ops, sysdata, &resources);
This is actually a bit redundant, since "next_busno" should be the
same as hose->busnr_space->start. So if we adopted this approach, we
might want to eventually drop the "next_busno" argument.
Yeah that would be nice, the call would certainly make more sense that
way.
no, I don't think so.
using pci_add_resource will need to create dummy resource abut bus range.
I don't see the problem here. The bus number aperture is a
fundamental property of a host bridge, and any firmware or native
bridge driver that tells you about a bridge but doesn't tell you the
bus number aperture is just broken.
Every arch already has struct resources for the MMIO and IO port
regions available on the bus. ACPI already has a resource for the bus
number range. It makes sense to me that the arch should supply a bus
number resource.
Conceptually, it's just like the MMIO and IO resources, so it makes
sense to me that the code for bus numbers should look like the code
for MMIO and IO ports.
there is lots of pci_scan_root_bus(), and those user does not bus end
yet before scan.
so could just hide pci_insert_busn_res in pci_scan_root_bus, and
update busn_res end there.
pci_scan_child_bus() does NOT tell you the end of the bus number
aperture, and we shouldn't pretend that it does. It might give you a
lower bound on the end of the aperture (as long as you're willing to
trust the current PCI config and you don't change anything).
other arch like x86, ia64, powerpc, sparc, will insert exact bus range
between pci_create_root_bus and
pci_scan_child_bus, will not need to update busn_res end.
please check v7 of this patchset.
git://git.kernel.org/pub/scm/linux/kernel/git/yinghai/linux-yinghai.git
for-pci-busn-alloc
I looked at your git tree, but I can't tell whether what's there is v7
or not and it's too much trouble to try to figure it out.
Bjorn