@@ -0,0 +1,108 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright2003-2008BroadcomCorporation+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/++#include <asm/memory.h>++#include <linux/linkage.h>+#include <linux/init.h>++#define __virt_to_phys(x) ((x) - PAGE_OFFSET)++/*+*v7_l1_cache_invalidate+*+*InvalidatecontentsofL1cachewithoutflushingitscontents+*intooutercacheandmemory.Thisisneededwhenthecontents+*ofthecacheareunpredictableafterpower-up.+*+*corruptsr0-r6+*/+ENTRY(v7_l1_cache_invalidate)+movr0,#0+mcrp15,2,r0,c0,c0,0@setcachelevelto1+mrcp15,1,r0,c0,c0,0@readCLIDR++ldrr1,=0x7fff+andr2,r1,r0,lsr#13 @ get max # of index size++ldrr1,=0x3ff+andr3,r1,r0,lsr#3 @ NumWays - 1+addr2,r2,#1 @ NumSets++andr0,r0,#0x7+addr0,r0,#4 @ SetShift++clzr1,r3@WayShift+addr4,r3,#1 @ NumWays+1:subr2,r2,#1 @ NumSets--+movr3,r4@Temp=NumWays+2:subsr3,r3,#1 @ Temp--+movr5,r3,lslr1+movr6,r2,lslr0+orrr5,r5,r6@Reg=(Temp<<WayShift)|(NumSets<<SetShift)+mcrp15,0,r5,c7,c6,2@Invalidateline+bgt2b+cmpr2,#0+bgt1b+dsb+movr0,#0+mcrp15,0,r0,c7,c5,0/*Invalidateicache*/+isb+movpc,lr+ENDPROC(v7_l1_cache_invalidate)++/*+*PlatformspecificentrypointforsecondaryCPUs.This+*providesa"holding pen"intowhichallsecondarycoresareheld+*untilwe're ready for them to initialise.+*/+__CPUINIT+ENTRY(bcm5301x_secondary_startup)+/*+*GethardwareCPUidofours+*/+mrcp15,0,r0,c0,c0,5+andr0,r0,#15+/*+*Waiton<pen_release>variablebyphysicaladdress+*tocontainourhardwareCPUid+*/+#ifdef CONFIG_SPARSEMEM+ldrr2,=(PAGE_OFFSET+SZ_128M)+ldrr1,=pen_release+cmpr1,r2+bge1f+ldrr2,=PAGE_OFFSET+subr6,r1,r2+b2f+1:+subr1,r1,r2+ldrr2,=PHYS_OFFSET2+addr6,r1,r2+2:+#else+ldrr6,=__virt_to_phys(pen_release)+#endif+pen:ldrr7,[r6]+cmpr7,r0+bnepen+nop+/*+*IncaseL1cachehasunpredictablecontentsatpower-up+*cleanitscontentswithoutflushing.+*/+blv7_l1_cache_invalidate+nop+/*+*we've been released from the holding pen: secondary_stack+*shouldnowcontaintheSVCstackforthiscore+*/+bsecondary_startup++ENDPROC(bcm5301x_secondary_startup)+.ltorg
@@ -0,0 +1,185 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright(C)2002ARMLtd.+*Copyright(C)2015Rafa?Mi?ecki<zajec5@gmail.com>+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/++#include<asm/cacheflush.h>+#include<asm/delay.h>+#include<asm/smp_scu.h>++#include<linux/clockchips.h>++/*+*Thereisa1KBLUTlocatedat0xFFFF0400-0xFFFFFFFF,anditsfirstentry+*iswherethesecondaryentrypointneedstobewritten+*/+#define SOC_ROM_BASE_PA 0xffff0000+#define SOC_ROM_LUT_OFF 0x400++/* ENTRY in bcm5301x_headsmp.S */+externvoidbcm5301x_secondary_startup(void);++staticDEFINE_SPINLOCK(boot_lock);++staticvoid__cpuinitwrite_pen_release(intval)+{+pen_release=val;+/* Make sure this store is visible to other CPUs */+smp_wmb();+__cpuc_flush_dcache_area((void*)&pen_release,sizeof(pen_release));+outer_clean_range(__pa(&pen_release),__pa(&pen_release+1));+}++staticvoid__initbcm5301x_smp_secondary_set_entry(void(*entry_point)(void))+{+void__iomem*rombase=NULL;+phys_addr_tlut_pa;+u32offset,mask;+u32val;++mask=(1UL<<PAGE_SHIFT)-1;++lut_pa=SOC_ROM_BASE_PA&~mask;+offset=SOC_ROM_BASE_PA&mask;+offset+=SOC_ROM_LUT_OFF;++rombase=ioremap(lut_pa,PAGE_SIZE);+if(!rombase)+return;+val=virt_to_phys(entry_point);++writel(val,rombase+offset);++smp_wmb();/* probably not needed - io regs are not cached */+dsb_sev();/* Exit WFI */+mb();++iounmap(rombase);+}++staticvoid__initbcm5301x_smp_prepare_cpus(unsignedintmax_cpus)+{+void__iomem*scu_base;+unsignedintncores;++if(!scu_a9_has_base()){+pr_warn("Unknown SCU base\n");+return;+}++scu_base=ioremap((phys_addr_t)scu_a9_get_base(),SZ_256);+if(!scu_base){+pr_err("Failed to remap SCU\n");+return;+}++ncores=scu_get_core_count(scu_base);+if(max_cpus>ncores){+unsignedinti;++pr_warn("Possible CPU mask exceeds available cores, reducing to %u\n",+ncores);+for(i=ncores-1;i<max_cpus;i++)+set_cpu_present(i,false);+max_cpus=ncores;+}++if(max_cpus>1){+/* nobody is to be released from the pen yet */+pen_release=-1;++/* Initialise the SCU */+scu_enable(scu_base);++/* Let CPUs know where to start */+bcm5301x_smp_secondary_set_entry(bcm5301x_secondary_startup);+}++iounmap(scu_base);+}++staticvoid__cpuinitbcm5301x_smp_secondary_init(unsignedintcpu)+{+trace_hardirqs_off();++/*+*lettheprimaryprocessorknowwe'reoutofthe+*pen,thenheadoffintotheCentrypoint+*/+write_pen_release(-1);++/*+*Synchronisewiththebootthread.+*/+spin_lock(&boot_lock);+spin_unlock(&boot_lock);+}++staticint__cpuinitbcm5301x_smp_boot_secondary(unsignedintcpu,+structtask_struct*idle)+{+unsignedlongtimeout;++/*+*setsynchronisationstatebetweenthisbootprocessor+*andthesecondaryone+*/+spin_lock(&boot_lock);++/*+*Thesecondaryprocessoriswaitingtobereleasedfrom+*theholdingpen-releaseit,thenwaitforittoflag+*thatithasbeenreleasedbyresettingpen_release.+*+*Notethat"pen_release"isthehardwareCPUID,whereas+*"cpu"isLinux'sinternalID.+*/+write_pen_release(cpu);++dsb_sev();++/*+*Timeoutsetonpurposeinjiffiessothatonslowprocessors+*thatmustalsohavelowHZitwillwaitlonger.+*/+timeout=jiffies+(HZ*10);++udelay(100);++/*+*IfthesecondaryCPUwaswaitingonWFE,itshould+*bealreadywatching<pen_release>,oritcouldbe+*waitinginWFI,senditanIPItobesureitwakes.+*/+if(pen_release!=-1)+tick_broadcast(cpumask_of(cpu));++while(time_before(jiffies,timeout)){+smp_rmb();+if(pen_release==-1)+break;++udelay(10);+}++/*+*nowthesecondarycoreisstartingupletitrunits+*calibrations,thenwaitforittofinish+*/+spin_unlock(&boot_lock);++returnpen_release!=-1?-ENOSYS:0;+}++staticstructsmp_operationsbcm5301x_smp_ops__initdata={+.smp_prepare_cpus=bcm5301x_smp_prepare_cpus,+.smp_secondary_init=bcm5301x_smp_secondary_init,+.smp_boot_secondary=bcm5301x_smp_boot_secondary,+};++CPU_METHOD_OF_DECLARE(bcm5301x_smp,"brcm,bcm4708-smp",+&bcm5301x_smp_ops);
Thanks for working on this.
Someone else with more knowledge about arm cortex A9 SMP stuff should
look at this patch.
Did you had a look at mach-rockchip/platsmp.c ? While I was looking at
SMP stuff this code looked clean to me and they are also using a Cortex A9.
There are some comments in the code.
Hauke
On 02/10/2015 09:32 PM, Rafa? Mi?ecki wrote:
@@ -0,0 +1,108 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright2003-2008BroadcomCorporation+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/++#include <asm/memory.h>++#include <linux/linkage.h>+#include <linux/init.h>++#define __virt_to_phys(x) ((x) - PAGE_OFFSET)++/*+*v7_l1_cache_invalidate+*+*InvalidatecontentsofL1cachewithoutflushingitscontents+*intooutercacheandmemory.Thisisneededwhenthecontents+*ofthecacheareunpredictableafterpower-up.+*+*corruptsr0-r6+*/+ENTRY(v7_l1_cache_invalidate)+movr0,#0+mcrp15,2,r0,c0,c0,0@setcachelevelto1+mrcp15,1,r0,c0,c0,0@readCLIDR++ldrr1,=0x7fff+andr2,r1,r0,lsr#13 @ get max # of index size++ldrr1,=0x3ff+andr3,r1,r0,lsr#3 @ NumWays - 1+addr2,r2,#1 @ NumSets++andr0,r0,#0x7+addr0,r0,#4 @ SetShift++clzr1,r3@WayShift+addr4,r3,#1 @ NumWays+1:subr2,r2,#1 @ NumSets--+movr3,r4@Temp=NumWays+2:subsr3,r3,#1 @ Temp--+movr5,r3,lslr1+movr6,r2,lslr0+orrr5,r5,r6@Reg=(Temp<<WayShift)|(NumSets<<SetShift)+mcrp15,0,r5,c7,c6,2@Invalidateline+bgt2b+cmpr2,#0+bgt1b+dsb+movr0,#0+mcrp15,0,r0,c7,c5,0/*Invalidateicache*/+isb+movpc,lr+ENDPROC(v7_l1_cache_invalidate)
This function looks similar to v7_invalidate_l1 and v7_flush_icache_all
in arch/arm/mm/cache-v7.S if it is different and it is intended could
you please point it out in the comment.
quoted hunk
+/*
+ * Platform specific entry point for secondary CPUs. This
+ * provides a "holding pen" into which all secondary cores are held
+ * until we're ready for them to initialise.
+ */
+ __CPUINIT
+ENTRY(bcm5301x_secondary_startup)
+ /*
+ * Get hardware CPU id of ours
+ */
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ /*
+ * Wait on <pen_release> variable by physical address
+ * to contain our hardware CPU id
+ */
+#ifdef CONFIG_SPARSEMEM
+ ldr r2, =(PAGE_OFFSET+SZ_128M)
+ ldr r1, =pen_release
+ cmp r1, r2
+ bge 1f
+ ldr r2, =PAGE_OFFSET
+ sub r6, r1, r2
+ b 2f
+1:
+ sub r1, r1, r2
+ ldr r2, =PHYS_OFFSET2
+ add r6, r1, r2
+2:
+#else
+ ldr r6, =__virt_to_phys(pen_release)
+#endif
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+ nop
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_l1_cache_invalidate
+ nop
+ /*
+ * we've been released from the holding pen: secondary_stack
+ * should now contain the SVC stack for this core
+ */
+ b secondary_startup
+
+ENDPROC(bcm5301x_secondary_startup)
+ .ltorg
+
+/* ENTRY in bcm5301x_headsmp.S */
+extern void bcm5301x_secondary_startup(void);
Shouldn't this go into some common header? I think we do not have such a
header so it is not that of a problem for now.
+
+static DEFINE_SPINLOCK(boot_lock);
+
+static void __cpuinit write_pen_release(int val)
+{
+ pen_release = val;
+ /* Make sure this store is visible to other CPUs */
+ smp_wmb();
+ __cpuc_flush_dcache_area((void *)&pen_release, sizeof(pen_release));
+ outer_clean_range(__pa(&pen_release), __pa(&pen_release + 1));
+}
+
+static void __init bcm5301x_smp_secondary_set_entry(void (*entry_point)(void))
+{
+ void __iomem *rombase = NULL;
+ phys_addr_t lut_pa;
+ u32 offset, mask;
+ u32 val;
+
+ mask = (1UL << PAGE_SHIFT) - 1;
+
+ lut_pa = SOC_ROM_BASE_PA & ~mask;
+ offset = SOC_ROM_BASE_PA & mask;
+ offset += SOC_ROM_LUT_OFF;
+
+ rombase = ioremap(lut_pa, PAGE_SIZE);
+ if (!rombase)
+ return;
+ val = virt_to_phys(entry_point);
+
+ writel(val, rombase + offset);
+
+ smp_wmb(); /* probably not needed - io regs are not cached */
+ dsb_sev(); /* Exit WFI */
+ mb();
+
+ iounmap(rombase);
+}
+
+static void __init bcm5301x_smp_prepare_cpus(unsigned int max_cpus)
+{
+ void __iomem *scu_base;
+ unsigned int ncores;
+
+ if (!scu_a9_has_base()) {
+ pr_warn("Unknown SCU base\n");
+ return;
+ }
+
+ scu_base = ioremap((phys_addr_t)scu_a9_get_base(), SZ_256);
+ if (!scu_base) {
+ pr_err("Failed to remap SCU\n");
+ return;
+ }
+
+ ncores = scu_get_core_count(scu_base);
+ if (max_cpus > ncores) {
+ unsigned int i;
+
+ pr_warn("Possible CPU mask exceeds available cores, reducing to %u\n",
+ ncores);
+ for (i = ncores - 1; i < max_cpus; i++)
+ set_cpu_present(i, false);
+ max_cpus = ncores;
+ }
+
+ if (max_cpus > 1) {
+ /* nobody is to be released from the pen yet */
+ pen_release = -1;
+
+ /* Initialise the SCU */
+ scu_enable(scu_base);
+
+ /* Let CPUs know where to start */
+ bcm5301x_smp_secondary_set_entry(bcm5301x_secondary_startup);
+ }
+
+ iounmap(scu_base);
+}
+
+static void __cpuinit bcm5301x_smp_secondary_init(unsigned int cpu)
+{
+ trace_hardirqs_off();
+
+ /*
+ * let the primary processor know we're out of the
+ * pen, then head off into the C entry point
+ */
+ write_pen_release(-1);
+
+ /*
+ * Synchronise with the boot thread.
+ */
+ spin_lock(&boot_lock);
+ spin_unlock(&boot_lock);
+}
+
+static int __cpuinit bcm5301x_smp_boot_secondary(unsigned int cpu,
+ struct task_struct *idle)
+{
+ unsigned long timeout;
+
+ /*
+ * set synchronisation state between this boot processor
+ * and the secondary one
+ */
+ spin_lock(&boot_lock);
+
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu);
+
+ dsb_sev();
+
+ /*
+ * Timeout set on purpose in jiffies so that on slow processors
+ * that must also have low HZ it will wait longer.
+ */
+ timeout = jiffies + (HZ * 10);
+
+ udelay(100);
+
+ /*
+ * If the secondary CPU was waiting on WFE, it should
+ * be already watching <pen_release>, or it could be
+ * waiting in WFI, send it an IPI to be sure it wakes.
+ */
+ if (pen_release != -1)
+ tick_broadcast(cpumask_of(cpu));
+
+ while (time_before(jiffies, timeout)) {
+ smp_rmb();
+ if (pen_release == -1)
+ break;
+
+ udelay(10);
+ }
+
+ /*
+ * now the secondary core is starting up let it run its
+ * calibrations, then wait for it to finish
+ */
+ spin_unlock(&boot_lock);
+
+ return pen_release != -1 ? -ENOSYS : 0;
+}
+
+static struct smp_operations bcm5301x_smp_ops __initdata = {
+ .smp_prepare_cpus = bcm5301x_smp_prepare_cpus,
+ .smp_secondary_init = bcm5301x_smp_secondary_init,
+ .smp_boot_secondary = bcm5301x_smp_boot_secondary,
+};
+
+CPU_METHOD_OF_DECLARE(bcm5301x_smp, "brcm,bcm4708-smp",
+ &bcm5301x_smp_ops);
This must be documented.
We really should be getting to the point where we have a small number of
standard(ish) enable methods rather than just adding a load of new IMP
DEF methods with pointless differences.
SUrely you're missing a dsb after the i-cache maintenance?
+ mov pc, lr
+ENDPROC(v7_l1_cache_invalidate)
This looks like a total mess. If you _really_ need this, factor it out
of the existing cache flush infrastructure. We don't need more broken
copies.
Do you have a guarantee that the CPU won't write back any of this
naturally before the invalidate is complete?
Is the CPU coherent at this point?
+
+/*
+ * Platform specific entry point for secondary CPUs. This
+ * provides a "holding pen" into which all secondary cores are held
+ * until we're ready for them to initialise.
+ */
+ __CPUINIT
+ENTRY(bcm5301x_secondary_startup)
+ /*
+ * Get hardware CPU id of ours
+ */
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
Test all of the MPIDR.Aff* bits, please.
+ /*
+ * Wait on <pen_release> variable by physical address
+ * to contain our hardware CPU id
+ */
+#ifdef CONFIG_SPARSEMEM
+ ldr r2, =(PAGE_OFFSET+SZ_128M)
+ ldr r1, =pen_release
+ cmp r1, r2
+ bge 1f
+ ldr r2, =PAGE_OFFSET
+ sub r6, r1, r2
+ b 2f
+1:
+ sub r1, r1, r2
+ ldr r2, =PHYS_OFFSET2
+ add r6, r1, r2
+2:
Huh? We really shouldn't have to care about SPARSEMEM in this kind of
code. I assume the fundamental issue here is your custom __virt_to_phys
implementation.
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_l1_cache_invalidate
+ nop
Another pointless nop?
quoted hunk
+ /*
+ * we've been released from the holding pen: secondary_stack
+ * should now contain the SVC stack for this core
+ */
+ b secondary_startup
+
+ENDPROC(bcm5301x_secondary_startup)
+ .ltorg
+ if (max_cpus > ncores) {
+ unsigned int i;
+
+ pr_warn("Possible CPU mask exceeds available cores, reducing to %u\n",
+ ncores);
+ for (i = ncores - 1; i < max_cpus; i++)
+ set_cpu_present(i, false);
+ max_cpus = ncores;
+ }
+
+ if (max_cpus > 1) {
+ /* nobody is to be released from the pen yet */
+ pen_release = -1;
+
+ /* Initialise the SCU */
+ scu_enable(scu_base);
+
+ /* Let CPUs know where to start */
+ bcm5301x_smp_secondary_set_entry(bcm5301x_secondary_startup);
+ }
+
+ iounmap(scu_base);
+}
[...]
+static int __cpuinit bcm5301x_smp_boot_secondary(unsigned int cpu,
+ struct task_struct *idle)
+{
+ unsigned long timeout;
+
+ /*
+ * set synchronisation state between this boot processor
+ * and the secondary one
+ */
+ spin_lock(&boot_lock);
+
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu);
As far as I can tell you're relying on the logical ID being equivalent
to MPDR.Aff0, which isn't necessarily true. Either use the physical ID
or use the actual logical ID.
+
+ dsb_sev();
+
+ /*
+ * Timeout set on purpose in jiffies so that on slow processors
+ * that must also have low HZ it will wait longer.
+ */
+ timeout = jiffies + (HZ * 10);
+
+ udelay(100);
+
+ /*
+ * If the secondary CPU was waiting on WFE, it should
+ * be already watching <pen_release>, or it could be
+ * waiting in WFI, send it an IPI to be sure it wakes.
+ */
+ if (pen_release != -1)
+ tick_broadcast(cpumask_of(cpu));
NAK. This is not what tick_broadcast is intended for.
If you need an IPI then send an IPI, don't piggyback on the timekeeping
infrastructure.
Mark.
@@ -6,3 +6,20 @@ Boards with the BCM4708 SoC shall have the following properties: Required root node property: compatible = "brcm,bcm4708";++Optional sub-node properties:++compatible = "brcm,bcm4708-sysram" SYSRAM for SMP bringup.+ SMP-capable SoCs use part of the SYSRAM for storing+ location of code to be executed by the extra cores.++Example:++ {+ compatible = "brcm,bcm4708";++ smp-sysram {+ compatible = "brcm,bcm4708-sysram";+ reg = <0xffff0000 0x1000>;+ };+ };
@@ -188,6 +188,7 @@ nodes to be present and contain the properties described below. can be one of: "allwinner,sun6i-a31" "arm,psci"+ "brcm,bcm4708-smp" "brcm,brahma-b15" "marvell,armada-375-smp" "marvell,armada-380-smp"
@@ -0,0 +1,46 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright (c)2003ARMLimited+*AllRightsReserved+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/+#include <linux/linkage.h>++/*+*BCM5301XspecificentrypointforsecondaryCPUs.+*/+ENTRY(bcm5301x_secondary_startup)+mrcp15,0,r0,c0,c0,5+andr0,r0,#15+adrr4,1f+ldmiar4,{r5,r6}+subr4,r4,r5+addr6,r6,r4+pen:ldrr7,[r6]+cmpr7,r0+bnepen++/*+*IncaseL1cachehasunpredictablecontentsatpower-up+*cleanitscontentswithoutflushing.+*/+/*blv7_l1_cache_invalidate*/+blv7_invalidate_l1++movr0,#0+mcrp15,0,r0,c7,c5,0/*Invalidateicache*/+dsb+isb++/*+*we've been released from the holding pen: secondary_stack+*shouldnowcontaintheSVCstackforthiscore+*/+bsecondary_startup+ENDPROC(bcm5301x_secondary_startup)++.align2+1:.long.+.longpen_release
@@ -0,0 +1,160 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright(C)2002ARMLtd.+*Copyright(C)2015Rafa?Mi?ecki<zajec5@gmail.com>+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/++#include<asm/cacheflush.h>+#include<asm/delay.h>+#include<asm/smp_plat.h>+#include<asm/smp_scu.h>++#include<linux/clockchips.h>+#include<linux/of.h>+#include<linux/of_address.h>++#define SOC_ROM_LUT_OFF 0x400++externvoidbcm5301x_secondary_startup(void);++staticvoid__cpuinitwrite_pen_release(intval)+{+pen_release=val;+smp_wmb();+sync_cache_w(&pen_release);+}++staticDEFINE_SPINLOCK(boot_lock);++staticvoid__initbcm5301x_smp_secondary_set_entry(void(*entry_point)(void))+{+void__iomem*sysram_base_addr=NULL;+structdevice_node*node;++for_each_compatible_node(node,NULL,"brcm,bcm4708-sysram"){+if(!of_device_is_available(node))+continue;+sysram_base_addr=of_iomap(node,0);+break;+}++if(!sysram_base_addr){+pr_warn("Failed to map sysram\n");+return;+}++writel(virt_to_phys(entry_point),sysram_base_addr+SOC_ROM_LUT_OFF);++dsb_sev();/* Exit WFI */+mb();/* make sure write buffer is drained */++iounmap(sysram_base_addr);+}++staticvoid__initbcm5301x_smp_prepare_cpus(unsignedintmax_cpus)+{+void__iomem*scu_base;++if(!scu_a9_has_base()){+pr_warn("Unknown SCU base\n");+return;+}++scu_base=ioremap((phys_addr_t)scu_a9_get_base(),SZ_256);+if(!scu_base){+pr_err("Failed to remap SCU\n");+return;+}++/* Initialise the SCU */+scu_enable(scu_base);++/* Let CPUs know where to start */+bcm5301x_smp_secondary_set_entry(bcm5301x_secondary_startup);++iounmap(scu_base);+}++staticvoid__cpuinitbcm5301x_smp_secondary_init(unsignedintcpu)+{+trace_hardirqs_off();++/*+*lettheprimaryprocessorknowwe'reoutofthe+*pen,thenheadoffintotheCentrypoint+*/+write_pen_release(-1);++/*+*Synchronisewiththebootthread.+*/+spin_lock(&boot_lock);+spin_unlock(&boot_lock);+}++staticint__cpuinitbcm5301x_smp_boot_secondary(unsignedintcpu,+structtask_struct*idle)+{+unsignedlongtimeout;++/*+*setsynchronisationstatebetweenthisbootprocessor+*andthesecondaryone+*/+spin_lock(&boot_lock);++/*+*Thesecondaryprocessoriswaitingtobereleasedfrom+*theholdingpen-releaseit,thenwaitforittoflag+*thatithasbeenreleasedbyresettingpen_release.+*+*Notethat"pen_release"isthehardwareCPUID,whereas+*"cpu"isLinux'sinternalID.+*/+write_pen_release(cpu_logical_map(cpu));++/* Send the secondary CPU SEV */+dsb_sev();++udelay(100);++/*+*SendthesecondaryCPUasoftinterrupt,therebycausing+*thebootmonitortoreadthesystemwideflagsregister,+*andbranchtotheaddressfoundthere.+*/+arch_send_wakeup_ipi_mask(cpumask_of(cpu));++/*+*Timeoutsetonpurposeinjiffiessothatonslowprocessors+*thatmustalsohavelowHZitwillwaitlonger.+*/+timeout=jiffies+(HZ*10);+while(time_before(jiffies,timeout)){+smp_rmb();+if(pen_release==-1)+break;++udelay(10);+}++/*+*nowthesecondarycoreisstartingupletitrunits+*calibrations,thenwaitforittofinish+*/+spin_unlock(&boot_lock);++returnpen_release!=-1?-ENOSYS:0;+}++staticstructsmp_operationsbcm5301x_smp_ops__initdata={+.smp_prepare_cpus=bcm5301x_smp_prepare_cpus,+.smp_secondary_init=bcm5301x_smp_secondary_init,+.smp_boot_secondary=bcm5301x_smp_boot_secondary,+};++CPU_METHOD_OF_DECLARE(bcm5301x_smp,"brcm,bcm4708-smp",+&bcm5301x_smp_ops);
+Optional sub-node properties:
+
+compatible = "brcm,bcm4708-sysram" SYSRAM for SMP bringup.
+ SMP-capable SoCs use part of the SYSRAM for storing
+ location of code to be executed by the extra cores.
Is this a regular kind of SRAM? If so, can you use "mmio-sram" as a
compatible fallback?
[snip]
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ /* bl v7_l1_cache_invalidate */
For consistency with the previous lines, you would want to space operands.
quoted hunk
+ dsb
+ isb
+
+ /*
+ * we've been released from the holding pen: secondary_stack
+ * should now contain the SVC stack for this core
+ */
+ b secondary_startup
+ENDPROC(bcm5301x_secondary_startup)
+
+ .align 2
+1: .long .
+ .long pen_release
@@ -6,3 +6,27 @@ Boards with the BCM4708 SoC shall have the following properties: Required root node property: compatible = "brcm,bcm4708";++Optional sub-node properties:++compatible = "mmio-sram" for SRAM access with IO memory region+ This is needed for SMP-capable SoCs which use part of+ SRAM for storing location of code to be executed by the+ extra cores.+ SMP support requires another sub-node with compatible+ property "brcm,bcm4708-sysram".++Example:++ sysram at ffff0000 {+ compatible = "mmio-sram";+ reg = <0xffff0000 0x10000>;+ #address-cells = <1>;+ #size-cells = <1>;+ ranges = <0 0xffff0000 0x10000>;++ smp-sysram at 0 {+ compatible = "brcm,bcm4708-sysram";+ reg = <0x0 0x1000>;+ };+ };
@@ -189,6 +189,7 @@ nodes to be present and contain the properties described below. can be one of: "allwinner,sun6i-a31" "arm,psci"+ "brcm,bcm4708-smp" "brcm,brahma-b15" "marvell,armada-375-smp" "marvell,armada-380-smp"
@@ -0,0 +1,45 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright (c)2003ARMLimited+*AllRightsReserved+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/+#include <linux/linkage.h>++/*+*BCM5301XspecificentrypointforsecondaryCPUs.+*/+ENTRY(bcm5301x_secondary_startup)+mrcp15,0,r0,c0,c0,5+andr0,r0,#15+adrr4,1f+ldmiar4,{r5,r6}+subr4,r4,r5+addr6,r6,r4+pen:ldrr7,[r6]+cmpr7,r0+bnepen++/*+*IncaseL1cachehasunpredictablecontentsatpower-up+*cleanitscontentswithoutflushing.+*/+blv7_invalidate_l1++movr0,#0+mcrp15,0,r0,c7,c5,0/*Invalidateicache*/+dsb+isb++/*+*we've been released from the holding pen: secondary_stack+*shouldnowcontaintheSVCstackforthiscore+*/+bsecondary_startup+ENDPROC(bcm5301x_secondary_startup)++.align2+1:.long.+.longpen_release
@@ -0,0 +1,158 @@+/*+*BroadcomBCM470X/BCM5301XARMplatformcode.+*+*Copyright(C)2002ARMLtd.+*Copyright(C)2015Rafa?Mi?ecki<zajec5@gmail.com>+*+*LicensedundertheGNU/GPL.SeeCOPYINGfordetails.+*/++#include<asm/cacheflush.h>+#include<asm/delay.h>+#include<asm/smp_plat.h>+#include<asm/smp_scu.h>++#include<linux/clockchips.h>+#include<linux/of.h>+#include<linux/of_address.h>++#define SOC_ROM_LUT_OFF 0x400++externvoidbcm5301x_secondary_startup(void);++staticvoid__cpuinitwrite_pen_release(intval)+{+pen_release=val;+smp_wmb();+sync_cache_w(&pen_release);+}++staticDEFINE_SPINLOCK(boot_lock);++staticvoid__initbcm5301x_smp_secondary_set_entry(void(*entry_point)(void))+{+void__iomem*sysram_base_addr=NULL;+structdevice_node*node;++node=of_find_compatible_node(NULL,NULL,"brcm,bcm4708-sysram");+if(!of_device_is_available(node))+return;++sysram_base_addr=of_iomap(node,0);+if(!sysram_base_addr){+pr_warn("Failed to map sysram\n");+return;+}++writel(virt_to_phys(entry_point),sysram_base_addr+SOC_ROM_LUT_OFF);++dsb_sev();/* Exit WFI */+mb();/* make sure write buffer is drained */++iounmap(sysram_base_addr);+}++staticvoid__initbcm5301x_smp_prepare_cpus(unsignedintmax_cpus)+{+void__iomem*scu_base;++if(!scu_a9_has_base()){+pr_warn("Unknown SCU base\n");+return;+}++scu_base=ioremap((phys_addr_t)scu_a9_get_base(),SZ_256);+if(!scu_base){+pr_err("Failed to remap SCU\n");+return;+}++/* Initialise the SCU */+scu_enable(scu_base);++/* Let CPUs know where to start */+bcm5301x_smp_secondary_set_entry(bcm5301x_secondary_startup);++iounmap(scu_base);+}++staticvoid__cpuinitbcm5301x_smp_secondary_init(unsignedintcpu)+{+trace_hardirqs_off();++/*+*lettheprimaryprocessorknowwe'reoutofthe+*pen,thenheadoffintotheCentrypoint+*/+write_pen_release(-1);++/*+*Synchronisewiththebootthread.+*/+spin_lock(&boot_lock);+spin_unlock(&boot_lock);+}++staticint__cpuinitbcm5301x_smp_boot_secondary(unsignedintcpu,+structtask_struct*idle)+{+unsignedlongtimeout;++/*+*setsynchronisationstatebetweenthisbootprocessor+*andthesecondaryone+*/+spin_lock(&boot_lock);++/*+*Thesecondaryprocessoriswaitingtobereleasedfrom+*theholdingpen-releaseit,thenwaitforittoflag+*thatithasbeenreleasedbyresettingpen_release.+*+*Notethat"pen_release"isthehardwareCPUID,whereas+*"cpu"isLinux'sinternalID.+*/+write_pen_release(cpu_logical_map(cpu));++/* Send the secondary CPU SEV */+dsb_sev();++udelay(100);++/*+*SendthesecondaryCPUasoftinterrupt,therebycausing+*thebootmonitortoreadthesystemwideflagsregister,+*andbranchtotheaddressfoundthere.+*/+arch_send_wakeup_ipi_mask(cpumask_of(cpu));++/*+*Timeoutsetonpurposeinjiffiessothatonslowprocessors+*thatmustalsohavelowHZitwillwaitlonger.+*/+timeout=jiffies+(HZ*10);+while(time_before(jiffies,timeout)){+smp_rmb();+if(pen_release==-1)+break;++udelay(10);+}++/*+*nowthesecondarycoreisstartingupletitrunits+*calibrations,thenwaitforittofinish+*/+spin_unlock(&boot_lock);++returnpen_release!=-1?-ENOSYS:0;+}++staticstructsmp_operationsbcm5301x_smp_ops__initdata={+.smp_prepare_cpus=bcm5301x_smp_prepare_cpus,+.smp_secondary_init=bcm5301x_smp_secondary_init,+.smp_boot_secondary=bcm5301x_smp_boot_secondary,+};++CPU_METHOD_OF_DECLARE(bcm5301x_smp,"brcm,bcm4708-smp",+&bcm5301x_smp_ops);
From: Russell King - ARM Linux <hidden> Date: 2015-03-26 12:00:26
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
--
FTTC broadband for 0.8mile line: currently at 10.5Mbps down 400kbps up
according to speedtest.net.
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
On 10/13/2015 3:29 PM, Hauke Mehrtens wrote:
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
From: Kapil Hali <hidden> Date: 2015-10-14 13:42:15
On 10/14/2015 4:18 AM, Ray Jui wrote:
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
Ray, I don't have complete/other patch sets for this change. It would
be good if I get those patch sets as well or complete e-mail thread.
I think if we have a cleaner solutions for SMP, we can consolidate
the change required for NS and NSP. I have few points to add, which
are inline in this e-mail.
On 10/13/2015 3:29 PM, Hauke Mehrtens wrote:
quoted
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Are you sure this is an issue of unpredictable L1 cache contents at
power-up? AFAIK, 5301X had an issue with secondary core initialization.
Secondary core which waits on WFE would let it out of the pen as soon as the
first spin_*lock executes. This was because of a BOOTROM bug in NS, so the
work around was to reset the address for the secondary processor to go back
and wait for the signal from the primary core. This vector fixup is required
so that the secondary core doesn't start executing kernel instructions until
we've patched its jump address during wakeup_secondary().
Also, v7 setup function should invalidate L1 cache and we should remove all
v7_invalidate_l1 calls in all headsmp.S in platform specific directories.
quoted
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
Ray, I don't have complete/other patch sets for this change. It would
be good if I get those patch sets as well or complete e-mail thread.
I think if we have a cleaner solutions for SMP, we can consolidate
the change required for NS and NSP. I have few points to add, which
are inline in this e-mail.
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Are you sure this is an issue of unpredictable L1 cache contents at
power-up? AFAIK, 5301X had an issue with secondary core initialization.
Secondary core which waits on WFE would let it out of the pen as soon as the
first spin_*lock executes. This was because of a BOOTROM bug in NS, so the
work around was to reset the address for the secondary processor to go back
and wait for the signal from the primary core. This vector fixup is required
so that the secondary core doesn't start executing kernel instructions until
we've patched its jump address during wakeup_secondary().
Also, v7 setup function should invalidate L1 cache and we should remove all
v7_invalidate_l1 calls in all headsmp.S in platform specific directories.
quoted
quoted
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
From: Russell King - ARM Linux <hidden> Date: 2015-10-15 08:17:47
On Wed, Oct 14, 2015 at 12:29:19AM +0200, Hauke Mehrtens wrote:
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Things have changed in this area: ARMv7 CPU support now conforms with
other Linux CPU support, and we invalidate the caches prior to enabling
them.
However, I still come back to the point I said above, which you have not
addressed. If the L1 cache contains unpredictable contents at power up,
how can you guarnatee that any code the CPU executes _is read by the CPU_
rather than the random contents of the L1 cache.
This isn't about the pen stuff. It's about whether you can execute any
code reliably at this point.
There are several possibilities that the ARM ARM allows:
* the caches are reset at CPU reset and contain no valid or dirty lines
* the caches are not reset at CPU reset, but are not searched until
the enable bits are set
* the caches are not reset at CPU reset, but are searched
The last case is the "special" case that requires a implementation
specific code to initialise them safely - but anyone implementing that
would be really silly to do so, so I doubt this is what you have.
So, I'd just get rid of that unnecessary cache flushing, especially as
I've said above, the ARMv7 CPU entry has been fixed to invalidate caches,
rather than flush dirty data (hence potentially random data in the second
case above) out to memory.
--
FTTC broadband for 0.8mile line: currently at 9.6Mbps down 400kbps up
according to speedtest.net.
From: Kapil Hali <hidden> Date: 2015-10-15 15:50:57
On 10/14/2015 11:52 PM, Hauke Mehrtens wrote:
On 10/14/2015 03:42 PM, Kapil Hali wrote:
quoted
On 10/14/2015 4:18 AM, Ray Jui wrote:
quoted
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
Ray, I don't have complete/other patch sets for this change. It would
be good if I get those patch sets as well or complete e-mail thread.
I think if we have a cleaner solutions for SMP, we can consolidate
the change required for NS and NSP. I have few points to add, which
are inline in this e-mail.
You are right, we can use the common SMP code for NS and NSP, however, I am
only concerned about the secondary_startup() function in case of NS as is
being discussed in the e-mail chain.
As there are multiple versions of NS SoC, and AFAIK, some of those have
anomalous BOOTROM. We should have a cleaner generic solution which can be
used for both NS and NSP.
Hauke
quoted
quoted
On 10/13/2015 3:29 PM, Hauke Mehrtens wrote:
quoted
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Are you sure this is an issue of unpredictable L1 cache contents at
power-up? AFAIK, 5301X had an issue with secondary core initialization.
Secondary core which waits on WFE would let it out of the pen as soon as the
first spin_*lock executes. This was because of a BOOTROM bug in NS, so the
work around was to reset the address for the secondary processor to go back
and wait for the signal from the primary core. This vector fixup is required
so that the secondary core doesn't start executing kernel instructions until
we've patched its jump address during wakeup_secondary().
Also, v7 setup function should invalidate L1 cache and we should remove all
v7_invalidate_l1 calls in all headsmp.S in platform specific directories.
quoted
quoted
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
From: Jon Mason <hidden> Date: 2015-10-22 20:30:13
On Wed, Oct 14, 2015 at 08:22:54PM +0200, Hauke Mehrtens wrote:
On 10/14/2015 03:42 PM, Kapil Hali wrote:
quoted
On 10/14/2015 4:18 AM, Ray Jui wrote:
quoted
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
Ray, I don't have complete/other patch sets for this change. It would
be good if I get those patch sets as well or complete e-mail thread.
I think if we have a cleaner solutions for SMP, we can consolidate
the change required for NS and NSP. I have few points to add, which
are inline in this e-mail.
Hello Hauke,
With my last patch (https://lkml.org/lkml/2015/10/15/690), SMP on 4708
is working with the NSP SMP patch
(https://lkml.org/lkml/2015/10/14/769). That patch does not have the
issues that Russell King was mentioning in this patch series (and
supports more SoCs). Have you had a chance to verify that it works
for you?
Thanks,
Jon
Hauke
quoted
quoted
On 10/13/2015 3:29 PM, Hauke Mehrtens wrote:
quoted
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Are you sure this is an issue of unpredictable L1 cache contents at
power-up? AFAIK, 5301X had an issue with secondary core initialization.
Secondary core which waits on WFE would let it out of the pen as soon as the
first spin_*lock executes. This was because of a BOOTROM bug in NS, so the
work around was to reset the address for the secondary processor to go back
and wait for the signal from the primary core. This vector fixup is required
so that the secondary core doesn't start executing kernel instructions until
we've patched its jump address during wakeup_secondary().
Also, v7 setup function should invalidate L1 cache and we should remove all
v7_invalidate_l1 calls in all headsmp.S in platform specific directories.
quoted
quoted
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke
On Wed, Oct 14, 2015 at 08:22:54PM +0200, Hauke Mehrtens wrote:
quoted
On 10/14/2015 03:42 PM, Kapil Hali wrote:
quoted
On 10/14/2015 4:18 AM, Ray Jui wrote:
quoted
+ bcm-kernel-feedback-list.
Kapil, you might want to take a look at this. Not sure how this is
related to your SMP patches for NSP.
Ray, I don't have complete/other patch sets for this change. It would
be good if I get those patch sets as well or complete e-mail thread.
I think if we have a cleaner solutions for SMP, we can consolidate
the change required for NS and NSP. I have few points to add, which
are inline in this e-mail.
Hello Hauke,
With my last patch (https://lkml.org/lkml/2015/10/15/690), SMP on 4708
is working with the NSP SMP patch
(https://lkml.org/lkml/2015/10/14/769). That patch does not have the
issues that Russell King was mentioning in this patch series (and
supports more SoCs). Have you had a chance to verify that it works
for you?
Hi,
I tested this patch on my device now.
What does the loader do before Linux gets started on the second CPU and
what is ensured?
On 03/26/2015 01:00 PM, Russell King - ARM Linux wrote:
quoted
On Sun, Mar 22, 2015 at 02:20:15PM +0100, Rafa? Mi?ecki wrote:
quoted
+/*
+ * BCM5301X specific entry point for secondary CPUs.
+ */
+ENTRY(bcm5301x_secondary_startup)
+ mrc p15, 0, r0, c0, c0, 5
+ and r0, r0, #15
+ adr r4, 1f
+ ldmia r4, {r5, r6}
+ sub r4, r4, r5
+ add r6, r6, r4
+pen: ldr r7, [r6]
+ cmp r7, r0
+ bne pen
+
+ /*
+ * In case L1 cache has unpredictable contents at power-up
+ * clean its contents without flushing.
+ */
+ bl v7_invalidate_l1
+
+ mov r0, #0
+ mcr p15, 0, r0, c7, c5, 0 /* Invalidate icache */
+ dsb
+ isb
So if your I-cache contains unpredictable contents, how do you execute
the code to this point? Shouldn't the I-cache invalidate be the very
first instruction you execute followed by the dsb and isb (oh, and iirc
it ignores the value in the register).
In the case where a CPU has unpredictable contents at power up, the
ARM ARM requires that an implementation specific sequence is followed
to initialise the caches. I doubt that such a sequence includes testing
a pen value.
Are you sure this is an issue of unpredictable L1 cache contents at
power-up? AFAIK, 5301X had an issue with secondary core initialization.
Secondary core which waits on WFE would let it out of the pen as soon as the
first spin_*lock executes. This was because of a BOOTROM bug in NS, so the
work around was to reset the address for the secondary processor to go back
and wait for the signal from the primary core. This vector fixup is required
so that the secondary core doesn't start executing kernel instructions until
we've patched its jump address during wakeup_secondary().
Also, v7 setup function should invalidate L1 cache and we should remove all
v7_invalidate_l1 calls in all headsmp.S in platform specific directories.
quoted
quoted
When I remove the test for the pen value the CPU does not come up any
more, I get this log output:
[ 0.132292] CPU: Testing write buffer coherency: ok
[ 0.137635] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143675] Setting up static identity map for 0x82a0 - 0x82d4
[ 10.149786] CPU1: failed to boot: -38
[ 10.153651] Brought up 1 CPUs
This was caused just by removing the "cmp r7, r0" and "bne pen"
instructions.
With these instructions are added it works and I get this:
[ 0.132329] CPU: Testing write buffer coherency: ok
[ 0.137682] CPU0: thread -1, cpu 0, socket 0, mpidr 80000000
[ 0.143708] Setting up static identity map for 0x82a0 - 0x82d4
[ 0.189788] CPU1: thread -1, cpu 1, socket 0, mpidr 80000001
[ 0.189892] Brought up 2 CPUs
[ 0.198889] SMP: Total of 2 processors activated (3188.32 BogoMIPS).
Currently this is 100% reproducible.
Could it be that the second CPU needs some time till it is synchronised
correctly?
I do not know if and why the cache clearing is needed, I do not have
access to the SoC documentation or the ASIC/firmware developer.
Which WFI? This seems to imply that you have some kind of initial
firmware. If so, that should be taking care of the cache initialisation,
not the kernel.
quoted
+ mb(); /* make sure write buffer is drained */
writel() already ensures that.
quoted
+ /*
+ * The secondary processor is waiting to be released from
+ * the holding pen - release it, then wait for it to flag
+ * that it has been released by resetting pen_release.
+ *
+ * Note that "pen_release" is the hardware CPU ID, whereas
+ * "cpu" is Linux's internal ID.
+ */
+ write_pen_release(cpu_logical_map(cpu));
+
+ /* Send the secondary CPU SEV */
+ dsb_sev();
If you even need any of the pen code, if you're having to send a SEV here,
wouldn't having a WFE in the pen assembly loop above be a good idea?
I have to read more on how WFE and co works.
Hauke