This series is ver.4 of page table walker patchset.
I reflected comments in the previous version (thanks Kirill, Jerome).
And rebased onto v3.16-rc3. Recently code around queue_pages_range()
is changed by commit d05f0cdcbe63 "mm: fix crashes from mbind() merging
vma", which affects this series a little.
Thanks,
Naoya Horiguchi
Tree: git@github.com:Naoya-Horiguchi/linux.git
Branch: v3.16-rc3/page_table_walker.ver4
---
Summary:
Kirill A. Shutemov (1):
mm: /proc/pid/clear_refs: avoid split_huge_page()
Naoya Horiguchi (12):
mm/pagewalk: remove pgd_entry() and pud_entry()
pagewalk: improve vma handling
pagewalk: add walk_page_vma()
smaps: remove mem_size_stats->vma and use walk_page_vma()
clear_refs: remove clear_refs_private->vma and introduce clear_refs_test_walk()
pagemap: use walk->vma instead of calling find_vma()
numa_maps: fix typo in gather_hugetbl_stats
numa_maps: remove numa_maps->vma
memcg: cleanup preparation for page table walk
arch/powerpc/mm/subpage-prot.c: use walk->vma and walk_page_vma()
mempolicy: apply page table walker on queue_pages_range()
mincore: apply page table walker on do_mincore()
arch/powerpc/mm/subpage-prot.c | 6 +-
fs/proc/task_mmu.c | 150 ++++++++++++++++-----------
include/linux/mm.h | 22 ++--
mm/huge_memory.c | 20 ----
mm/memcontrol.c | 49 +++------
mm/mempolicy.c | 224 ++++++++++++++++------------------------
mm/mincore.c | 173 +++++++++++--------------------
mm/pagewalk.c | 228 ++++++++++++++++++++++++-----------------
8 files changed, 409 insertions(+), 463 deletions(-)
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
Page table walker has the information of the current vma in mm_walk, so
we don't have to call find_vma() in each pagemap_hugetlb_range() call.
NULL-vma check is omitted because we assume that we never run hugetlb_entry()
callback on the address without vma. And even if it were broken, null pointer
dereference would be detected, so we can get enough information for debugging.
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
fs/proc/task_mmu.c | 7 ++-----
1 file changed, 2 insertions(+), 5 deletions(-)
@@ -1080,15 +1080,12 @@ static int pagemap_hugetlb_range(pte_t *pte, unsigned long hmask,structmm_walk*walk){structpagemapread*pm=walk->private;-structvm_area_struct*vma;+structvm_area_struct*vma=walk->vma;interr=0;intflags2;pagemap_entry_tpme;-vma=find_vma(walk->mm,addr);-WARN_ON_ONCE(!vma);--if(vma&&(vma->vm_flags&VM_SOFTDIRTY))+if(vma->vm_flags&VM_SOFTDIRTY)flags2=__PM_SOFT_DIRTY;elseflags2=0;
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
clear_refs_write() has some prechecks to determine if we really walk over
a given vma. Now we have a test_walk() callback to filter vmas, so let's
utilize it.
ChangeLog v4:
- use walk_page_range instead of walk_page_vma with for loop
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
fs/proc/task_mmu.c | 54 ++++++++++++++++++++++++++----------------------------
1 file changed, 26 insertions(+), 28 deletions(-)
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
@@ -1347,7 +1347,7 @@ static int gather_pte_stats(pmd_t *pmd, unsigned long addr,return0;}#ifdef CONFIG_HUGETLB_PAGE-staticintgather_hugetbl_stats(pte_t*pte,unsignedlonghmask,+staticintgather_hugetlb_stats(pte_t*pte,unsignedlonghmask,unsignedlongaddr,unsignedlongend,structmm_walk*walk){structnuma_maps*md;
@@ -1366,7 +1366,7 @@ static int gather_hugetbl_stats(pte_t *pte, unsigned long hmask,}#else-staticintgather_hugetbl_stats(pte_t*pte,unsignedlonghmask,+staticintgather_hugetlb_stats(pte_t*pte,unsignedlonghmask,unsignedlongaddr,unsignedlongend,structmm_walk*walk){return0;
@@ -1398,7 +1398,7 @@ static int show_numa_map(struct seq_file *m, void *v, int is_pid)md->vma=vma;-walk.hugetlb_entry=gather_hugetbl_stats;+walk.hugetlb_entry=gather_hugetlb_stats;walk.pmd_entry=gather_pte_stats;walk.private=md;walk.mm=mm;
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
pagewalk.c can handle vma in itself, so we don't have to pass vma via
walk->private. And show_numa_map() walks pages on vma basis, so using
walk_page_vma() is preferable.
ChangeLog v4:
- remove redundant vma
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
fs/proc/task_mmu.c | 29 +++++++++++++----------------
1 file changed, 13 insertions(+), 16 deletions(-)
@@ -1337,7 +1335,7 @@ static int gather_pte_stats(pmd_t *pmd, unsigned long addr,return0;orig_pte=pte=pte_offset_map_lock(walk->mm,pmd,addr,&ptl);do{-structpage*page=can_gather_numa_stats(*pte,md->vma,addr);+structpage*page=can_gather_numa_stats(*pte,vma,addr);if(!page)continue;gather_stats(page,md,pte_dirty(*pte),1);
@@ -1385,7 +1383,12 @@ static int show_numa_map(struct seq_file *m, void *v, int is_pid)structfile*file=vma->vm_file;structtask_struct*task=proc_priv->task;structmm_struct*mm=vma->vm_mm;-structmm_walkwalk={};+structmm_walkwalk={+.hugetlb_entry=gather_hugetlb_stats,+.pmd_entry=gather_pte_stats,+.private=md,+.mm=mm,+};structmempolicy*pol;charbuffer[64];intnid;
@@ -1396,13 +1399,6 @@ static int show_numa_map(struct seq_file *m, void *v, int is_pid)/* Ensure we start with an empty set of numa_maps statistics. */memset(md,0,sizeof(*md));-md->vma=vma;--walk.hugetlb_entry=gather_hugetlb_stats;-walk.pmd_entry=gather_pte_stats;-walk.private=md;-walk.mm=mm;-pol=get_vma_policy(task,vma,vma->vm_start);mpol_to_str(buffer,sizeof(buffer),pol);mpol_cond_put(pol);
@@ -1432,7 +1428,8 @@ static int show_numa_map(struct seq_file *m, void *v, int is_pid)if(is_vm_hugetlb_page(vma))seq_puts(m," huge");-walk_page_range(vma->vm_start,vma->vm_end,&walk);+/* mmap_sem is held by m_start */+walk_page_vma(vma,&walk);if(!md->pages)gotoout;
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
We don't have to use mm_walk->private to pass vma to the callback function
because of mm_walk->vma. And walk_page_vma() is useful if we walk over a
single vma.
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
arch/powerpc/mm/subpage-prot.c | 6 ++----
1 file changed, 2 insertions(+), 4 deletions(-)
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
This patch makes do_mincore() use walk_page_vma(), which reduces many lines
of code by using common page table walk code.
ChangeLog v4:
- remove redundant vma
ChangeLog v3:
- add NULL vma check in mincore_unmapped_range()
- don't use pte_entry()
ChangeLog v2:
- change type of args of callbacks to void *
- move definition of mincore_walk to the start of the function to fix compiler
warning
Signed-off-by: Naoya Horiguchi <redacted>
---
mm/huge_memory.c | 20 -------
mm/mincore.c | 173 ++++++++++++++++++++-----------------------------------
2 files changed, 62 insertions(+), 131 deletions(-)
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
From: "Kirill A. Shutemov" <redacted>
Currently pagewalker splits all THP pages on any clear_refs request. It's
not necessary. We can handle this on PMD level.
One side effect is that soft dirty will potentially see more dirty memory,
since we will mark whole THP page dirty at once.
Sanity checked with CRIU test suite. More testing is required.
ChangeLog:
- move code for thp to clear_refs_pte_range()
Signed-off-by: Kirill A. Shutemov <redacted>
Cc: Pavel Emelyanov <redacted>
Cc: Andrea Arcangeli <redacted>
Cc: Dave Hansen <redacted>
Signed-off-by: Naoya Horiguchi <redacted>
Cc: Cyrill Gorcunov <redacted>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
---
fs/proc/task_mmu.c | 47 ++++++++++++++++++++++++++++++++++++++++++++---
1 file changed, 44 insertions(+), 3 deletions(-)
@@ -752,7 +778,22 @@ static int clear_refs_pte_range(pmd_t *pmd, unsigned long addr,spinlock_t*ptl;structpage*page;-split_huge_page_pmd(vma,addr,pmd);+if(pmd_trans_huge_lock(pmd,vma,&ptl)==1){+if(cp->type==CLEAR_REFS_SOFT_DIRTY){+clear_soft_dirty_pmd(vma,addr,pmd);+gotoout;+}++page=pmd_page(*pmd);++/* Clear accessed and referenced bits. */+pmdp_test_and_clear_young(vma,addr,pmd);+ClearPageReferenced(page);+out:+spin_unlock(ptl);+return0;+}+if(pmd_trans_unstable(pmd))return0;
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
queue_pages_range() does page table walking in its own way now, but there
is some code duplicate. This patch applies page table walker to reduce
lines of code.
queue_pages_range() has to do some precheck to determine whether we really
walk over the vma or just skip it. Now we have test_walk() callback in
mm_walk for this purpose, so we can do this replacement cleanly.
queue_pages_test_walk() depends on not only the current vma but also the
previous one, so queue_pages->prev is introduced to remember it.
ChangeLog v4:
- rebase to v3.16-rc3, where the return value of queue_pages_range()
becomes 0 in success instead of the first found vma, and use -EFAILT
instead of ERR_PTR() in failure.
Signed-off-by: Naoya Horiguchi <redacted>
---
mm/mempolicy.c | 224 +++++++++++++++++++++++----------------------------------
1 file changed, 90 insertions(+), 134 deletions(-)
@@ -509,114 +519,46 @@ static int queue_pages_pte_range(struct vm_area_struct *vma, pmd_t *pmd,if(PageReserved(page))continue;nid=page_to_nid(page);-if(node_isset(nid,*nodes)==!!(flags&MPOL_MF_INVERT))+if(node_isset(nid,*qp->nmask)==!!(flags&MPOL_MF_INVERT))continue;if(flags&(MPOL_MF_MOVE|MPOL_MF_MOVE_ALL))-migrate_page_add(page,private,flags);-else-break;-}while(pte++,addr+=PAGE_SIZE,addr!=end);-pte_unmap_unlock(orig_pte,ptl);-returnaddr!=end;+migrate_page_add(page,qp->pagelist,flags);+}+pte_unmap_unlock(pte-1,ptl);+cond_resched();+return0;}-staticvoidqueue_pages_hugetlb_pmd_range(structvm_area_struct*vma,-pmd_t*pmd,constnodemask_t*nodes,unsignedlongflags,-void*private)+staticintqueue_pages_hugetlb(pte_t*pte,unsignedlonghmask,+unsignedlongaddr,unsignedlongend,+structmm_walk*walk){#ifdef CONFIG_HUGETLB_PAGE+structqueue_pages*qp=walk->private;+unsignedlongflags=qp->flags;intnid;structpage*page;spinlock_t*ptl;pte_tentry;-ptl=huge_pte_lock(hstate_vma(vma),vma->vm_mm,(pte_t*)pmd);-entry=huge_ptep_get((pte_t*)pmd);+ptl=huge_pte_lock(hstate_vma(walk->vma),walk->mm,pte);+entry=huge_ptep_get(pte);if(!pte_present(entry))gotounlock;page=pte_page(entry);nid=page_to_nid(page);-if(node_isset(nid,*nodes)==!!(flags&MPOL_MF_INVERT))+if(node_isset(nid,*qp->nmask)==!!(flags&MPOL_MF_INVERT))gotounlock;/* With MPOL_MF_MOVE, we migrate only unshared hugepage. */if(flags&(MPOL_MF_MOVE_ALL)||(flags&MPOL_MF_MOVE&&page_mapcount(page)==1))-isolate_huge_page(page,private);+isolate_huge_page(page,qp->pagelist);unlock:spin_unlock(ptl);#elseBUG();#endif-}--staticinlineintqueue_pages_pmd_range(structvm_area_struct*vma,pud_t*pud,-unsignedlongaddr,unsignedlongend,-constnodemask_t*nodes,unsignedlongflags,-void*private)-{-pmd_t*pmd;-unsignedlongnext;--pmd=pmd_offset(pud,addr);-do{-next=pmd_addr_end(addr,end);-if(!pmd_present(*pmd))-continue;-if(pmd_huge(*pmd)&&is_vm_hugetlb_page(vma)){-queue_pages_hugetlb_pmd_range(vma,pmd,nodes,-flags,private);-continue;-}-split_huge_page_pmd(vma,addr,pmd);-if(pmd_none_or_trans_huge_or_clear_bad(pmd))-continue;-if(queue_pages_pte_range(vma,pmd,addr,next,nodes,-flags,private))-return-EIO;-}while(pmd++,addr=next,addr!=end);-return0;-}--staticinlineintqueue_pages_pud_range(structvm_area_struct*vma,pgd_t*pgd,-unsignedlongaddr,unsignedlongend,-constnodemask_t*nodes,unsignedlongflags,-void*private)-{-pud_t*pud;-unsignedlongnext;--pud=pud_offset(pgd,addr);-do{-next=pud_addr_end(addr,end);-if(pud_huge(*pud)&&is_vm_hugetlb_page(vma))-continue;-if(pud_none_or_clear_bad(pud))-continue;-if(queue_pages_pmd_range(vma,pud,addr,next,nodes,-flags,private))-return-EIO;-}while(pud++,addr=next,addr!=end);-return0;-}--staticinlineintqueue_pages_pgd_range(structvm_area_struct*vma,-unsignedlongaddr,unsignedlongend,-constnodemask_t*nodes,unsignedlongflags,-void*private)-{-pgd_t*pgd;-unsignedlongnext;--pgd=pgd_offset(vma->vm_mm,addr);-do{-next=pgd_addr_end(addr,end);-if(pgd_none_or_clear_bad(pgd))-continue;-if(queue_pages_pud_range(vma,pgd,addr,next,nodes,-flags,private))-return-EIO;-}while(pgd++,addr=next,addr!=end);return0;}
@@ -649,6 +591,44 @@ static unsigned long change_prot_numa(struct vm_area_struct *vma,}#endif /* CONFIG_NUMA_BALANCING */+staticintqueue_pages_test_walk(unsignedlongstart,unsignedlongend,+structmm_walk*walk)+{+structvm_area_struct*vma=walk->vma;+structqueue_pages*qp=walk->private;+unsignedlongendvma=vma->vm_end;+unsignedlongflags=qp->flags;++if(endvma>end)+endvma=end;+if(vma->vm_start>start)+start=vma->vm_start;++if(!(flags&MPOL_MF_DISCONTIG_OK)){+if(!vma->vm_next&&vma->vm_end<end)+return-EFAULT;+if(qp->prev&&qp->prev->vm_end<vma->vm_start)+return-EFAULT;+}++qp->prev=vma;++if(vma->vm_flags&VM_PFNMAP)+return1;++if(flags&MPOL_MF_LAZY){+change_prot_numa(vma,start,endvma);+return1;+}++if((flags&MPOL_MF_STRICT)||+((flags&(MPOL_MF_MOVE|MPOL_MF_MOVE_ALL))&&+vma_migratable(vma)))+/* queue pages from current vma */+return0;+return1;+}+/**Walkthroughpagetablesandcollectpagestobemigrated.*
@@ -658,48 +638,24 @@ static unsigned long change_prot_numa(struct vm_area_struct *vma,*/staticintqueue_pages_range(structmm_struct*mm,unsignedlongstart,unsignedlongend,-constnodemask_t*nodes,unsignedlongflags,void*private)-{-interr=0;-structvm_area_struct*vma,*prev;--vma=find_vma(mm,start);-if(!vma)-return-EFAULT;-prev=NULL;-for(;vma&&vma->vm_start<end;vma=vma->vm_next){-unsignedlongendvma=vma->vm_end;--if(endvma>end)-endvma=end;-if(vma->vm_start>start)-start=vma->vm_start;--if(!(flags&MPOL_MF_DISCONTIG_OK)){-if(!vma->vm_next&&vma->vm_end<end)-return-EFAULT;-if(prev&&prev->vm_end<vma->vm_start)-return-EFAULT;-}--if(flags&MPOL_MF_LAZY){-change_prot_numa(vma,start,endvma);-gotonext;-}--if((flags&MPOL_MF_STRICT)||-((flags&(MPOL_MF_MOVE|MPOL_MF_MOVE_ALL))&&-vma_migratable(vma))){--err=queue_pages_pgd_range(vma,start,endvma,nodes,-flags,private);-if(err)-break;-}-next:-prev=vma;-}-returnerr;+nodemask_t*nodes,unsignedlongflags,+structlist_head*pagelist)+{+structqueue_pagesqp={+.pagelist=pagelist,+.flags=flags,+.nmask=nodes,+.prev=NULL,+};+structmm_walkqueue_pages_walk={+.hugetlb_entry=queue_pages_hugetlb,+.pmd_entry=queue_pages_pte_range,+.test_walk=queue_pages_test_walk,+.mm=mm,+.private=&qp,+};++returnwalk_page_range(start,end,&queue_pages_walk);}/*
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
Current implementation of page table walker has a fundamental problem
in vma handling, which started when we tried to handle vma(VM_HUGETLB).
Because it's done in pgd loop, considering vma boundary makes code
complicated and bug-prone.
From the users viewpoint, some user checks some vma-related condition to
determine whether the user really does page walk over the vma.
In order to solve these, this patch moves vma check outside pgd loop and
introduce a new callback ->test_walk().
ChangeLog v4:
- avoid walking over the regions where vma is NULL if pte_hole() is
undefined
- use vma->vm_next instead of repeating find_vma()
- use min() in walk_page_range "outside vma" branch
- fix return value of walk_hugetlb_range()
ChangeLog v3:
- drop walk->skip control
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
include/linux/mm.h | 15 +++-
mm/pagewalk.c | 203 ++++++++++++++++++++++++++++++-----------------------
2 files changed, 129 insertions(+), 89 deletions(-)
@@ -1107,10 +1107,16 @@ void unmap_vmas(struct mmu_gather *tlb, struct vm_area_struct *start_vma,*@pte_entry:ifset,calledforeachnon-emptyPTE(4th-level)entry*@pte_hole:ifset,calledforeachholeatalllevels*@hugetlb_entry:ifset,calledforeachhugetlbentry-**Caution*:Thecallermustholdmmap_sem()if@hugetlb_entry-*isused.+*@test_walk:callerspecificcallbackfunctiontodeterminewhether+*wewalkoverthecurrentvmaornot.Apositivereturned+*valuemeans"do page table walk over the current vma,"+*andanegativeonemeans"abort current page table walk+*rightnow." 0 means "skipthecurrentvma."+*@mm:mm_structrepresentingthetargetprocessofpagetablewalk+*@vma:vmacurrentlywalked(NULLifwalkingoutsidevmas)+*@private:privatedataforcallbacks'usage*-*(seewalk_page_rangeformoredetails)+*(seethecommentonwalk_page_range()formoredetails)*/structmm_walk{int(*pmd_entry)(pmd_t*pmd,unsignedlongaddr,
@@ -59,7 +59,7 @@ static int walk_pmd_range(pud_t *pud, unsigned long addr, unsigned long end,continue;split_huge_page_pmd_mm(walk->mm,addr,pmd);-if(pmd_none_or_trans_huge_or_clear_bad(pmd))+if(pmd_trans_unstable(pmd))gotoagain;err=walk_pte_range(pmd,addr,next,walk);if(err)
@@ -95,6 +95,32 @@ static int walk_pud_range(pgd_t *pgd, unsigned long addr, unsigned long end,returnerr;}+staticintwalk_pgd_range(unsignedlongaddr,unsignedlongend,+structmm_walk*walk)+{+pgd_t*pgd;+unsignedlongnext;+interr=0;++pgd=pgd_offset(walk->mm,addr);+do{+next=pgd_addr_end(addr,end);+if(pgd_none_or_clear_bad(pgd)){+if(walk->pte_hole)+err=walk->pte_hole(addr,next,walk);+if(err)+break;+continue;+}+if(walk->pmd_entry||walk->pte_entry)+err=walk_pud_range(pgd,addr,next,walk);+if(err)+break;+}while(pgd++,addr=next,addr!=end);++returnerr;+}+#ifdef CONFIG_HUGETLB_PAGEstaticunsignedlonghugetlb_entry_end(structhstate*h,unsignedlongaddr,unsignedlongend)
@@ -103,10 +129,10 @@ static unsigned long hugetlb_entry_end(struct hstate *h, unsigned long addr,returnboundary<end?boundary:end;}-staticintwalk_hugetlb_range(structvm_area_struct*vma,-unsignedlongaddr,unsignedlongend,+staticintwalk_hugetlb_range(unsignedlongaddr,unsignedlongend,structmm_walk*walk){+structvm_area_struct*vma=walk->vma;structhstate*h=hstate_vma(vma);unsignedlongnext;unsignedlonghmask=huge_page_mask(h);
@@ -135,109 +160,115 @@ static int walk_hugetlb_range(struct vm_area_struct *vma,#endif /* CONFIG_HUGETLB_PAGE */+/*+*Decidewhetherwereallywalkoverthecurrentvmaon[@start,@end)+*orskipitviathereturnedvalue.Return0ifwedowalkoverthe+*currentvma,andreturn1ifweskipthevma.Negativevaluesmeans+*error,whereweabortthecurrentwalk.+*+*Defaultcheck(onlyVM_PFNMAPcheckfornow)isusedwhenthecaller+*doesn'tdefinetest_walk()callback.+*/+staticintwalk_page_test(unsignedlongstart,unsignedlongend,+structmm_walk*walk)+{+structvm_area_struct*vma=walk->vma;+if(walk->test_walk)+returnwalk->test_walk(start,end,walk);++/*+*Donotwalkovervma(VM_PFNMAP),becausewehavenovalidstruct+*pagebackingaVM_PFNMAPrange.Seealsocommita9ff785e4437.+*/+if(vma->vm_flags&VM_PFNMAP)+return1;+return0;+}++staticint__walk_page_range(unsignedlongstart,unsignedlongend,+structmm_walk*walk)+{+interr=0;+structvm_area_struct*vma=walk->vma;++if(vma&&is_vm_hugetlb_page(vma)){+if(walk->hugetlb_entry)+err=walk_hugetlb_range(start,end,walk);+}else+err=walk_pgd_range(start,end,walk);++returnerr;+}/**-*walk_page_range-walkamemorymap'spagetableswithacallback-*@addr:startingaddress-*@end:endingaddress-*@walk:setofcallbackstoinvokeforeachlevelofthetree-*-*RecursivelywalkthepagetableforthememoryareainaVMA,-*callingsuppliedcallbacks.Callbacksarecalledin-order(first-*PGD,firstPUD,firstPMD,firstPTE,secondPTE...secondPMD,-*etc.).Iflower-levelcallbacksareomitted,walkingdepthisreduced.+*walk_page_range-walkpagetablewithcallerspecificcallbacks*-*Eachcallbackreceivesanentrypointerandthestartandendofthe-*associatedrange,andacopyoftheoriginalmm_walkforaccessto-*the->privateor->mmfields.+*Recursivelywalkthepagetabletreeoftheprocessrepresentedby@walk->mm+*withinthevirtualaddressrange[@start,@end).Duringwalking,wecando+*somecaller-specificworksforeachentry,bysettinguppmd_entry(),+*pte_entry(),and/orhugetlb_entry().Ifyoudon'tsetupforsomeofthese+*callbacks,theassociatedentries/pagesarejustignored.+*Thereturnvaluesofthesecallbacksarecommonlydefinedlikebelow:+*-0:succeededtohandlethecurrententry,andifyoudon'treachthe+*endaddressyet,continuetowalk.+*->0:succeededtohandlethecurrententry,andreturntothecaller+*withcallerspecificvalue.+*-<0:failedtohandlethecurrententry,andreturntothecaller+*witherrorcode.*-*Usuallynolocksaretaken,butsplittingtransparenthugepagemay-*takepagetablelock.AndthebottomleveliteratorwillmapPTE-*directoriesfromhighmemifnecessary.+*Beforestartingtowalkpagetable,somecallerswanttocheckwhether+*theyreallywanttowalkoverthecurrentvma,typicallybychecking+*itsvm_flags.walk_page_test()and@walk->test_walk()areusedforthis+*purpose.*-*Ifanycallbackreturnsanon-zerovalue,thewalkisabortedand-*thereturnvalueispropagatedbacktothecaller.Otherwise0isreturned.+*structmm_walkkeepscurrentvaluesofsomecommondatalikevmaandpmd,+*whichareusefulfortheaccessfromcallbacks.Ifyouwanttopasssome+*caller-specificdatatocallbacks,@walk->privateshouldbehelpful.*-*walk->mm->mmap_semmustbeheldforatleastreadifwalk->hugetlb_entry-*is!NULL.+*Locking:+*Callersofwalk_page_range()andwalk_page_vma()shouldhold+*@walk->mm->mmap_sem,becausethesefunctiontraversevmalistand/or+*accesstovma'sdata.*/-intwalk_page_range(unsignedlongaddr,unsignedlongend,+intwalk_page_range(unsignedlongstart,unsignedlongend,structmm_walk*walk){-pgd_t*pgd;-unsignedlongnext;interr=0;+unsignedlongnext;+structvm_area_struct*vma;-if(addr>=end)-returnerr;+if(start>=end)+return-EINVAL;if(!walk->mm)return-EINVAL;VM_BUG_ON(!rwsem_is_locked(&walk->mm->mmap_sem));-pgd=pgd_offset(walk->mm,addr);+vma=find_vma(walk->mm,start);do{-structvm_area_struct*vma=NULL;+if(!vma){/* after the last vma */+walk->vma=NULL;+next=end;+}elseif(start<vma->vm_start){/* outside vma */+walk->vma=NULL;+next=min(end,vma->vm_start);+}else{/* inside vma */+walk->vma=vma;+next=min(end,vma->vm_end);+vma=vma->vm_next;-next=pgd_addr_end(addr,end);--/*-*Thisfunctionwasnotintendedtobevmabased.-*Buttherearevmaspecialcasestobehandled:-*-hugetlbvma's-*-VM_PFNMAPvma's-*/-vma=find_vma(walk->mm,addr);-if(vma){-/*-*TherearenopagestructuresbackingaVM_PFNMAP-*range,sodonotallowsplit_huge_page_pmd().-*/-if((vma->vm_start<=addr)&&-(vma->vm_flags&VM_PFNMAP)){-next=vma->vm_end;-pgd=pgd_offset(walk->mm,next);+err=walk_page_test(start,next,walk);+if(err>0)continue;-}-/*-*Handlehugetlbvmaindividuallybecausepagetable-*walkforthehugetlbpageisdependentonthe-*architectureandwecan'thandleditinthesame-*mannerasnon-hugepages.-*/-if(walk->hugetlb_entry&&(vma->vm_start<=addr)&&-is_vm_hugetlb_page(vma)){-if(vma->vm_end<next)-next=vma->vm_end;-/*-*Hugepageisverytightlycoupledwithvma,-*sowalkthroughhugetlbentrieswithina-*givenvma.-*/-err=walk_hugetlb_range(vma,addr,next,walk);-if(err)-break;-pgd=pgd_offset(walk->mm,next);-continue;-}-}--if(pgd_none_or_clear_bad(pgd)){-if(walk->pte_hole)-err=walk->pte_hole(addr,next,walk);-if(err)+if(err<0)break;-pgd++;-continue;}-if(walk->pmd_entry||walk->pte_entry)-err=walk_pud_range(pgd,addr,next,walk);+if(walk->vma||walk->pte_hole)+err=__walk_page_range(start,next,walk);if(err)break;-pgd++;-}while(addr=next,addr<end);-+}while(start=next,start<end);returnerr;}
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
pagewalk.c can handle vma in itself, so we don't have to pass vma via
walk->private. And show_smap() walks pages on vma basis, so using
walk_page_vma() is preferable.
ChangeLog v4:
- remove redundant vma
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
fs/proc/task_mmu.c | 9 +++------
1 file changed, 3 insertions(+), 6 deletions(-)
@@ -501,7 +500,7 @@ static int smaps_pte_range(pmd_t *pmd, unsigned long addr, unsigned long end,structmm_walk*walk){structmem_size_stats*mss=walk->private;-structvm_area_struct*vma=mss->vma;+structvm_area_struct*vma=walk->vma;pte_t*pte;spinlock_t*ptl;
@@ -594,10 +593,8 @@ static int show_smap(struct seq_file *m, void *v, int is_pid)};memset(&mss,0,sizeofmss);-mss.vma=vma;/* mmap_sem is held in m_start */-if(vma->vm_mm&&!is_vm_hugetlb_page(vma))-walk_page_range(vma->vm_start,vma->vm_end,&smaps_walk);+walk_page_vma(vma,&smaps_walk);show_map_vma(m,vma,is_pid);
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
pagewalk.c can handle vma in itself, so we don't have to pass vma via
walk->private. And both of mem_cgroup_count_precharge() and
mem_cgroup_move_charge() do for each vma loop themselves, but now it's
done in pagewalk.c, so let's clean up them.
ChangeLog v4:
- use walk_page_range() instead of walk_page_vma() with for loop.
Signed-off-by: Naoya Horiguchi <redacted>
---
mm/memcontrol.c | 49 ++++++++++++++++---------------------------------
1 file changed, 16 insertions(+), 33 deletions(-)
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
Introduces walk_page_vma(), which is useful for the callers which want to
walk over a given vma. It's used by later patches.
ChangeLog v3:
- check walk_page_test's return value instead of walk->skip
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
include/linux/mm.h | 1 +
mm/pagewalk.c | 18 ++++++++++++++++++
2 files changed, 19 insertions(+)
@@ -272,3 +272,21 @@ int walk_page_range(unsigned long start, unsigned long end,}while(start=next,start<end);returnerr;}++intwalk_page_vma(structvm_area_struct*vma,structmm_walk*walk)+{+interr;++if(!walk->mm)+return-EINVAL;++VM_BUG_ON(!rwsem_is_locked(&walk->mm->mmap_sem));+VM_BUG_ON(!vma);+walk->vma=vma;+err=walk_page_test(vma->vm_start,vma->vm_end,walk);+if(err>0)+return0;+if(err<0)+returnerr;+return__walk_page_range(vma->vm_start,vma->vm_end,walk);+}
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
Currently no user of page table walker sets ->pgd_entry() or ->pud_entry(),
so checking their existence in each loop is just wasting CPU cycle.
So let's remove it to reduce overhead.
Signed-off-by: Naoya Horiguchi <redacted>
Acked-by: Kirill A. Shutemov <redacted>
---
include/linux/mm.h | 6 ------
mm/pagewalk.c | 9 ++-------
2 files changed, 2 insertions(+), 13 deletions(-)
@@ -86,9 +86,7 @@ static int walk_pud_range(pgd_t *pgd, unsigned long addr, unsigned long end,break;continue;}-if(walk->pud_entry)-err=walk->pud_entry(pud,addr,next,walk);-if(!err&&(walk->pmd_entry||walk->pte_entry))+if(walk->pmd_entry||walk->pte_entry)err=walk_pmd_range(pud,addr,next,walk);if(err)break;
@@ -234,10 +232,7 @@ int walk_page_range(unsigned long addr, unsigned long end,pgd++;continue;}-if(walk->pgd_entry)-err=walk->pgd_entry(pgd,addr,next,walk);-if(!err&&-(walk->pud_entry||walk->pmd_entry||walk->pte_entry))+if(walk->pmd_entry||walk->pte_entry)err=walk_pud_range(pgd,addr,next,walk);if(err)break;
--
1.9.3
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
From: Dave Hansen <hidden> Date: 2014-07-01 21:00:40
On 07/01/2014 10:07 AM, Naoya Horiguchi wrote:
queue_pages_range() does page table walking in its own way now, but there
is some code duplicate. This patch applies page table walker to reduce
lines of code.
queue_pages_range() has to do some precheck to determine whether we really
walk over the vma or just skip it. Now we have test_walk() callback in
mm_walk for this purpose, so we can do this replacement cleanly.
queue_pages_test_walk() depends on not only the current vma but also the
previous one, so queue_pages->prev is introduced to remember it.
Hi Naoya,
The previous version of this patch caused a performance regression which
was reported to you:
http://marc.info/?l=linux-kernel&m=140375975525069&w=2
Has that been dealt with in this version somehow?
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
On Tue, Jul 01, 2014 at 02:00:32PM -0700, Dave Hansen wrote:
On 07/01/2014 10:07 AM, Naoya Horiguchi wrote:
quoted
queue_pages_range() does page table walking in its own way now, but there
is some code duplicate. This patch applies page table walker to reduce
lines of code.
queue_pages_range() has to do some precheck to determine whether we really
walk over the vma or just skip it. Now we have test_walk() callback in
mm_walk for this purpose, so we can do this replacement cleanly.
queue_pages_test_walk() depends on not only the current vma but also the
previous one, so queue_pages->prev is introduced to remember it.
I believe so, in previous version we called ->pte_entry() callback
for each pte entries, but in this version I stop doing this and
most of works are done in ->pmd_entry() callback, so the number
of function calls are reduced by about 1/512. And rather than that,
I just cleaned up queue_pages_* without major behavioral changes, so
the visible regression should be solved.
Thanks,
Naoya Horiguchi
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
From: Kirill A. Shutemov <hidden> Date: 2014-07-09 13:35:00
On Tue, Jul 01, 2014 at 01:07:31PM -0400, Naoya Horiguchi wrote:
This patch makes do_mincore() use walk_page_vma(), which reduces many lines
of code by using common page table walk code.
ChangeLog v4:
- remove redundant vma
ChangeLog v3:
- add NULL vma check in mincore_unmapped_range()
- don't use pte_entry()
ChangeLog v2:
- change type of args of callbacks to void *
- move definition of mincore_walk to the start of the function to fix compiler
warning
Signed-off-by: Naoya Horiguchi <redacted>
On Wed, Jul 09, 2014 at 04:34:36PM +0300, Kirill A. Shutemov wrote:
On Tue, Jul 01, 2014 at 01:07:31PM -0400, Naoya Horiguchi wrote:
quoted
This patch makes do_mincore() use walk_page_vma(), which reduces many lines
of code by using common page table walk code.
ChangeLog v4:
- remove redundant vma
ChangeLog v3:
- add NULL vma check in mincore_unmapped_range()
- don't use pte_entry()
ChangeLog v2:
- change type of args of callbacks to void *
- move definition of mincore_walk to the start of the function to fix compiler
warning
Signed-off-by: Naoya Horiguchi <redacted>
Trinity crases this implementation of mincore pretty easily:
[ 42.775369] BUG: unable to handle kernel paging request at ffff88007bb61000
[ 42.776656] IP: [<ffffffff81126f8f>] mincore_unmapped_range+0xdf/0x100
Thanks for your testing/reporting.
...
Looks like 'vec' overflow. I don't see what could prevent do_mincore() to
write more than PAGE_SIZE to 'vec'.
I found the miscalculation of walk->private (vec) on thp and hugetlbfs.
I confirmed that the reported problem is fixed (I checked that trinity
never triggers the reported BUG) with the following changes on this patch.
@@ -34,7 +34,7 @@ static int mincore_hugetlb(pte_t *pte, unsigned long hmask, unsigned long addr,present=pte&&!huge_pte_none(huge_ptep_get(pte));for(;addr!=end;vec++,addr+=PAGE_SIZE)*vec=present;-walk->private+=(end-addr)>>PAGE_SHIFT;+walk->private=vec;#elseBUG();#endif
@@ -118,8 +118,10 @@ static int mincore_pte_range(pmd_t *pmd, unsigned long addr, unsigned long end,return0;}-if(pmd_trans_unstable(pmd))+if(pmd_trans_unstable(pmd)){+walk->private+=(end-addr)>>PAGE_SHIFT;return0;+}ptep=pte_offset_map_lock(walk->mm,pmd,addr,&ptl);for(;addr!=end;ptep++,addr+=PAGE_SIZE){
Thanks,
Naoya Horiguchi
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
From: Kirill A. Shutemov <hidden> Date: 2014-07-10 10:06:22
On Wed, Jul 09, 2014 at 05:36:24PM -0400, Naoya Horiguchi wrote:
On Wed, Jul 09, 2014 at 04:34:36PM +0300, Kirill A. Shutemov wrote:
quoted
On Tue, Jul 01, 2014 at 01:07:31PM -0400, Naoya Horiguchi wrote:
quoted
This patch makes do_mincore() use walk_page_vma(), which reduces many lines
of code by using common page table walk code.
ChangeLog v4:
- remove redundant vma
ChangeLog v3:
- add NULL vma check in mincore_unmapped_range()
- don't use pte_entry()
ChangeLog v2:
- change type of args of callbacks to void *
- move definition of mincore_walk to the start of the function to fix compiler
warning
Signed-off-by: Naoya Horiguchi <redacted>
Trinity crases this implementation of mincore pretty easily:
[ 42.775369] BUG: unable to handle kernel paging request at ffff88007bb61000
[ 42.776656] IP: [<ffffffff81126f8f>] mincore_unmapped_range+0xdf/0x100
Thanks for your testing/reporting.
...
quoted
Looks like 'vec' overflow. I don't see what could prevent do_mincore() to
write more than PAGE_SIZE to 'vec'.
I found the miscalculation of walk->private (vec) on thp and hugetlbfs.
I confirmed that the reported problem is fixed (I checked that trinity
never triggers the reported BUG) with the following changes on this patch.
From: Kirill A. Shutemov <hidden> Date: 2014-07-10 11:32:40
On Tue, Jul 01, 2014 at 01:07:23PM -0400, Naoya Horiguchi wrote:
quoted hunk
@@ -822,38 +844,14 @@ static ssize_t clear_refs_write(struct file *file, const char __user *buf, }; struct mm_walk clear_refs_walk = { .pmd_entry = clear_refs_pte_range,+ .test_walk = clear_refs_test_walk, .mm = mm, .private = &cp, }; down_read(&mm->mmap_sem); if (type == CLEAR_REFS_SOFT_DIRTY) mmu_notifier_invalidate_range_start(mm, 0, -1);- for (vma = mm->mmap; vma; vma = vma->vm_next) {- cp.vma = vma;- if (is_vm_hugetlb_page(vma))- continue;- /*- * Writing 1 to /proc/pid/clear_refs affects all pages.- *- * Writing 2 to /proc/pid/clear_refs only affects- * Anonymous pages.- *- * Writing 3 to /proc/pid/clear_refs only affects file- * mapped pages.- *- * Writing 4 to /proc/pid/clear_refs affects all pages.- */- if (type == CLEAR_REFS_ANON && vma->vm_file)- continue;- if (type == CLEAR_REFS_MAPPED && !vma->vm_file)- continue;- if (type == CLEAR_REFS_SOFT_DIRTY) {- if (vma->vm_flags & VM_SOFTDIRTY)- vma->vm_flags &= ~VM_SOFTDIRTY;- }- walk_page_range(vma->vm_start, vma->vm_end,- &clear_refs_walk);- }+ walk_page_range(0, ~0UL, &clear_refs_walk);
'vma' variable is now unused in the clear_refs_write().
--
Kirill A. Shutemov
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
On Thu, Jul 10, 2014 at 02:32:19PM +0300, Kirill A. Shutemov wrote:
On Tue, Jul 01, 2014 at 01:07:23PM -0400, Naoya Horiguchi wrote:
quoted
@@ -822,38 +844,14 @@ static ssize_t clear_refs_write(struct file *file, const char __user *buf, }; struct mm_walk clear_refs_walk = { .pmd_entry = clear_refs_pte_range,+ .test_walk = clear_refs_test_walk, .mm = mm, .private = &cp, }; down_read(&mm->mmap_sem); if (type == CLEAR_REFS_SOFT_DIRTY) mmu_notifier_invalidate_range_start(mm, 0, -1);- for (vma = mm->mmap; vma; vma = vma->vm_next) {- cp.vma = vma;- if (is_vm_hugetlb_page(vma))- continue;- /*- * Writing 1 to /proc/pid/clear_refs affects all pages.- *- * Writing 2 to /proc/pid/clear_refs only affects- * Anonymous pages.- *- * Writing 3 to /proc/pid/clear_refs only affects file- * mapped pages.- *- * Writing 4 to /proc/pid/clear_refs affects all pages.- */- if (type == CLEAR_REFS_ANON && vma->vm_file)- continue;- if (type == CLEAR_REFS_MAPPED && !vma->vm_file)- continue;- if (type == CLEAR_REFS_SOFT_DIRTY) {- if (vma->vm_flags & VM_SOFTDIRTY)- vma->vm_flags &= ~VM_SOFTDIRTY;- }- walk_page_range(vma->vm_start, vma->vm_end,- &clear_refs_walk);- }+ walk_page_range(0, ~0UL, &clear_refs_walk);
'vma' variable is now unused in the clear_refs_write().
Yes, will remove it.
Naoya
--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>
On Thu, Jul 10, 2014 at 01:06:00PM +0300, Kirill A. Shutemov wrote:
On Wed, Jul 09, 2014 at 05:36:24PM -0400, Naoya Horiguchi wrote:
quoted
On Wed, Jul 09, 2014 at 04:34:36PM +0300, Kirill A. Shutemov wrote:
quoted
On Tue, Jul 01, 2014 at 01:07:31PM -0400, Naoya Horiguchi wrote:
quoted
This patch makes do_mincore() use walk_page_vma(), which reduces many lines
of code by using common page table walk code.
ChangeLog v4:
- remove redundant vma
ChangeLog v3:
- add NULL vma check in mincore_unmapped_range()
- don't use pte_entry()
ChangeLog v2:
- change type of args of callbacks to void *
- move definition of mincore_walk to the start of the function to fix compiler
warning
Signed-off-by: Naoya Horiguchi <redacted>
Trinity crases this implementation of mincore pretty easily:
[ 42.775369] BUG: unable to handle kernel paging request at ffff88007bb61000
[ 42.776656] IP: [<ffffffff81126f8f>] mincore_unmapped_range+0xdf/0x100
Thanks for your testing/reporting.
...
quoted
Looks like 'vec' overflow. I don't see what could prevent do_mincore() to
write more than PAGE_SIZE to 'vec'.
I found the miscalculation of walk->private (vec) on thp and hugetlbfs.
I confirmed that the reported problem is fixed (I checked that trinity
never triggers the reported BUG) with the following changes on this patch.
I don't do it explicitly, so adding it is one solution.
But I think the problem comes from using walk_page_range() instead of
walk_page_vma() which forcibly sets the walk range from vm->vm_start to
vm->vm_end.
As the original code does, limiting the range to [addr, addr + pages <<
PAGE_SHIFT) is fine because it implicitly prevents buffer overflow.
Here is the revised fix for this patch. Please remove the one I replied
yesterday because it was wrong.
Thanks,
Naoya Horiguchi
---
@@ -34,7 +34,7 @@ static int mincore_hugetlb(pte_t *pte, unsigned long hmask, unsigned long addr,present=pte&&!huge_pte_none(huge_ptep_get(pte));for(;addr!=end;vec++,addr+=PAGE_SIZE)*vec=present;-walk->private+=(end-addr)>>PAGE_SHIFT;+walk->private=vec;#elseBUG();#endif
@@ -118,8 +118,10 @@ static int mincore_pte_range(pmd_t *pmd, unsigned long addr, unsigned long end,return0;}-if(pmd_trans_unstable(pmd))+if(pmd_trans_unstable(pmd)){+mincore_unmapped_range(addr,end,walk);return0;+}ptep=pte_offset_map_lock(walk->mm,pmd,addr,&ptl);for(;addr!=end;ptep++,addr+=PAGE_SIZE){
@@ -168,6 +170,7 @@ static int mincore_pte_range(pmd_t *pmd, unsigned long addr, unsigned long end,staticlongdo_mincore(unsignedlongaddr,unsignedlongpages,unsignedchar*vec){structvm_area_struct*vma;+unsignedlongend;interr;structmm_walkmincore_walk={.pmd_entry=mincore_pte_range,
@@ -180,16 +183,11 @@ static long do_mincore(unsigned long addr, unsigned long pages, unsigned char *vif(!vma||addr<vma->vm_start)return-ENOMEM;mincore_walk.mm=vma->vm_mm;--err=walk_page_vma(vma,&mincore_walk);+end=min(vma->vm_end,addr+(pages<<PAGE_SHIFT));+err=walk_page_range(addr,end,&mincore_walk);if(err<0)returnerr;-else{-unsignedlongend;--end=min(vma->vm_end,addr+(pages<<PAGE_SHIFT));-return(end-addr)>>PAGE_SHIFT;-}+return(end-addr)>>PAGE_SHIFT;}/*--
To unsubscribe, send a message with 'unsubscribe linux-mm' in
the body to majordomo@kvack.org. For more info on Linux MM,
see: http://www.linux-mm.org/ .
Don't email: <a href=mailto:"dont@kvack.org"> email@kvack.org </a>