diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2014-08-04 20:15:45 -0400 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2014-08-04 20:15:45 -0400 |
| commit | ce4747963252a30613ebf1c1df3d83b9526a342e (patch) | |
| tree | 6c61d1b1045a72965006324ae3805280be296e53 | |
| parent | 76f09aa464a1913efd596dd0edbf88f932fde08c (diff) | |
| parent | a5102476a24bce364b74f1110005542a2c964103 (diff) | |
Merge branch 'x86-mm-for-linus' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip
Pull x86 mm changes from Ingo Molnar:
"The main change in this cycle is the rework of the TLB range flushing
code, to simplify, fix and consolidate the code. By Dave Hansen"
* 'x86-mm-for-linus' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip:
x86/mm: Set TLB flush tunable to sane value (33)
x86/mm: New tunable for single vs full TLB flush
x86/mm: Add tracepoints for TLB flushes
x86/mm: Unify remote INVLPG code
x86/mm: Fix missed global TLB flush stat
x86/mm: Rip out complicated, out-of-date, buggy TLB flushing
x86/mm: Clean up the TLB flushing code
x86/smep: Be more informative when signalling an SMEP fault
| -rw-r--r-- | Documentation/x86/tlb.txt | 75 | ||||
| -rw-r--r-- | arch/x86/include/asm/mmu_context.h | 6 | ||||
| -rw-r--r-- | arch/x86/include/asm/processor.h | 1 | ||||
| -rw-r--r-- | arch/x86/kernel/cpu/amd.c | 7 | ||||
| -rw-r--r-- | arch/x86/kernel/cpu/common.c | 13 | ||||
| -rw-r--r-- | arch/x86/kernel/cpu/intel.c | 26 | ||||
| -rw-r--r-- | arch/x86/mm/fault.c | 6 | ||||
| -rw-r--r-- | arch/x86/mm/init.c | 7 | ||||
| -rw-r--r-- | arch/x86/mm/tlb.c | 103 | ||||
| -rw-r--r-- | include/linux/mm_types.h | 8 | ||||
| -rw-r--r-- | include/trace/events/tlb.h | 40 |
11 files changed, 193 insertions, 99 deletions
diff --git a/Documentation/x86/tlb.txt b/Documentation/x86/tlb.txt new file mode 100644 index 000000000000..2b3a82e69151 --- /dev/null +++ b/Documentation/x86/tlb.txt | |||
| @@ -0,0 +1,75 @@ | |||
| 1 | When the kernel unmaps or modified the attributes of a range of | ||
| 2 | memory, it has two choices: | ||
| 3 | 1. Flush the entire TLB with a two-instruction sequence. This is | ||
| 4 | a quick operation, but it causes collateral damage: TLB entries | ||
| 5 | from areas other than the one we are trying to flush will be | ||
| 6 | destroyed and must be refilled later, at some cost. | ||
| 7 | 2. Use the invlpg instruction to invalidate a single page at a | ||
| 8 | time. This could potentialy cost many more instructions, but | ||
| 9 | it is a much more precise operation, causing no collateral | ||
| 10 | damage to other TLB entries. | ||
| 11 | |||
| 12 | Which method to do depends on a few things: | ||
| 13 | 1. The size of the flush being performed. A flush of the entire | ||
| 14 | address space is obviously better performed by flushing the | ||
| 15 | entire TLB than doing 2^48/PAGE_SIZE individual flushes. | ||
| 16 | 2. The contents of the TLB. If the TLB is empty, then there will | ||
| 17 | be no collateral damage caused by doing the global flush, and | ||
| 18 | all of the individual flush will have ended up being wasted | ||
| 19 | work. | ||
| 20 | 3. The size of the TLB. The larger the TLB, the more collateral | ||
| 21 | damage we do with a full flush. So, the larger the TLB, the | ||
| 22 | more attrative an individual flush looks. Data and | ||
| 23 | instructions have separate TLBs, as do different page sizes. | ||
| 24 | 4. The microarchitecture. The TLB has become a multi-level | ||
| 25 | cache on modern CPUs, and the global flushes have become more | ||
| 26 | expensive relative to single-page flushes. | ||
| 27 | |||
| 28 | There is obviously no way the kernel can know all these things, | ||
| 29 | especially the contents of the TLB during a given flush. The | ||
| 30 | sizes of the flush will vary greatly depending on the workload as | ||
| 31 | well. There is essentially no "right" point to choose. | ||
| 32 | |||
| 33 | You may be doing too many individual invalidations if you see the | ||
| 34 | invlpg instruction (or instructions _near_ it) show up high in | ||
| 35 | profiles. If you believe that individual invalidations being | ||
| 36 | called too often, you can lower the tunable: | ||
| 37 | |||
| 38 | /sys/debug/kernel/x86/tlb_single_page_flush_ceiling | ||
| 39 | |||
| 40 | This will cause us to do the global flush for more cases. | ||
| 41 | Lowering it to 0 will disable the use of the individual flushes. | ||
| 42 | Setting it to 1 is a very conservative setting and it should | ||
| 43 | never need to be 0 under normal circumstances. | ||
| 44 | |||
| 45 | Despite the fact that a single individual flush on x86 is | ||
| 46 | guaranteed to flush a full 2MB [1], hugetlbfs always uses the full | ||
| 47 | flushes. THP is treated exactly the same as normal memory. | ||
| 48 | |||
| 49 | You might see invlpg inside of flush_tlb_mm_range() show up in | ||
| 50 | profiles, or you can use the trace_tlb_flush() tracepoints. to | ||
| 51 | determine how long the flush operations are taking. | ||
| 52 | |||
| 53 | Essentially, you are balancing the cycles you spend doing invlpg | ||
| 54 | with the cycles that you spend refilling the TLB later. | ||
| 55 | |||
| 56 | You can measure how expensive TLB refills are by using | ||
| 57 | performance counters and 'perf stat', like this: | ||
| 58 | |||
| 59 | perf stat -e | ||
| 60 | cpu/event=0x8,umask=0x84,name=dtlb_load_misses_walk_duration/, | ||
| 61 | cpu/event=0x8,umask=0x82,name=dtlb_load_misses_walk_completed/, | ||
| 62 | cpu/event=0x49,umask=0x4,name=dtlb_store_misses_walk_duration/, | ||
| 63 | cpu/event=0x49,umask=0x2,name=dtlb_store_misses_walk_completed/, | ||
| 64 | cpu/event=0x85,umask=0x4,name=itlb_misses_walk_duration/, | ||
| 65 | cpu/event=0x85,umask=0x2,name=itlb_misses_walk_completed/ | ||
| 66 | |||
| 67 | That works on an IvyBridge-era CPU (i5-3320M). Different CPUs | ||
| 68 | may have differently-named counters, but they should at least | ||
| 69 | be there in some form. You can use pmu-tools 'ocperf list' | ||
| 70 | (https://github.com/andikleen/pmu-tools) to find the right | ||
| 71 | counters for a given CPU. | ||
| 72 | |||
| 73 | 1. A footnote in Intel's SDM "4.10.4.2 Recommended Invalidation" | ||
| 74 | says: "One execution of INVLPG is sufficient even for a page | ||
| 75 | with size greater than 4 KBytes." | ||
diff --git a/arch/x86/include/asm/mmu_context.h b/arch/x86/include/asm/mmu_context.h index be12c534fd59..166af2a8e865 100644 --- a/arch/x86/include/asm/mmu_context.h +++ b/arch/x86/include/asm/mmu_context.h | |||
| @@ -3,6 +3,10 @@ | |||
| 3 | 3 | ||
| 4 | #include <asm/desc.h> | 4 | #include <asm/desc.h> |
| 5 | #include <linux/atomic.h> | 5 | #include <linux/atomic.h> |
| 6 | #include <linux/mm_types.h> | ||
| 7 | |||
| 8 | #include <trace/events/tlb.h> | ||
| 9 | |||
| 6 | #include <asm/pgalloc.h> | 10 | #include <asm/pgalloc.h> |
| 7 | #include <asm/tlbflush.h> | 11 | #include <asm/tlbflush.h> |
| 8 | #include <asm/paravirt.h> | 12 | #include <asm/paravirt.h> |
| @@ -44,6 +48,7 @@ static inline void switch_mm(struct mm_struct *prev, struct mm_struct *next, | |||
| 44 | 48 | ||
| 45 | /* Re-load page tables */ | 49 | /* Re-load page tables */ |
| 46 | load_cr3(next->pgd); | 50 | load_cr3(next->pgd); |
| 51 | trace_tlb_flush(TLB_FLUSH_ON_TASK_SWITCH, TLB_FLUSH_ALL); | ||
| 47 | 52 | ||
| 48 | /* Stop flush ipis for the previous mm */ | 53 | /* Stop flush ipis for the previous mm */ |
| 49 | cpumask_clear_cpu(cpu, mm_cpumask(prev)); | 54 | cpumask_clear_cpu(cpu, mm_cpumask(prev)); |
| @@ -71,6 +76,7 @@ static inline void switch_mm(struct mm_struct *prev, struct mm_struct *next, | |||
| 71 | * to make sure to use no freed page tables. | 76 | * to make sure to use no freed page tables. |
| 72 | */ | 77 | */ |
| 73 | load_cr3(next->pgd); | 78 | load_cr3(next->pgd); |
| 79 | trace_tlb_flush(TLB_FLUSH_ON_TASK_SWITCH, TLB_FLUSH_ALL); | ||
| 74 | load_LDT_nolock(&next->context); | 80 | load_LDT_nolock(&next->context); |
| 75 | } | 81 | } |
| 76 | } | 82 | } |
diff --git a/arch/x86/include/asm/processor.h b/arch/x86/include/asm/processor.h index 32cc237f8e20..ee30b9f0b91c 100644 --- a/arch/x86/include/asm/processor.h +++ b/arch/x86/include/asm/processor.h | |||
| @@ -72,7 +72,6 @@ extern u16 __read_mostly tlb_lld_4k[NR_INFO]; | |||
| 72 | extern u16 __read_mostly tlb_lld_2m[NR_INFO]; | 72 | extern u16 __read_mostly tlb_lld_2m[NR_INFO]; |
| 73 | extern u16 __read_mostly tlb_lld_4m[NR_INFO]; | 73 | extern u16 __read_mostly tlb_lld_4m[NR_INFO]; |
| 74 | extern u16 __read_mostly tlb_lld_1g[NR_INFO]; | 74 | extern u16 __read_mostly tlb_lld_1g[NR_INFO]; |
| 75 | extern s8 __read_mostly tlb_flushall_shift; | ||
| 76 | 75 | ||
| 77 | /* | 76 | /* |
| 78 | * CPU type and hardware bug flags. Kept separately for each CPU. | 77 | * CPU type and hardware bug flags. Kept separately for each CPU. |
diff --git a/arch/x86/kernel/cpu/amd.c b/arch/x86/kernel/cpu/amd.c index bc360d3df60e..60e5497681f5 100644 --- a/arch/x86/kernel/cpu/amd.c +++ b/arch/x86/kernel/cpu/amd.c | |||
| @@ -724,11 +724,6 @@ static unsigned int amd_size_cache(struct cpuinfo_x86 *c, unsigned int size) | |||
| 724 | } | 724 | } |
| 725 | #endif | 725 | #endif |
| 726 | 726 | ||
| 727 | static void cpu_set_tlb_flushall_shift(struct cpuinfo_x86 *c) | ||
| 728 | { | ||
| 729 | tlb_flushall_shift = 6; | ||
| 730 | } | ||
| 731 | |||
| 732 | static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) | 727 | static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) |
| 733 | { | 728 | { |
| 734 | u32 ebx, eax, ecx, edx; | 729 | u32 ebx, eax, ecx, edx; |
| @@ -776,8 +771,6 @@ static void cpu_detect_tlb_amd(struct cpuinfo_x86 *c) | |||
| 776 | tlb_lli_2m[ENTRIES] = eax & mask; | 771 | tlb_lli_2m[ENTRIES] = eax & mask; |
| 777 | 772 | ||
| 778 | tlb_lli_4m[ENTRIES] = tlb_lli_2m[ENTRIES] >> 1; | 773 | tlb_lli_4m[ENTRIES] = tlb_lli_2m[ENTRIES] >> 1; |
| 779 | |||
| 780 | cpu_set_tlb_flushall_shift(c); | ||
| 781 | } | 774 | } |
| 782 | 775 | ||
| 783 | static const struct cpu_dev amd_cpu_dev = { | 776 | static const struct cpu_dev amd_cpu_dev = { |
diff --git a/arch/x86/kernel/cpu/common.c b/arch/x86/kernel/cpu/common.c index 188a8c5cc094..333fd5209336 100644 --- a/arch/x86/kernel/cpu/common.c +++ b/arch/x86/kernel/cpu/common.c | |||
| @@ -481,26 +481,17 @@ u16 __read_mostly tlb_lld_2m[NR_INFO]; | |||
| 481 | u16 __read_mostly tlb_lld_4m[NR_INFO]; | 481 | u16 __read_mostly tlb_lld_4m[NR_INFO]; |
| 482 | u16 __read_mostly tlb_lld_1g[NR_INFO]; | 482 | u16 __read_mostly tlb_lld_1g[NR_INFO]; |
| 483 | 483 | ||
| 484 | /* | ||
| 485 | * tlb_flushall_shift shows the balance point in replacing cr3 write | ||
| 486 | * with multiple 'invlpg'. It will do this replacement when | ||
| 487 | * flush_tlb_lines <= active_lines/2^tlb_flushall_shift. | ||
| 488 | * If tlb_flushall_shift is -1, means the replacement will be disabled. | ||
| 489 | */ | ||
| 490 | s8 __read_mostly tlb_flushall_shift = -1; | ||
| 491 | |||
| 492 | void cpu_detect_tlb(struct cpuinfo_x86 *c) | 484 | void cpu_detect_tlb(struct cpuinfo_x86 *c) |
| 493 | { | 485 | { |
| 494 | if (this_cpu->c_detect_tlb) | 486 | if (this_cpu->c_detect_tlb) |
