diff options
| author | Linus Torvalds <torvalds@linux-foundation.org> | 2012-05-23 20:08:40 -0400 |
|---|---|---|
| committer | Linus Torvalds <torvalds@linux-foundation.org> | 2012-05-23 20:08:40 -0400 |
| commit | c80ddb526331a72c9e9d1480f85f6fd7c74e3d2d (patch) | |
| tree | 0212803a009f171990032abb94fad84156baa153 | |
| parent | 2c13bc0f8f0d3e13b42be70bf74fec8e56b58324 (diff) | |
| parent | 1dff2b87a34a1ac1d1898ea109bf97ed396aca53 (diff) | |
Merge tag 'md-3.5' of git://neil.brown.name/md
Pull md updates from NeilBrown:
"It's been a busy cycle for md - lots of fun stuff here.. if you like
this kind of thing :-)
Main features:
- RAID10 arrays can be reshaped - adding and removing devices and
changing chunks (not 'far' array though)
- allow RAID5 arrays to be reshaped with a backup file (not tested
yet, but the priciple works fine for RAID10).
- arrays can be reshaped while a bitmap is present - you no longer
need to remove it first
- SSSE3 support for RAID6 syndrome calculations
and of course a number of minor fixes etc."
* tag 'md-3.5' of git://neil.brown.name/md: (56 commits)
md/bitmap: record the space available for the bitmap in the superblock.
md/raid10: Remove extras after reshape to smaller number of devices.
md/raid5: improve removal of extra devices after reshape.
md: check the return of mddev_find()
MD RAID1: Further conditionalize 'fullsync'
DM RAID: Use md_error() in place of simply setting Faulty bit
DM RAID: Record and handle missing devices
DM RAID: Set recovery flags on resume
md/raid5: Allow reshape while a bitmap is present.
md/raid10: resize bitmap when required during reshape.
md: allow array to be resized while bitmap is present.
md/bitmap: make sure reshape request are reflected in superblock.
md/bitmap: add bitmap_resize function to allow bitmap resizing.
md/bitmap: use DIV_ROUND_UP instead of open-code
md/bitmap: create a 'struct bitmap_counts' substructure of 'struct bitmap'
md/bitmap: make bitmap bitops atomic.
md/bitmap: make _page_attr bitops atomic.
md/bitmap: merge bitmap_file_unmap and bitmap_file_put.
md/bitmap: remove async freeing of bitmap file.
md/bitmap: convert some spin_lock_irqsave to spin_lock_irq
...
| -rw-r--r-- | arch/x86/Makefile | 5 | ||||
| -rw-r--r-- | arch/x86/include/asm/xor_32.h | 6 | ||||
| -rw-r--r-- | arch/x86/include/asm/xor_64.h | 8 | ||||
| -rw-r--r-- | arch/x86/include/asm/xor_avx.h | 214 | ||||
| -rw-r--r-- | crypto/xor.c | 13 | ||||
| -rw-r--r-- | drivers/md/bitmap.c | 1100 | ||||
| -rw-r--r-- | drivers/md/bitmap.h | 60 | ||||
| -rw-r--r-- | drivers/md/dm-raid.c | 22 | ||||
| -rw-r--r-- | drivers/md/md.c | 370 | ||||
| -rw-r--r-- | drivers/md/md.h | 12 | ||||
| -rw-r--r-- | drivers/md/raid1.c | 22 | ||||
| -rw-r--r-- | drivers/md/raid10.c | 1281 | ||||
| -rw-r--r-- | drivers/md/raid10.h | 34 | ||||
| -rw-r--r-- | drivers/md/raid5.c | 252 | ||||
| -rw-r--r-- | drivers/md/raid5.h | 7 | ||||
| -rw-r--r-- | include/linux/raid/md_p.h | 15 | ||||
| -rw-r--r-- | include/linux/raid/pq.h | 18 | ||||
| -rw-r--r-- | lib/raid6/Makefile | 2 | ||||
| -rw-r--r-- | lib/raid6/algos.c | 127 | ||||
| -rw-r--r-- | lib/raid6/mktables.c | 25 | ||||
| -rw-r--r-- | lib/raid6/recov.c | 15 | ||||
| -rw-r--r-- | lib/raid6/recov_ssse3.c | 335 | ||||
| -rw-r--r-- | lib/raid6/test/Makefile | 2 | ||||
| -rw-r--r-- | lib/raid6/test/test.c | 32 | ||||
| -rw-r--r-- | lib/raid6/x86.h | 15 |
25 files changed, 3124 insertions, 868 deletions
diff --git a/arch/x86/Makefile b/arch/x86/Makefile index dc611a40a336..1f2521434554 100644 --- a/arch/x86/Makefile +++ b/arch/x86/Makefile | |||
| @@ -115,9 +115,10 @@ cfi-sections := $(call as-instr,.cfi_sections .debug_frame,-DCONFIG_AS_CFI_SECTI | |||
| 115 | 115 | ||
| 116 | # does binutils support specific instructions? | 116 | # does binutils support specific instructions? |
| 117 | asinstr := $(call as-instr,fxsaveq (%rax),-DCONFIG_AS_FXSAVEQ=1) | 117 | asinstr := $(call as-instr,fxsaveq (%rax),-DCONFIG_AS_FXSAVEQ=1) |
| 118 | avx_instr := $(call as-instr,vxorps %ymm0$(comma)%ymm1$(comma)%ymm2,-DCONFIG_AS_AVX=1) | ||
| 118 | 119 | ||
| 119 | KBUILD_AFLAGS += $(cfi) $(cfi-sigframe) $(cfi-sections) $(asinstr) | 120 | KBUILD_AFLAGS += $(cfi) $(cfi-sigframe) $(cfi-sections) $(asinstr) $(avx_instr) |
| 120 | KBUILD_CFLAGS += $(cfi) $(cfi-sigframe) $(cfi-sections) $(asinstr) | 121 | KBUILD_CFLAGS += $(cfi) $(cfi-sigframe) $(cfi-sections) $(asinstr) $(avx_instr) |
| 121 | 122 | ||
| 122 | LDFLAGS := -m elf_$(UTS_MACHINE) | 123 | LDFLAGS := -m elf_$(UTS_MACHINE) |
| 123 | 124 | ||
diff --git a/arch/x86/include/asm/xor_32.h b/arch/x86/include/asm/xor_32.h index 133b40a0f495..454570891bdc 100644 --- a/arch/x86/include/asm/xor_32.h +++ b/arch/x86/include/asm/xor_32.h | |||
| @@ -861,6 +861,9 @@ static struct xor_block_template xor_block_pIII_sse = { | |||
| 861 | .do_5 = xor_sse_5, | 861 | .do_5 = xor_sse_5, |
| 862 | }; | 862 | }; |
| 863 | 863 | ||
| 864 | /* Also try the AVX routines */ | ||
| 865 | #include "xor_avx.h" | ||
| 866 | |||
| 864 | /* Also try the generic routines. */ | 867 | /* Also try the generic routines. */ |
| 865 | #include <asm-generic/xor.h> | 868 | #include <asm-generic/xor.h> |
| 866 | 869 | ||
| @@ -871,6 +874,7 @@ do { \ | |||
| 871 | xor_speed(&xor_block_8regs_p); \ | 874 | xor_speed(&xor_block_8regs_p); \ |
| 872 | xor_speed(&xor_block_32regs); \ | 875 | xor_speed(&xor_block_32regs); \ |
| 873 | xor_speed(&xor_block_32regs_p); \ | 876 | xor_speed(&xor_block_32regs_p); \ |
| 877 | AVX_XOR_SPEED; \ | ||
| 874 | if (cpu_has_xmm) \ | 878 | if (cpu_has_xmm) \ |
| 875 | xor_speed(&xor_block_pIII_sse); \ | 879 | xor_speed(&xor_block_pIII_sse); \ |
| 876 | if (cpu_has_mmx) { \ | 880 | if (cpu_has_mmx) { \ |
| @@ -883,6 +887,6 @@ do { \ | |||
| 883 | We may also be able to load into the L1 only depending on how the cpu | 887 | We may also be able to load into the L1 only depending on how the cpu |
| 884 | deals with a load to a line that is being prefetched. */ | 888 | deals with a load to a line that is being prefetched. */ |
| 885 | #define XOR_SELECT_TEMPLATE(FASTEST) \ | 889 | #define XOR_SELECT_TEMPLATE(FASTEST) \ |
| 886 | (cpu_has_xmm ? &xor_block_pIII_sse : FASTEST) | 890 | AVX_SELECT(cpu_has_xmm ? &xor_block_pIII_sse : FASTEST) |
| 887 | 891 | ||
| 888 | #endif /* _ASM_X86_XOR_32_H */ | 892 | #endif /* _ASM_X86_XOR_32_H */ |
diff --git a/arch/x86/include/asm/xor_64.h b/arch/x86/include/asm/xor_64.h index 1549b5e261f6..b9b2323e90fe 100644 --- a/arch/x86/include/asm/xor_64.h +++ b/arch/x86/include/asm/xor_64.h | |||
| @@ -347,15 +347,21 @@ static struct xor_block_template xor_block_sse = { | |||
| 347 | .do_5 = xor_sse_5, | 347 | .do_5 = xor_sse_5, |
| 348 | }; | 348 | }; |
| 349 | 349 | ||
| 350 | |||
| 351 | /* Also try the AVX routines */ | ||
| 352 | #include "xor_avx.h" | ||
| 353 | |||
| 350 | #undef XOR_TRY_TEMPLATES | 354 | #undef XOR_TRY_TEMPLATES |
| 351 | #define XOR_TRY_TEMPLATES \ | 355 | #define XOR_TRY_TEMPLATES \ |
| 352 | do { \ | 356 | do { \ |
| 357 | AVX_XOR_SPEED; \ | ||
| 353 | xor_speed(&xor_block_sse); \ | 358 | xor_speed(&xor_block_sse); \ |
| 354 | } while (0) | 359 | } while (0) |
| 355 | 360 | ||
| 356 | /* We force the use of the SSE xor block because it can write around L2. | 361 | /* We force the use of the SSE xor block because it can write around L2. |
| 357 | We may also be able to load into the L1 only depending on how the cpu | 362 | We may also be able to load into the L1 only depending on how the cpu |
| 358 | deals with a load to a line that is being prefetched. */ | 363 | deals with a load to a line that is being prefetched. */ |
| 359 | #define XOR_SELECT_TEMPLATE(FASTEST) (&xor_block_sse) | 364 | #define XOR_SELECT_TEMPLATE(FASTEST) \ |
| 365 | AVX_SELECT(&xor_block_sse) | ||
| 360 | 366 | ||
| 361 | #endif /* _ASM_X86_XOR_64_H */ | 367 | #endif /* _ASM_X86_XOR_64_H */ |
diff --git a/arch/x86/include/asm/xor_avx.h b/arch/x86/include/asm/xor_avx.h new file mode 100644 index 000000000000..2510d35f480e --- /dev/null +++ b/arch/x86/include/asm/xor_avx.h | |||
| @@ -0,0 +1,214 @@ | |||
| 1 | #ifndef _ASM_X86_XOR_AVX_H | ||
| 2 | #define _ASM_X86_XOR_AVX_H | ||
| 3 | |||
| 4 | /* | ||
| 5 | * Optimized RAID-5 checksumming functions for AVX | ||
| 6 | * | ||
| 7 | * Copyright (C) 2012 Intel Corporation | ||
| 8 | * Author: Jim Kukunas <james.t.kukunas@linux.intel.com> | ||
| 9 | * | ||
| 10 | * Based on Ingo Molnar and Zach Brown's respective MMX and SSE routines | ||
| 11 | * | ||
| 12 | * This program is free software; you can redistribute it and/or | ||
| 13 | * modify it under the terms of the GNU General Public License | ||
| 14 | * as published by the Free Software Foundation; version 2 | ||
| 15 | * of the License. | ||
| 16 | */ | ||
| 17 | |||
| 18 | #ifdef CONFIG_AS_AVX | ||
| 19 | |||
| 20 | #include <linux/compiler.h> | ||
| 21 | #include <asm/i387.h> | ||
| 22 | |||
| 23 | #define ALIGN32 __aligned(32) | ||
| 24 | |||
| 25 | #define YMM_SAVED_REGS 4 | ||
| 26 | |||
| 27 | #define YMMS_SAVE \ | ||
| 28 | do { \ | ||
| 29 | preempt_disable(); \ | ||
| 30 | cr0 = read_cr0(); \ | ||
| 31 | clts(); \ | ||
| 32 | asm volatile("vmovaps %%ymm0, %0" : "=m" (ymm_save[0]) : : "memory"); \ | ||
| 33 | asm volatile("vmovaps %%ymm1, %0" : "=m" (ymm_save[32]) : : "memory"); \ | ||
| 34 | asm volatile("vmovaps %%ymm2, %0" : "=m" (ymm_save[64]) : : "memory"); \ | ||
| 35 | asm volatile("vmovaps %%ymm3, %0" : "=m" (ymm_save[96]) : : "memory"); \ | ||
| 36 | } while (0); | ||
| 37 | |||
| 38 | #define YMMS_RESTORE \ | ||
| 39 | do { \ | ||
| 40 | asm volatile("sfence" : : : "memory"); \ | ||
| 41 | asm volatile("vmovaps %0, %%ymm3" : : "m" (ymm_save[96])); \ | ||
| 42 | asm volatile("vmovaps %0, %%ymm2" : : "m" (ymm_save[64])); \ | ||
| 43 | asm volatile("vmovaps %0, %%ymm1" : : "m" (ymm_save[32])); \ | ||
| 44 | asm volatile("vmovaps %0, %%ymm0" : : "m" (ymm_save[0])); \ | ||
| 45 | write_cr0(cr0); \ | ||
| 46 | preempt_enable(); \ | ||
| 47 | } while (0); | ||
| 48 | |||
| 49 | #define BLOCK4(i) \ | ||
| 50 | BLOCK(32 * i, 0) \ | ||
| 51 | BLOCK(32 * (i + 1), 1) \ | ||
| 52 | BLOCK(32 * (i + 2), 2) \ | ||
| 53 | BLOCK(32 * (i + 3), 3) | ||
| 54 | |||
| 55 | #define BLOCK16() \ | ||
| 56 | BLOCK4(0) \ | ||
| 57 | BLOCK4(4) \ | ||
| 58 | BLOCK4(8) \ | ||
| 59 | BLOCK4(12) | ||
| 60 | |||
| 61 | static void xor_avx_2(unsigned long bytes, unsigned long *p0, unsigned long *p1) | ||
| 62 | { | ||
| 63 | unsigned long cr0, lines = bytes >> 9; | ||
| 64 | char ymm_save[32 * YMM_SAVED_REGS] ALIGN32; | ||
| 65 | |||
