| 1 |
// Copyright 2016 Ferry Toth, Exalon Delft BV, The Netherlands |
| 2 |
/* |
| 3 |
This software is provided 'as-is', without any express or implied |
| 4 |
warranty. In no event will the author be held liable for any damages |
| 5 |
arising from the use of this software. |
| 6 |
|
| 7 |
Permission is granted to anyone to use this software for any purpose, |
| 8 |
including commercial applications, and to alter it and redistribute it |
| 9 |
freely, subject to the following restrictions: |
| 10 |
|
| 11 |
1. The origin of this software must not be misrepresented; you must not |
| 12 |
claim that you wrote the original software. If you use this software |
| 13 |
in a product, an acknowledgment in the product documentation would be |
| 14 |
appreciated but is not required. |
| 15 |
2. Altered source versions must be plainly marked as such, and must not be |
| 16 |
misrepresented as being the original software. |
| 17 |
3. This notice may not be removed or altered from any source distribution. |
| 18 |
|
| 19 |
Ferry Toth |
| 20 |
ftoth@exalondelft.nl |
| 21 |
*/ |
| 22 |
|
| 23 |
/* Use hardware CRC instruction on Intel SSE 4.2 processors. This computes a |
| 24 |
CRC-32C, *not* the CRC-32 used by Ethernet and zip, gzip, etc. Where efficient |
| 25 |
3 crc32q instructions are used which a single core can execute in parallel. |
| 26 |
This compensates for the latency of a single crc32q instruction. Combining the |
| 27 |
3 CRC-32C bytes is done using the pclmulqdq instruction, which has overhead of |
| 28 |
its own, and makes this code path only efficient for buffer sizes above 216 bytes. |
| 29 |
All code requiring a crc32q instruction is done inside a macro, for which alternative |
| 30 |
code is generated in case of a 32 bit platform. |
| 31 |
|
| 32 |
This code is a port of Intels crc_iscsi_v_pcl.asm assembly code (which is part of |
| 33 |
this project as well as in a modified form the linux kernel) and reaches the same |
| 34 |
throughput on 64bit platforms. The main advantage of this port is that it was |
| 35 |
relatively easy to port to 32bit platforms (like Intel Edison which currently has |
| 36 |
only 32bit support). Being written in C it is of course easier to maintain and possibly |
| 37 |
optimize further */ |
| 38 |
|
| 39 |
/* Version history: |
| 40 |
1.0 07 May 2016 Ferry Toth - First version |
| 41 |
*/ |
| 42 |
|
| 43 |
#ifndef __LP64__ |
| 44 |
#define CRC_NATIVE uint32_t |
| 45 |
#else |
| 46 |
#define CRC_NATIVE uint64_t |
| 47 |
#endif |
| 48 |
|
| 49 |
#ifndef __LP64__ |
| 50 |
#define CRCtriplet(crc, buf, offset) \ |
| 51 |
crc ## 0 = __builtin_ia32_crc32si(crc ## 0, *((uint32_t*) buf ## 0 + 2 * offset)); \ |
| 52 |
crc ## 1 = __builtin_ia32_crc32si(crc ## 1, *((uint32_t*) buf ## 1 + 2 * offset)); \ |
| 53 |
crc ## 2 = __builtin_ia32_crc32si(crc ## 2, *((uint32_t*) buf ## 2 + 2 * offset)); \ |
| 54 |
crc ## 0 = __builtin_ia32_crc32si(crc ## 0, *((uint32_t*) buf ## 0 + 1 + 2 * offset)); \ |
| 55 |
crc ## 1 = __builtin_ia32_crc32si(crc ## 1, *((uint32_t*) buf ## 1 + 1 + 2 * offset)); \ |
| 56 |
crc ## 2 = __builtin_ia32_crc32si(crc ## 2, *((uint32_t*) buf ## 2 + 1 + 2 * offset)); |
| 57 |
#else |
| 58 |
#define CRCtriplet(crc, buf, offset) \ |
| 59 |
crc ## 0 = __builtin_ia32_crc32di(crc ## 0, *(buf ## 0 + offset)); \ |
| 60 |
crc ## 1 = __builtin_ia32_crc32di(crc ## 1, *(buf ## 1 + offset)); \ |
| 61 |
crc ## 2 = __builtin_ia32_crc32di(crc ## 2, *(buf ## 2 + offset)); |
| 62 |
#endif |
| 63 |
|
| 64 |
#ifndef __LP64__ |
| 65 |
#define CRCduplet(crc, buf, offset) \ |
| 66 |
crc ## 0 = __builtin_ia32_crc32si(crc ## 0, *((uint32_t*) buf ## 0 + 2 * offset)); \ |
| 67 |
crc ## 1 = __builtin_ia32_crc32si(crc ## 1, *((uint32_t*) buf ## 1 + 2 * offset)); \ |
| 68 |
crc ## 0 = __builtin_ia32_crc32si(crc ## 0, *((uint32_t*) buf ## 0 + 1 + 2 * offset)); \ |
| 69 |
crc ## 1 = __builtin_ia32_crc32si(crc ## 1, *((uint32_t*) buf ## 1 + 1 + 2 * offset)); |
| 70 |
#else |
| 71 |
#define CRCduplet(crc, buf, offset) \ |
| 72 |
crc ## 0 = __builtin_ia32_crc32di(crc ## 0, *(buf ## 0 + offset)); \ |
| 73 |
crc ## 1 = __builtin_ia32_crc32di(crc ## 1, *(buf ## 1 + offset)); |
| 74 |
#endif |
| 75 |
|
| 76 |
|
| 77 |
#ifndef __LP64__ |
| 78 |
#define CRCsinglet(crc, buf, offset) \ |
| 79 |
crc = __builtin_ia32_crc32si(crc, *(uint32_t*)(buf + offset)); \ |
| 80 |
crc = __builtin_ia32_crc32si(crc, *(uint32_t*)(buf + offset + sizeof(uint32_t))); |
| 81 |
#else |
| 82 |
#define CRCsinglet(crc, buf, offset) crc = __builtin_ia32_crc32di(crc, *(uint64_t*)(buf + offset)); |
| 83 |
#endif |
| 84 |
|
| 85 |
|
| 86 |
/* |
| 87 |
* CombineCRC performs pclmulqdq multiplication of 2 partial CRC's and a well chosen constant |
| 88 |
* and xor's these with the remaining CRC. I (Ferry Toth) could not find a way to implement this in |
| 89 |
* C, so the 64bit code following here is from Intel. As that code runs only on 64 bit (due to movq |
| 90 |
* instructions), I am providing a 32bit variant that does the same but using movd. The 32bit |
| 91 |
* version keeps intermediate results longer in the xmm registers to do the 2nd xor, then moves the |
| 92 |
* longs in 2 steps for the final crc32l |
| 93 |
* |
| 94 |
*/ |
| 95 |
|
| 96 |
#ifndef __LP64__ |
| 97 |
#define CombineCRC()\ |
| 98 |
__asm__ __volatile__ (\ |
| 99 |
"movdqu (%3), %%xmm0\n\t"\ |
| 100 |
"movd %0, %%xmm1\n\t"\ |
| 101 |
"pclmullqlqdq %%xmm0, %%xmm1\n\t"\ |
| 102 |
"movd %2, %%xmm2\n\t"\ |
| 103 |
"pclmullqhqdq %%xmm0, %%xmm2\n\t"\ |
| 104 |
"pxor %%xmm2, %%xmm1\n\t"\ |
| 105 |
"movdqu (%4), %%xmm2\n\t"\ |
| 106 |
"pxor %%xmm2, %%xmm1\n\t"\ |
| 107 |
"movd %%xmm1, %0\n\t"\ |
| 108 |
"crc32l %0, %5\n\t"\ |
| 109 |
"pextrd $1, %%xmm1, %1\n\t"\ |
| 110 |
"crc32l %1, %5\n\t"\ |
| 111 |
"movl %5, %0"\ |
| 112 |
: "=r" ( crc0 )\ |
| 113 |
: "0" ( crc0 ), "r" ( crc1 ), "r" ( crc32cIntelC_K + block_size - 1 ), "r" ( ( uint64_t* ) next2 - 1 ), "r" ( crc2 )\ |
| 114 |
: "%xmm0", "%xmm1", "%xmm2"\ |
| 115 |
); |
| 116 |
#else |
| 117 |
#define CombineCRC()\ |
| 118 |
__asm__ __volatile__ (\ |
| 119 |
"movdqa (%3), %%xmm0\n\t"\ |
| 120 |
"movq %0, %%xmm1\n\t"\ |
| 121 |
"pclmullqlqdq %%xmm0, %%xmm1\n\t"\ |
| 122 |
"movq %2, %%xmm2\n\t"\ |
| 123 |
"pclmullqhqdq %%xmm0, %%xmm2\n\t"\ |
| 124 |
"pxor %%xmm2, %%xmm1\n\t"\ |
| 125 |
"movq %%xmm1, %0"\ |
| 126 |
: "=r" ( crc0 ) \ |
| 127 |
: "0" ( crc0 ), "r" ( crc1 ), "r" ( crc32cIntelC_K + block_size - 1 ) \ |
| 128 |
: "%xmm0", "%xmm1", "%xmm2"\ |
| 129 |
); \ |
| 130 |
crc0 = crc0 ^ * ( ( uint64_t* ) next2 - 1 );\ |
| 131 |
crc2 = __builtin_ia32_crc32di ( crc2, crc0 );\ |
| 132 |
crc0 = crc2; |
| 133 |
#endif |
| 134 |
// kate: indent-mode cstyle; indent-width 4; replace-tabs on; |