2 * Blowfish Cipher Algorithm (x86_64)
4 * Copyright (C) 2011 Jussi Kivilinna <jussi.kivilinna@mbnet.fi>
6 * This program is free software; you can redistribute it and/or modify
7 * it under the terms of the GNU General Public License as published by
8 * the Free Software Foundation; either version 2 of the License, or
9 * (at your option) any later version.
11 * This program is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14 * GNU General Public License for more details.
16 * You should have received a copy of the GNU General Public License
17 * along with this program; if not, write to the Free Software
18 * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307
23 #include <linux/linkage.h>
25 .file "blowfish-x86_64-asm.S"
28 /* structure of crypto context */
30 #define s0 ((16 + 2) * 4)
31 #define s1 ((16 + 2 + (1 * 256)) * 4)
32 #define s2 ((16 + 2 + (2 * 256)) * 4)
33 #define s3 ((16 + 2 + (3 * 256)) * 4)
71 /***********************************************************************
73 ***********************************************************************/
79 movl s0(CTX,RT0,4), RT0d; \
80 addl s1(CTX,RT1,4), RT0d; \
84 xorl s2(CTX,RT1,4), RT0d; \
85 addl s3(CTX,RT2,4), RT0d; \
88 #define add_roundkey_enc(n) \
89 xorq p+4*(n)(CTX), RX0;
91 #define round_enc(n) \
92 add_roundkey_enc(n); \
97 #define add_roundkey_dec(n) \
98 movq p+4*(n-1)(CTX), RT0; \
102 #define round_dec(n) \
103 add_roundkey_dec(n); \
108 #define read_block() \
113 #define write_block() \
117 #define xor_block() \
121 ENTRY(__blowfish_enc_blk)
126 * %rcx: bool, if true: xor output
143 add_roundkey_enc(16);
156 ENDPROC(__blowfish_enc_blk)
158 ENTRY(blowfish_dec_blk)
187 ENDPROC(blowfish_dec_blk)
189 /**********************************************************************
190 4-way blowfish, four blocks parallel
191 **********************************************************************/
193 /* F() for 4-way. Slower when used alone/1-way, but faster when used
194 * parallel/4-way (tested on AMD Phenom II & Intel Xeon E7330).
197 movzbl x ## bh, RT1d; \
198 movzbl x ## bl, RT3d; \
200 movzbl x ## bh, RT0d; \
201 movzbl x ## bl, RT2d; \
203 movl s0(CTX,RT0,4), RT0d; \
204 addl s1(CTX,RT2,4), RT0d; \
205 xorl s2(CTX,RT1,4), RT0d; \
206 addl s3(CTX,RT3,4), RT0d; \
209 #define add_preloaded_roundkey4() \
215 #define preload_roundkey_enc(n) \
216 movq p+4*(n)(CTX), RKEY;
218 #define add_roundkey_enc4(n) \
219 add_preloaded_roundkey4(); \
220 preload_roundkey_enc(n + 2);
222 #define round_enc4(n) \
223 add_roundkey_enc4(n); \
235 #define preload_roundkey_dec(n) \
236 movq p+4*((n)-1)(CTX), RKEY; \
239 #define add_roundkey_dec4(n) \
240 add_preloaded_roundkey4(); \
241 preload_roundkey_dec(n - 2);
243 #define round_dec4(n) \
244 add_roundkey_dec4(n); \
256 #define read_block4() \
273 #define write_block4() \
286 #define xor_block4() \
299 ENTRY(__blowfish_enc_blk_4way)
304 * %rcx: bool, if true: xor output
310 preload_roundkey_enc(0);
325 add_preloaded_roundkey4();
345 ENDPROC(__blowfish_enc_blk_4way)
347 ENTRY(blowfish_dec_blk_4way)
355 preload_roundkey_dec(17);
370 add_preloaded_roundkey4();
379 ENDPROC(blowfish_dec_blk_4way)