Use uint16 in SHA256_fresh in scrypt kernel.
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329
diff --git a/scrypt120713.cl b/scrypt120713.cl
index 900ccce..768a866 100644
--- a/scrypt120713.cl
+++ b/scrypt120713.cl
@@ -237,194 +237,194 @@ void SHA256_fresh(uint4*restrict state0,uint4*restrict state1, const uint4 block
#define G (*state1).z
#define H (*state1).w
- uint4 W[4];
+ uint16 W;
- W[0].x = block0.x;
- D=0x98c7e2a2U+W[0].x;
- H=0xfc08884dU+W[0].x;
+ W.s0 = block0.x;
+ D=0x98c7e2a2U+W.s0;
+ H=0xfc08884dU+W.s0;
- W[0].y = block0.y;
- C=0xcd2a11aeU+Tr1(D)+Ch(D,0x510e527fU,0x9b05688cU)+W[0].y;
+ W.s1 = block0.y;
+ C=0xcd2a11aeU+Tr1(D)+Ch(D,0x510e527fU,0x9b05688cU)+W.s1;
G=0xC3910C8EU+C+Tr2(H)+Ch(H,0xfb6feee7U,0x2a01a605U);
- W[0].z = block0.z;
- B=0x0c2e12e0U+Tr1(C)+Ch(C,D,0x510e527fU)+W[0].z;
+ W.s2 = block0.z;
+ B=0x0c2e12e0U+Tr1(C)+Ch(C,D,0x510e527fU)+W.s2;
F=0x4498517BU+B+Tr2(G)+Maj(G,H,0x6a09e667U);
- W[0].w = block0.w;
- A=0xa4ce148bU+Tr1(B)+Ch(B,C,D)+W[0].w;
+ W.s3 = block0.w;
+ A=0xa4ce148bU+Tr1(B)+Ch(B,C,D)+W.s3;
E=0x95F61999U+A+Tr2(F)+Maj(F,G,H);
- W[1].x = block1.x;
- RND(E,F,G,H,A,B,C,D, W[1].x+0x3956c25bU);
- W[1].y = block1.y;
- RND(D,E,F,G,H,A,B,C, W[1].y+0x59f111f1U);
- W[1].z = block1.z;
- RND(C,D,E,F,G,H,A,B, W[1].z+0x923f82a4U);
- W[1].w = block1.w;
- RND(B,C,D,E,F,G,H,A, W[1].w+0xab1c5ed5U);
+ W.s4 = block1.x;
+ RND(E,F,G,H,A,B,C,D, W.s4+0x3956c25bU);
+ W.s5 = block1.y;
+ RND(D,E,F,G,H,A,B,C, W.s5+0x59f111f1U);
+ W.s6 = block1.z;
+ RND(C,D,E,F,G,H,A,B, W.s6+0x923f82a4U);
+ W.s7 = block1.w;
+ RND(B,C,D,E,F,G,H,A, W.s7+0xab1c5ed5U);
- W[2].x = block2.x;
- RND(A,B,C,D,E,F,G,H, W[2].x+0xd807aa98U);
- W[2].y = block2.y;
- RND(H,A,B,C,D,E,F,G, W[2].y+0x12835b01U);
- W[2].z = block2.z;
- RND(G,H,A,B,C,D,E,F, W[2].z+0x243185beU);
- W[2].w = block2.w;
- RND(F,G,H,A,B,C,D,E, W[2].w+0x550c7dc3U);
+ W.s8 = block2.x;
+ RND(A,B,C,D,E,F,G,H, W.s8+0xd807aa98U);
+ W.s9 = block2.y;
+ RND(H,A,B,C,D,E,F,G, W.s9+0x12835b01U);
+ W.sa = block2.z;
+ RND(G,H,A,B,C,D,E,F, W.sa+0x243185beU);
+ W.sb = block2.w;
+ RND(F,G,H,A,B,C,D,E, W.sb+0x550c7dc3U);
- W[3].x = block3.x;
- RND(E,F,G,H,A,B,C,D, W[3].x+0x72be5d74U);
- W[3].y = block3.y;
- RND(D,E,F,G,H,A,B,C, W[3].y+0x80deb1feU);
- W[3].z = block3.z;
- RND(C,D,E,F,G,H,A,B, W[3].z+0x9bdc06a7U);
- W[3].w = block3.w;
- RND(B,C,D,E,F,G,H,A, W[3].w+0xc19bf174U);
+ W.sc = block3.x;
+ RND(E,F,G,H,A,B,C,D, W.sc+0x72be5d74U);
+ W.sd = block3.y;
+ RND(D,E,F,G,H,A,B,C, W.sd+0x80deb1feU);
+ W.se = block3.z;
+ RND(C,D,E,F,G,H,A,B, W.se+0x9bdc06a7U);
+ W.sf = block3.w;
+ RND(B,C,D,E,F,G,H,A, W.sf+0xc19bf174U);
- W[0].x += Wr1(W[3].z) + W[2].y + Wr2(W[0].y);
- RND(A,B,C,D,E,F,G,H, W[0].x+0xe49b69c1U);
+ W.s0 += Wr1(W.se) + W.s9 + Wr2(W.s1);
+ RND(A,B,C,D,E,F,G,H, W.s0+0xe49b69c1U);
- W[0].y += Wr1(W[3].w) + W[2].z + Wr2(W[0].z);
- RND(H,A,B,C,D,E,F,G, W[0].y+0xefbe4786U);
+ W.s1 += Wr1(W.sf) + W.sa + Wr2(W.s2);
+ RND(H,A,B,C,D,E,F,G, W.s1+0xefbe4786U);
- W[0].z += Wr1(W[0].x) + W[2].w + Wr2(W[0].w);
- RND(G,H,A,B,C,D,E,F, W[0].z+0x0fc19dc6U);
+ W.s2 += Wr1(W.s0) + W.sb + Wr2(W.s3);
+ RND(G,H,A,B,C,D,E,F, W.s2+0x0fc19dc6U);
- W[0].w += Wr1(W[0].y) + W[3].x + Wr2(W[1].x);
- RND(F,G,H,A,B,C,D,E, W[0].w+0x240ca1ccU);
+ W.s3 += Wr1(W.s1) + W.sc + Wr2(W.s4);
+ RND(F,G,H,A,B,C,D,E, W.s3+0x240ca1ccU);
- W[1].x += Wr1(W[0].z) + W[3].y + Wr2(W[1].y);
- RND(E,F,G,H,A,B,C,D, W[1].x+0x2de92c6fU);
+ W.s4 += Wr1(W.s2) + W.sd + Wr2(W.s5);
+ RND(E,F,G,H,A,B,C,D, W.s4+0x2de92c6fU);
- W[1].y += Wr1(W[0].w) + W[3].z + Wr2(W[1].z);
- RND(D,E,F,G,H,A,B,C, W[1].y+0x4a7484aaU);
+ W.s5 += Wr1(W.s3) + W.se + Wr2(W.s6);
+ RND(D,E,F,G,H,A,B,C, W.s5+0x4a7484aaU);
- W[1].z += Wr1(W[1].x) + W[3].w + Wr2(W[1].w);
- RND(C,D,E,F,G,H,A,B, W[1].z+0x5cb0a9dcU);
+ W.s6 += Wr1(W.s4) + W.sf + Wr2(W.s7);
+ RND(C,D,E,F,G,H,A,B, W.s6+0x5cb0a9dcU);
- W[1].w += Wr1(W[1].y) + W[0].x + Wr2(W[2].x);
- RND(B,C,D,E,F,G,H,A, W[1].w+0x76f988daU);
+ W.s7 += Wr1(W.s5) + W.s0 + Wr2(W.s8);
+ RND(B,C,D,E,F,G,H,A, W.s7+0x76f988daU);
- W[2].x += Wr1(W[1].z) + W[0].y + Wr2(W[2].y);
- RND(A,B,C,D,E,F,G,H, W[2].x+0x983e5152U);
+ W.s8 += Wr1(W.s6) + W.s1 + Wr2(W.s9);
+ RND(A,B,C,D,E,F,G,H, W.s8+0x983e5152U);
- W[2].y += Wr1(W[1].w) + W[0].z + Wr2(W[2].z);
- RND(H,A,B,C,D,E,F,G, W[2].y+0xa831c66dU);
+ W.s9 += Wr1(W.s7) + W.s2 + Wr2(W.sa);
+ RND(H,A,B,C,D,E,F,G, W.s9+0xa831c66dU);
- W[2].z += Wr1(W[2].x) + W[0].w + Wr2(W[2].w);
- RND(G,H,A,B,C,D,E,F, W[2].z+0xb00327c8U);
+ W.sa += Wr1(W.s8) + W.s3 + Wr2(W.sb);
+ RND(G,H,A,B,C,D,E,F, W.sa+0xb00327c8U);
- W[2].w += Wr1(W[2].y) + W[1].x + Wr2(W[3].x);
- RND(F,G,H,A,B,C,D,E, W[2].w+0xbf597fc7U);
+ W.sb += Wr1(W.s9) + W.s4 + Wr2(W.sc);
+ RND(F,G,H,A,B,C,D,E, W.sb+0xbf597fc7U);
- W[3].x += Wr1(W[2].z) + W[1].y + Wr2(W[3].y);
- RND(E,F,G,H,A,B,C,D, W[3].x+0xc6e00bf3U);
+ W.sc += Wr1(W.sa) + W.s5 + Wr2(W.sd);
+ RND(E,F,G,H,A,B,C,D, W.sc+0xc6e00bf3U);
- W[3].y += Wr1(W[2].w) + W[1].z + Wr2(W[3].z);
- RND(D,E,F,G,H,A,B,C, W[3].y+0xd5a79147U);
+ W.sd += Wr1(W.sb) + W.s6 + Wr2(W.se);
+ RND(D,E,F,G,H,A,B,C, W.sd+0xd5a79147U);
- W[3].z += Wr1(W[3].x) + W[1].w + Wr2(W[3].w);
- RND(C,D,E,F,G,H,A,B, W[3].z+0x06ca6351U);
+ W.se += Wr1(W.sc) + W.s7 + Wr2(W.sf);
+ RND(C,D,E,F,G,H,A,B, W.se+0x06ca6351U);
- W[3].w += Wr1(W[3].y) + W[2].x + Wr2(W[0].x);
- RND(B,C,D,E,F,G,H,A, W[3].w+0x14292967U);
+ W.sf += Wr1(W.sd) + W.s8 + Wr2(W.s0);
+ RND(B,C,D,E,F,G,H,A, W.sf+0x14292967U);
- W[0].x += Wr1(W[3].z) + W[2].y + Wr2(W[0].y);
- RND(A,B,C,D,E,F,G,H, W[0].x+0x27b70a85U);
+ W.s0 += Wr1(W.se) + W.s9 + Wr2(W.s1);
+ RND(A,B,C,D,E,F,G,H, W.s0+0x27b70a85U);
- W[0].y += Wr1(W[3].w) + W[2].z + Wr2(W[0].z);
- RND(H,A,B,C,D,E,F,G, W[0].y+0x2e1b2138U);
+ W.s1 += Wr1(W.sf) + W.sa + Wr2(W.s2);
+ RND(H,A,B,C,D,E,F,G, W.s1+0x2e1b2138U);
- W[0].z += Wr1(W[0].x) + W[2].w + Wr2(W[0].w);
- RND(G,H,A,B,C,D,E,F, W[0].z+0x4d2c6dfcU);
+ W.s2 += Wr1(W.s0) + W.sb + Wr2(W.s3);
+ RND(G,H,A,B,C,D,E,F, W.s2+0x4d2c6dfcU);
- W[0].w += Wr1(W[0].y) + W[3].x + Wr2(W[1].x);
- RND(F,G,H,A,B,C,D,E, W[0].w+0x53380d13U);
+ W.s3 += Wr1(W.s1) + W.sc + Wr2(W.s4);
+ RND(F,G,H,A,B,C,D,E, W.s3+0x53380d13U);
- W[1].x += Wr1(W[0].z) + W[3].y + Wr2(W[1].y);
- RND(E,F,G,H,A,B,C,D, W[1].x+0x650a7354U);
+ W.s4 += Wr1(W.s2) + W.sd + Wr2(W.s5);
+ RND(E,F,G,H,A,B,C,D, W.s4+0x650a7354U);
- W[1].y += Wr1(W[0].w) + W[3].z + Wr2(W[1].z);
- RND(D,E,F,G,H,A,B,C, W[1].y+0x766a0abbU);
+ W.s5 += Wr1(W.s3) + W.se + Wr2(W.s6);
+ RND(D,E,F,G,H,A,B,C, W.s5+0x766a0abbU);
- W[1].z += Wr1(W[1].x) + W[3].w + Wr2(W[1].w);
- RND(C,D,E,F,G,H,A,B, W[1].z+0x81c2c92eU);
+ W.s6 += Wr1(W.s4) + W.sf + Wr2(W.s7);
+ RND(C,D,E,F,G,H,A,B, W.s6+0x81c2c92eU);
- W[1].w += Wr1(W[1].y) + W[0].x + Wr2(W[2].x);
- RND(B,C,D,E,F,G,H,A, W[1].w+0x92722c85U);
+ W.s7 += Wr1(W.s5) + W.s0 + Wr2(W.s8);
+ RND(B,C,D,E,F,G,H,A, W.s7+0x92722c85U);
- W[2].x += Wr1(W[1].z) + W[0].y + Wr2(W[2].y);
- RND(A,B,C,D,E,F,G,H, W[2].x+0xa2bfe8a1U);
+ W.s8 += Wr1(W.s6) + W.s1 + Wr2(W.s9);
+ RND(A,B,C,D,E,F,G,H, W.s8+0xa2bfe8a1U);
- W[2].y += Wr1(W[1].w) + W[0].z + Wr2(W[2].z);
- RND(H,A,B,C,D,E,F,G, W[2].y+0xa81a664bU);
+ W.s9 += Wr1(W.s7) + W.s2 + Wr2(W.sa);
+ RND(H,A,B,C,D,E,F,G, W.s9+0xa81a664bU);
- W[2].z += Wr1(W[2].x) + W[0].w + Wr2(W[2].w);
- RND(G,H,A,B,C,D,E,F, W[2].z+0xc24b8b70U);
+ W.sa += Wr1(W.s8) + W.s3 + Wr2(W.sb);
+ RND(G,H,A,B,C,D,E,F, W.sa+0xc24b8b70U);
- W[2].w += Wr1(W[2].y) + W[1].x + Wr2(W[3].x);
- RND(F,G,H,A,B,C,D,E, W[2].w+0xc76c51a3U);
+ W.sb += Wr1(W.s9) + W.s4 + Wr2(W.sc);
+ RND(F,G,H,A,B,C,D,E, W.sb+0xc76c51a3U);
- W[3].x += Wr1(W[2].z) + W[1].y + Wr2(W[3].y);
- RND(E,F,G,H,A,B,C,D, W[3].x+0xd192e819U);
+ W.sc += Wr1(W.sa) + W.s5 + Wr2(W.sd);
+ RND(E,F,G,H,A,B,C,D, W.sc+0xd192e819U);
- W[3].y += Wr1(W[2].w) + W[1].z + Wr2(W[3].z);
- RND(D,E,F,G,H,A,B,C, W[3].y+0xd6990624U);
+ W.sd += Wr1(W.sb) + W.s6 + Wr2(W.se);
+ RND(D,E,F,G,H,A,B,C, W.sd+0xd6990624U);
- W[3].z += Wr1(W[3].x) + W[1].w + Wr2(W[3].w);
- RND(C,D,E,F,G,H,A,B, W[3].z+0xf40e3585U);
+ W.se += Wr1(W.sc) + W.s7 + Wr2(W.sf);
+ RND(C,D,E,F,G,H,A,B, W.se+0xf40e3585U);
- W[3].w += Wr1(W[3].y) + W[2].x + Wr2(W[0].x);
- RND(B,C,D,E,F,G,H,A, W[3].w+0x106aa070U);
+ W.sf += Wr1(W.sd) + W.s8 + Wr2(W.s0);
+ RND(B,C,D,E,F,G,H,A, W.sf+0x106aa070U);
- W[0].x += Wr1(W[3].z) + W[2].y + Wr2(W[0].y);
- RND(A,B,C,D,E,F,G,H, W[0].x+0x19a4c116U);
+ W.s0 += Wr1(W.se) + W.s9 + Wr2(W.s1);
+ RND(A,B,C,D,E,F,G,H, W.s0+0x19a4c116U);
- W[0].y += Wr1(W[3].w) + W[2].z + Wr2(W[0].z);
- RND(H,A,B,C,D,E,F,G, W[0].y+0x1e376c08U);
+ W.s1 += Wr1(W.sf) + W.sa + Wr2(W.s2);
+ RND(H,A,B,C,D,E,F,G, W.s1+0x1e376c08U);
- W[0].z += Wr1(W[0].x) + W[2].w + Wr2(W[0].w);
- RND(G,H,A,B,C,D,E,F, W[0].z+0x2748774cU);
+ W.s2 += Wr1(W.s0) + W.sb + Wr2(W.s3);
+ RND(G,H,A,B,C,D,E,F, W.s2+0x2748774cU);
- W[0].w += Wr1(W[0].y) + W[3].x + Wr2(W[1].x);
- RND(F,G,H,A,B,C,D,E, W[0].w+0x34b0bcb5U);
+ W.s3 += Wr1(W.s1) + W.sc + Wr2(W.s4);
+ RND(F,G,H,A,B,C,D,E, W.s3+0x34b0bcb5U);
- W[1].x += Wr1(W[0].z) + W[3].y + Wr2(W[1].y);
- RND(E,F,G,H,A,B,C,D, W[1].x+0x391c0cb3U);
+ W.s4 += Wr1(W.s2) + W.sd + Wr2(W.s5);
+ RND(E,F,G,H,A,B,C,D, W.s4+0x391c0cb3U);
- W[1].y += Wr1(W[0].w) + W[3].z + Wr2(W[1].z);
- RND(D,E,F,G,H,A,B,C, W[1].y+0x4ed8aa4aU);
+ W.s5 += Wr1(W.s3) + W.se + Wr2(W.s6);
+ RND(D,E,F,G,H,A,B,C, W.s5+0x4ed8aa4aU);
- W[1].z += Wr1(W[1].x) + W[3].w + Wr2(W[1].w);
- RND(C,D,E,F,G,H,A,B, W[1].z+0x5b9cca4fU);
+ W.s6 += Wr1(W.s4) + W.sf + Wr2(W.s7);
+ RND(C,D,E,F,G,H,A,B, W.s6+0x5b9cca4fU);
- W[1].w += Wr1(W[1].y) + W[0].x + Wr2(W[2].x);
- RND(B,C,D,E,F,G,H,A, W[1].w+0x682e6ff3U);
+ W.s7 += Wr1(W.s5) + W.s0 + Wr2(W.s8);
+ RND(B,C,D,E,F,G,H,A, W.s7+0x682e6ff3U);
- W[2].x += Wr1(W[1].z) + W[0].y + Wr2(W[2].y);
- RND(A,B,C,D,E,F,G,H, W[2].x+0x748f82eeU);
+ W.s8 += Wr1(W.s6) + W.s1 + Wr2(W.s9);
+ RND(A,B,C,D,E,F,G,H, W.s8+0x748f82eeU);
- W[2].y += Wr1(W[1].w) + W[0].z + Wr2(W[2].z);
- RND(H,A,B,C,D,E,F,G, W[2].y+0x78a5636fU);
+ W.s9 += Wr1(W.s7) + W.s2 + Wr2(W.sa);
+ RND(H,A,B,C,D,E,F,G, W.s9+0x78a5636fU);
- W[2].z += Wr1(W[2].x) + W[0].w + Wr2(W[2].w);
- RND(G,H,A,B,C,D,E,F, W[2].z+0x84c87814U);
+ W.sa += Wr1(W.s8) + W.s3 + Wr2(W.sb);
+ RND(G,H,A,B,C,D,E,F, W.sa+0x84c87814U);
- W[2].w += Wr1(W[2].y) + W[1].x + Wr2(W[3].x);
- RND(F,G,H,A,B,C,D,E, W[2].w+0x8cc70208U);
+ W.sb += Wr1(W.s9) + W.s4 + Wr2(W.sc);
+ RND(F,G,H,A,B,C,D,E, W.sb+0x8cc70208U);
- W[3].x += Wr1(W[2].z) + W[1].y + Wr2(W[3].y);
- RND(E,F,G,H,A,B,C,D, W[3].x+0x90befffaU);
+ W.sc += Wr1(W.sa) + W.s5 + Wr2(W.sd);
+ RND(E,F,G,H,A,B,C,D, W.sc+0x90befffaU);
- W[3].y += Wr1(W[2].w) + W[1].z + Wr2(W[3].z);
- RND(D,E,F,G,H,A,B,C, W[3].y+0xa4506cebU);
+ W.sd += Wr1(W.sb) + W.s6 + Wr2(W.se);
+ RND(D,E,F,G,H,A,B,C, W.sd+0xa4506cebU);
- W[3].z += Wr1(W[3].x) + W[1].w + Wr2(W[3].w);
- RND(C,D,E,F,G,H,A,B, W[3].z+0xbef9a3f7U);
+ W.se += Wr1(W.sc) + W.s7 + Wr2(W.sf);
+ RND(C,D,E,F,G,H,A,B, W.se+0xbef9a3f7U);
- W[3].w += Wr1(W[3].y) + W[2].x + Wr2(W[0].x);
- RND(B,C,D,E,F,G,H,A, W[3].w+0xc67178f2U);
+ W.sf += Wr1(W.sd) + W.s8 + Wr2(W.s0);
+ RND(B,C,D,E,F,G,H,A, W.sf+0xc67178f2U);
#undef A
#undef B