v3.8.4

2026-05-30 00:32:28 +00:00 · 2018-03-18 12:51:03 -04:00
parent 157508bd07
commit 20fe05054c
19 changed files with 830 additions and 289 deletions
--- a/algo/cubehash/cube-hash-2way.c
+++ b/algo/cubehash/cube-hash-2way.c
@@ -0,0 +1,205 @@
+#if defined(__AVX2__)
+
+#include <stdbool.h>
+#include <unistd.h>
+#include <memory.h>
+#include "cube-hash-2way.h"
+
+// 2x128
+
+static void transform_2way( cube_2way_context *sp )
+{
+    int r;
+    const int rounds = sp->rounds;
+
+    __m256i x0, x1, x2, x3, x4, x5, x6, x7, y0, y1, y2, y3;
+
+    x0 = _mm256_load_si256( (__m256i*)sp->h     );
+    x1 = _mm256_load_si256( (__m256i*)sp->h + 1 );
+    x2 = _mm256_load_si256( (__m256i*)sp->h + 2 );
+    x3 = _mm256_load_si256( (__m256i*)sp->h + 3 );
+    x4 = _mm256_load_si256( (__m256i*)sp->h + 4 );
+    x5 = _mm256_load_si256( (__m256i*)sp->h + 5 );
+    x6 = _mm256_load_si256( (__m256i*)sp->h + 6 );
+    x7 = _mm256_load_si256( (__m256i*)sp->h + 7 );
+
+    for ( r = 0; r < rounds; ++r )
+    {
+        x4 = _mm256_add_epi32( x0, x4 );
+        x5 = _mm256_add_epi32( x1, x5 );
+        x6 = _mm256_add_epi32( x2, x6 );
+        x7 = _mm256_add_epi32( x3, x7 );
+        y0 = x2;
+        y1 = x3;
+        y2 = x0;
+        y3 = x1;
+        x0 = _mm256_xor_si256( _mm256_slli_epi32( y0,  7 ),
+                               _mm256_srli_epi32( y0, 25 ) );
+        x1 = _mm256_xor_si256( _mm256_slli_epi32( y1,  7 ),
+                               _mm256_srli_epi32( y1, 25 ) );
+        x2 = _mm256_xor_si256( _mm256_slli_epi32( y2,  7 ),
+                               _mm256_srli_epi32( y2, 25 ) );
+        x3 = _mm256_xor_si256( _mm256_slli_epi32( y3,  7 ),
+                               _mm256_srli_epi32( y3, 25 ) );
+        x0 = _mm256_xor_si256( x0, x4 );
+        x1 = _mm256_xor_si256( x1, x5 );
+        x2 = _mm256_xor_si256( x2, x6 );
+        x3 = _mm256_xor_si256( x3, x7 );
+        x4 = mm256_swap128_64( x4 );
+        x5 = mm256_swap128_64( x5 );
+        x6 = mm256_swap128_64( x6 );
+        x7 = mm256_swap128_64( x7 );
+        x4 = _mm256_add_epi32( x0, x4 );
+        x5 = _mm256_add_epi32( x1, x5 );
+        x6 = _mm256_add_epi32( x2, x6 );
+        x7 = _mm256_add_epi32( x3, x7 );
+        y0 = x1;
+        y1 = x0;
+        y2 = x3;
+        y3 = x2;
+        x0 = _mm256_xor_si256( _mm256_slli_epi32( y0, 11 ),
+                               _mm256_srli_epi32( y0, 21 ) );
+        x1 = _mm256_xor_si256( _mm256_slli_epi32( y1, 11 ),
+                               _mm256_srli_epi32( y1, 21 ) );
+        x2 = _mm256_xor_si256( _mm256_slli_epi32( y2, 11 ),
+                               _mm256_srli_epi32( y2, 21 ) );
+        x3 = _mm256_xor_si256( _mm256_slli_epi32( y3, 11 ),
+                               _mm256_srli_epi32( y3, 21 ) );
+        x0 = _mm256_xor_si256( x0, x4 );
+        x1 = _mm256_xor_si256( x1, x5 );
+        x2 = _mm256_xor_si256( x2, x6 );
+        x3 = _mm256_xor_si256( x3, x7 );
+        x4 = mm256_swap64_32( x4 );
+        x5 = mm256_swap64_32( x5 );
+        x6 = mm256_swap64_32( x6 );
+        x7 = mm256_swap64_32( x7 );
+    }
+
+    _mm256_store_si256( (__m256i*)sp->h,     x0 );
+    _mm256_store_si256( (__m256i*)sp->h + 1, x1 );
+    _mm256_store_si256( (__m256i*)sp->h + 2, x2 );
+    _mm256_store_si256( (__m256i*)sp->h + 3, x3 );
+    _mm256_store_si256( (__m256i*)sp->h + 4, x4 );
+    _mm256_store_si256( (__m256i*)sp->h + 5, x5 );
+    _mm256_store_si256( (__m256i*)sp->h + 6, x6 );
+    _mm256_store_si256( (__m256i*)sp->h + 7, x7 );
+
+}
+
+cube_2way_context cube_2way_ctx_cache __attribute__ ((aligned (64)));
+
+int cube_2way_reinit( cube_2way_context *sp )
+{
+   memcpy( sp, &cube_2way_ctx_cache, sizeof(cube_2way_context) );
+   return 0;
+
+}
+
+int cube_2way_init( cube_2way_context *sp, int hashbitlen, int rounds,
+                       int blockbytes )
+{
+    int i;
+
+    // all sizes of __m128i
+    cube_2way_ctx_cache.hashlen   = hashbitlen/128;
+    cube_2way_ctx_cache.blocksize = blockbytes/16;
+    cube_2way_ctx_cache.rounds    = rounds;
+    cube_2way_ctx_cache.pos       = 0;
+
+    for ( i = 0; i < 8; ++i )
+       cube_2way_ctx_cache.h[i] = m256_zero;
+
+    cube_2way_ctx_cache.h[0] = _mm256_set_epi32(
+                                   0, rounds, blockbytes, hashbitlen / 8,
+                                   0, rounds, blockbytes, hashbitlen / 8 );
+
+    for ( i = 0; i < 10; ++i )
+       transform_2way( &cube_2way_ctx_cache );
+
+    memcpy( sp, &cube_2way_ctx_cache, sizeof(cube_2way_context) );
+    return 0;
+}
+
+
+int cube_2way_update( cube_2way_context *sp, const void *data, size_t size )
+{
+    const int len = size / 16;
+    const __m256i *in = (__m256i*)data;
+    int i;
+
+    // It is assumed data is aligned to 256 bits and is a multiple of 128 bits.
+    // Current usage sata is either 64 or 80 bytes.
+
+    for ( i = 0; i < len; i++ )
+    {
+        sp->h[ sp->pos ] = _mm256_xor_si256( sp->h[ sp->pos ], in[i] );
+        sp->pos++;
+        if ( sp->pos == sp->blocksize )
+        {
+           transform_2way( sp );
+           sp->pos = 0;
+        }
+    }
+
+    return 0;
+}
+
+int cube_2way_close( cube_2way_context *sp, void *output )
+{
+    __m256i *hash = (__m256i*)output;
+    int i;
+
+    // pos is zero for 64 byte data, 1 for 80 byte data.
+    sp->h[ sp->pos ] = _mm256_xor_si256( sp->h[ sp->pos ],
+                    _mm256_set_epi8( 0,0,0,0, 0,0,0,0, 0,0,0,0, 0,0,0,0x80,
+                                     0,0,0,0, 0,0,0,0, 0,0,0,0, 0,0,0,0x80 ) );
+    transform_2way( sp );
+
+    sp->h[7] = _mm256_xor_si256( sp->h[7], _mm256_set_epi32( 1,0,0,0,
+                                                             1,0,0,0 ) );
+    for ( i = 0; i < 10; ++i )
+       transform_2way( &cube_2way_ctx_cache );
+
+    for ( i = 0; i < sp->hashlen; i++ )
+       hash[i] = sp->h[i];
+
+    return 0;
+}
+
+int cube_2way_update_close( cube_2way_context *sp, void *output,
+                               const void *data, size_t size )
+{
+    const int len = size / 16;
+    const __m256i *in = (__m256i*)data;
+    __m256i *hash = (__m256i*)output;
+    int i;
+
+    for ( i = 0; i < len; i++ )
+    {
+        sp->h[ sp->pos ] = _mm256_xor_si256( sp->h[ sp->pos ], in[i] );
+        sp->pos++;
+        if ( sp->pos == sp->blocksize )
+        {
+           transform_2way( sp );
+           sp->pos = 0;
+        }
+    }
+
+    // pos is zero for 64 byte data, 1 for 80 byte data.
+    sp->h[ sp->pos ] = _mm256_xor_si256( sp->h[ sp->pos ],
+                    _mm256_set_epi8( 0,0,0,0, 0,0,0,0, 0,0,0,0, 0,0,0,0x80,
+                                     0,0,0,0, 0,0,0,0, 0,0,0,0, 0,0,0,0x80 ) );
+    transform_2way( sp );
+
+    sp->h[7] = _mm256_xor_si256( sp->h[7], _mm256_set_epi32( 1,0,0,0,
+                                                             1,0,0,0 ) );
+    for ( i = 0; i < 10; ++i )
+       transform_2way( &cube_2way_ctx_cache );
+
+    for ( i = 0; i < sp->hashlen; i++ )
+       hash[i] = sp->h[i];
+
+    return 0;
+}
+
+#endif
--- a/algo/cubehash/cube-hash-2way.h
+++ b/algo/cubehash/cube-hash-2way.h
@@ -0,0 +1,36 @@
+#ifndef CUBE_HASH_2WAY_H__
+#define CUBE_HASH_2WAY_H__
+
+#if defined(__AVX2__)
+
+#include <stdint.h>
+#include "avxdefs.h"
+
+// 2x128, 2 way parallel SSE2
+
+struct _cube_2way_context
+{
+    int hashlen;           // __m128i
+    int rounds;
+    int blocksize;         // __m128i
+    int pos;               // number of __m128i read into x from current block
+    __m256i h[8] __attribute__ ((aligned (64)));
+};
+
+typedef struct _cube_2way_context cube_2way_context;
+
+int cube_2way_init( cube_2way_context* sp, int hashbitlen, int rounds,
+                       int blockbytes );
+// reinitialize context with same parameters, much faster.
+int cube_2way_reinit( cube_2way_context *sp );
+
+int cube_2way_update( cube_2way_context *sp, const void *data, size_t size );
+
+int cube_2way_close( cube_2way_context *sp, void *output );
+
+int cube_2way_update_close( cube_2way_context *sp, void *output,
+                            const void *data, size_t size );
+
+
+#endif
+#endif
--- a/algo/hodl/hodl-gate.c
+++ b/algo/hodl/hodl-gate.c
@@ -76,7 +76,6 @@ char* hodl_malloc_txs_request( struct work *work )
  return req;
 }

-
 void hodl_build_block_header( struct work* g_work, uint32_t version,
                              uint32_t *prevhash, uint32_t *merkle_tree,
                              uint32_t ntime, uint32_t nbits )
@@ -88,16 +87,16 @@ void hodl_build_block_header( struct work* g_work, uint32_t version,

   if ( have_stratum )
      for ( i = 0; i < 8; i++ )
-         g_work->data[1 + i] = le32dec( prevhash + i );
+         g_work->data[ 1+i ] = le32dec( prevhash + i );
   else
      for (i = 0; i < 8; i++)
         g_work->data[ 8-i ] = le32dec( prevhash + i );

   for ( i = 0; i < 8; i++ )
-      g_work->data[9 + i] = be32dec(  merkle_tree + i );
+      g_work->data[ 9+i ] = be32dec( merkle_tree + i );

-   g_work->data[ algo_gate.ntime_index ] =  ntime;
-   g_work->data[ algo_gate.nbits_index ] =  nbits;
+   g_work->data[ algo_gate.ntime_index ] = ntime;
+   g_work->data[ algo_gate.nbits_index ] = nbits;
   g_work->data[22] = 0x80000000;
   g_work->data[31] = 0x00000280;
 }
@@ -194,8 +193,13 @@ bool register_hodl_algo( algo_gate_t* gate )
  applog( LOG_ERR, "Only CPUs with AES are supported, use legacy version.");
  return false;
 #endif
+//  if ( TOTAL_CHUNKS % opt_n_threads )
+//  {
+//     applog(LOG_ERR,"Thread count must be power of 2.");
+//     return false;
+//  }
  pthread_barrier_init( &hodl_barrier, NULL, opt_n_threads );
-  gate->optimizations         = SSE2_OPT | AES_OPT | AVX_OPT | AVX2_OPT;
+  gate->optimizations         = AES_OPT | AVX_OPT | AVX2_OPT;
  gate->scanhash              = (void*)&hodl_scanhash;
  gate->get_new_work          = (void*)&hodl_get_new_work;
  gate->longpoll_rpc_call     = (void*)&hodl_longpoll_rpc_call;
--- a/algo/hodl/hodl-wolf.c
+++ b/algo/hodl/hodl-wolf.c
@@ -10,23 +10,26 @@

 #ifndef NO_AES_NI               

-void GenerateGarbageCore(CacheEntry *Garbage, int ThreadID, int ThreadCount, void *MidHash)
+void GenerateGarbageCore( CacheEntry *Garbage, int ThreadID, int ThreadCount,
+     void *MidHash )
 {
-#ifdef __AVX__
-    uint64_t* TempBufs[SHA512_PARALLEL_N] ;
-    uint64_t* desination[SHA512_PARALLEL_N];
+    const int Chunk = TOTAL_CHUNKS / ThreadCount;
+    const uint32_t StartChunk = ThreadID * Chunk;
+    const uint32_t EndChunk   = StartChunk + Chunk;

-    for ( int i=0; i<SHA512_PARALLEL_N; ++i )
+#ifdef __AVX__
+    uint64_t* TempBufs[ SHA512_PARALLEL_N ] ;
+    uint64_t* desination[ SHA512_PARALLEL_N ];
+
+    for ( int i=0; i < SHA512_PARALLEL_N; ++i )
    {
-        TempBufs[i] = (uint64_t*)malloc(32);
-        memcpy(TempBufs[i], MidHash, 32);
+        TempBufs[i] = (uint64_t*)malloc( 32 );
+        memcpy( TempBufs[i], MidHash, 32 );
    }

-    uint32_t StartChunk = ThreadID * (TOTAL_CHUNKS / ThreadCount);
-    for ( uint32_t i = StartChunk;
-          i < StartChunk + (TOTAL_CHUNKS / ThreadCount); i+= SHA512_PARALLEL_N )
+    for ( uint32_t i = StartChunk; i < EndChunk; i += SHA512_PARALLEL_N )
    {
-        for ( int j=0; j<SHA512_PARALLEL_N; ++j )
+        for ( int j = 0; j < SHA512_PARALLEL_N; ++j )
        {
            ( (uint32_t*)TempBufs[j] )[0] = i + j;
            desination[j] = (uint64_t*)( (uint8_t *)Garbage + ( (i+j)
@@ -35,15 +38,13 @@ void GenerateGarbageCore(CacheEntry *Garbage, int ThreadID, int ThreadCount, voi
        sha512Compute32b_parallel( TempBufs, desination );
    }

-    for ( int i=0; i<SHA512_PARALLEL_N; ++i )
+    for ( int i = 0; i < SHA512_PARALLEL_N; ++i )
        free( TempBufs[i] );
 #else
    uint32_t TempBuf[8];
    memcpy( TempBuf, MidHash, 32 );

-    uint32_t StartChunk = ThreadID * (TOTAL_CHUNKS / ThreadCount);
-    for ( uint32_t i = StartChunk;
-          i < StartChunk + (TOTAL_CHUNKS / ThreadCount); ++i )
+    for ( uint32_t i = StartChunk; i < EndChunk; ++i )
    {
        TempBuf[0] = i;
        SHA512( ( uint8_t *)TempBuf, 32,
--- a/algo/lyra2/sponge.h
+++ b/algo/lyra2/sponge.h
@@ -103,16 +103,16 @@ static inline uint64_t rotr64( const uint64_t w, const unsigned c ){
   b = mm_rotr_64( _mm_xor_si128( b, c ), 63 );

 #define LYRA_ROUND_AVX(s0,s1,s2,s3,s4,s5,s6,s7) \
-   G_2X64( s0, s2, s4, s6 ); \
-   G_2X64( s1, s3, s5, s7 ); \
-   mm_rotl256_1x64( s2, s3 ); \
-   mm_swap_128( s4, s5 ); \
-   mm_rotr256_1x64( s6, s7 ); \
   G_2X64( s0, s2, s4, s6 ); \
   G_2X64( s1, s3, s5, s7 ); \
   mm_rotr256_1x64( s2, s3 ); \
   mm_swap_128( s4, s5 ); \
-   mm_rotl256_1x64( s6, s7 );
+   mm_rotl256_1x64( s6, s7 ); \
+   G_2X64( s0, s2, s4, s6 ); \
+   G_2X64( s1, s3, s5, s7 ); \
+   mm_rotl256_1x64( s2, s3 ); \
+   mm_swap_128( s4, s5 ); \
+   mm_rotr256_1x64( s6, s7 );

 #define LYRA_12_ROUNDS_AVX(s0,s1,s2,s3,s4,s5,s6,s7) \
   LYRA_ROUND_AVX(s0,s1,s2,s3,s4,s5,s6,s7) \
--- a/algo/shavite/sph-shavite-aesni.c
+++ b/algo/shavite/sph-shavite-aesni.c
@@ -81,9 +81,9 @@ static const sph_u32 IV512[] = {
 // Similar to mm_rotr256_1x32 but only a partial rotation as lo is not
 // completed. It's faster than a full rotation.

-static inline __m128i mm_rotr256hi_1x32( __m128i hi, __m128i lo, int n )
-{   return _mm_or_si128( _mm_srli_si128( hi, n<<2 ),
-                        _mm_slli_si128( lo, 16 - (n<<2) ) );
+static inline __m128i mm_rotr256hi_1x32( __m128i hi, __m128i lo )
+{   return _mm_or_si128( _mm_srli_si128( hi,  4 ),
+                         _mm_slli_si128( lo, 12 ) );
 }

 #define AES_ROUND_NOKEY(x0, x1, x2, x3)   do { \
@@ -388,36 +388,36 @@ c512( sph_shavite_big_context *sc, const void *msg )

      // round 2, 6, 10

-      k00 = _mm_xor_si128( k00, mm_rotr256hi_1x32( k12, k13, 1 ) );
+      k00 = _mm_xor_si128( k00, mm_rotr256hi_1x32( k12, k13 ) );
      x = _mm_xor_si128( p3, k00 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k01 = _mm_xor_si128( k01, mm_rotr256hi_1x32( k13, k00, 1 ) );
+      k01 = _mm_xor_si128( k01, mm_rotr256hi_1x32( k13, k00 ) );
      x = _mm_xor_si128( x, k01 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k02 = _mm_xor_si128( k02, mm_rotr256hi_1x32( k00, k01, 1 ) );
+      k02 = _mm_xor_si128( k02, mm_rotr256hi_1x32( k00, k01 ) );
      x = _mm_xor_si128( x, k02 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k03 = _mm_xor_si128( k03, mm_rotr256hi_1x32( k01, k02, 1 ) );
+      k03 = _mm_xor_si128( k03, mm_rotr256hi_1x32( k01, k02 ) );
      x = _mm_xor_si128( x, k03 );
      x = _mm_aesenc_si128( x, m128_zero );

      p2 = _mm_xor_si128( p2, x );
-      k10 = _mm_xor_si128( k10, mm_rotr256hi_1x32( k02, k03, 1 ) );
+      k10 = _mm_xor_si128( k10, mm_rotr256hi_1x32( k02, k03 ) );
      x = _mm_xor_si128( p1, k10 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k11 = _mm_xor_si128( k11, mm_rotr256hi_1x32( k03, k10, 1 ) );
+      k11 = _mm_xor_si128( k11, mm_rotr256hi_1x32( k03, k10 ) );
      x = _mm_xor_si128( x, k11 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k12 = _mm_xor_si128( k12, mm_rotr256hi_1x32( k10, k11, 1 ) );
+      k12 = _mm_xor_si128( k12, mm_rotr256hi_1x32( k10, k11 ) );
      x = _mm_xor_si128( x, k12 );
      x = _mm_aesenc_si128( x, m128_zero );

-      k13 = _mm_xor_si128( k13, mm_rotr256hi_1x32( k11, k12, 1 ) );
+      k13 = _mm_xor_si128( k13, mm_rotr256hi_1x32( k11, k12 ) );
      x = _mm_xor_si128( x, k13 );
      x = _mm_aesenc_si128( x, m128_zero );
      p0 = _mm_xor_si128( p0, x );
@@ -470,36 +470,36 @@ c512( sph_shavite_big_context *sc, const void *msg )

      // round 4, 8, 12

-      k00 = _mm_xor_si128( k00, mm_rotr256hi_1x32( k12, k13, 1 ) );
+      k00 = _mm_xor_si128( k00, mm_rotr256hi_1x32( k12, k13 ) );

      x = _mm_xor_si128( p1, k00 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k01 = _mm_xor_si128( k01, mm_rotr256hi_1x32( k13, k00, 1 ) );
+      k01 = _mm_xor_si128( k01, mm_rotr256hi_1x32( k13, k00 ) );

      x = _mm_xor_si128( x, k01 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k02 = _mm_xor_si128( k02, mm_rotr256hi_1x32( k00, k01, 1 ) );
+      k02 = _mm_xor_si128( k02, mm_rotr256hi_1x32( k00, k01 ) );

      x = _mm_xor_si128( x, k02 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k03 = _mm_xor_si128( k03, mm_rotr256hi_1x32( k01, k02, 1 ) );
+      k03 = _mm_xor_si128( k03, mm_rotr256hi_1x32( k01, k02 ) );

      x = _mm_xor_si128( x, k03 );
      x = _mm_aesenc_si128( x, m128_zero );
      p0 = _mm_xor_si128( p0, x );
-      k10 = _mm_xor_si128( k10, mm_rotr256hi_1x32( k02, k03, 1 ) );
+      k10 = _mm_xor_si128( k10, mm_rotr256hi_1x32( k02, k03 ) );

      x = _mm_xor_si128( p3, k10 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k11 = _mm_xor_si128( k11, mm_rotr256hi_1x32( k03, k10, 1 ) );
+      k11 = _mm_xor_si128( k11, mm_rotr256hi_1x32( k03, k10 ) );

      x = _mm_xor_si128( x, k11 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k12 = _mm_xor_si128( k12, mm_rotr256hi_1x32( k10, k11, 1 ) );
+      k12 = _mm_xor_si128( k12, mm_rotr256hi_1x32( k10, k11 ) );

      x = _mm_xor_si128( x, k12 );
      x = _mm_aesenc_si128( x, m128_zero );
-      k13 = _mm_xor_si128( k13, mm_rotr256hi_1x32( k11, k12, 1 ) );
+      k13 = _mm_xor_si128( k13, mm_rotr256hi_1x32( k11, k12 ) );

      x = _mm_xor_si128( x, k13 );
      x = _mm_aesenc_si128( x, m128_zero );
--- a/algo/yescrypt/yescrypt-simd.c
+++ b/algo/yescrypt/yescrypt-simd.c
@@ -1363,10 +1363,11 @@ yescrypt_kdf(const yescrypt_shared_t * shared, yescrypt_local_t * local,
 	   {
 		HMAC_SHA256_CTX ctx;
 		HMAC_SHA256_Init(&ctx, buf, buflen);
-                if ( client_key_hack ) // GlobalBoost-Y buggy yescrypt
-			HMAC_SHA256_Update(&ctx, salt, saltlen);
-                else // Proper yescrypt
-                        HMAC_SHA256_Update(&ctx, "Client Key", 10);
+                if ( yescrypt_client_key )
+                    HMAC_SHA256_Update( &ctx, (uint8_t*)yescrypt_client_key,
+                                        yescrypt_client_key_len );
+                else
+                    HMAC_SHA256_Update( &ctx, salt, saltlen );
 		HMAC_SHA256_Final(sha256, &ctx);
 	   }
 	   /* Compute StoredKey */
--- a/algo/yescrypt/yescrypt.c
+++ b/algo/yescrypt/yescrypt.c
@@ -25,7 +25,7 @@
 #include "compat.h"

 #include "yescrypt.h"
-
+#include "sha256_Y.h"
 #include "algo-gate-api.h"

 #define BYTES2CHARS(bytes) \
@@ -366,7 +366,8 @@ static int yescrypt_bsty(const uint8_t * passwd, size_t passwdlen,
 uint64_t YESCRYPT_N;
 uint32_t YESCRYPT_R;
 uint32_t YESCRYPT_P;
-bool client_key_hack;
+char *yescrypt_client_key = NULL;
+int yescrypt_client_key_len = 0;

 /* main hash 80 bytes input */
 void yescrypt_hash( const char *input, char *output, uint32_t len )
@@ -436,7 +437,8 @@ bool register_yescrypt_algo( algo_gate_t* gate )
 {
   yescrypt_gate_base( gate );
   gate->get_max64  = (void*)&yescrypt_get_max64;
-   client_key_hack = true;
+   yescrypt_client_key = NULL;
+   yescrypt_client_key_len = 0;
   YESCRYPT_N = 2048;
   YESCRYPT_R = 8;
   YESCRYPT_P = 1;
@@ -447,7 +449,8 @@ bool register_yescryptr8_algo( algo_gate_t* gate )
 {
   yescrypt_gate_base( gate );
   gate->get_max64  = (void*)&yescrypt_get_max64;
-   client_key_hack = false;
+   yescrypt_client_key = "Client Key";
+   yescrypt_client_key_len = 10;
   YESCRYPT_N = 2048;
   YESCRYPT_R = 8;
   YESCRYPT_P = 1;
@@ -458,10 +461,23 @@ bool register_yescryptr16_algo( algo_gate_t* gate )
 {
   yescrypt_gate_base( gate );
   gate->get_max64  = (void*)&yescryptr16_get_max64;
-   client_key_hack = false;
+   yescrypt_client_key = "Client Key";
+   yescrypt_client_key_len = 10;
   YESCRYPT_N = 4096;   
   YESCRYPT_R = 16;   
   YESCRYPT_P = 1;   
   return true;
 }

+bool register_yescryptr32_algo( algo_gate_t* gate )
+{
+   yescrypt_gate_base( gate );
+   gate->get_max64  = (void*)&yescryptr16_get_max64;
+   yescrypt_client_key = "WaviBanana";
+   yescrypt_client_key_len = 10;
+   YESCRYPT_N = 4096;
+   YESCRYPT_R = 32;
+   YESCRYPT_P = 1;
+   return true;
+}
+
--- a/algo/yescrypt/yescrypt.h
+++ b/algo/yescrypt/yescrypt.h
@@ -108,7 +108,8 @@ typedef enum {
 	__YESCRYPT_INIT_SHARED = 0x30000
 } yescrypt_flags_t;

-extern bool client_key_hack;    // true for GlobalBoost-Y
+extern char *yescrypt_client_key;
+extern int yescrypt_client_key_len;


 #define YESCRYPT_KNOWN_FLAGS \