CompressedGemma commited on
Commit
b60d890
·
verified ·
1 Parent(s): d11de92

Upload 4 files

Browse files
Files changed (4) hide show
  1. gguf_format.h +29 -0
  2. hexstate_quantize.c +182 -74
  3. hexstate_requantize.py +155 -46
  4. iq2xs_grid.h +410 -0
gguf_format.h CHANGED
@@ -168,6 +168,31 @@ typedef struct {
168
 
169
  /* sizeof(BlockIQ1S) = 2 + 32 + 16 = 50 bytes for 256 weights */
170
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
171
  /* ═══════════════════════════════════════════════════════════════════════
172
  * FP16 ←→ FP32 CONVERSION
173
  *
@@ -688,6 +713,8 @@ static inline int64_t ggml_type_block_size(GGMLType type)
688
  case GGML_TYPE_Q4_K: return 256;
689
  case GGML_TYPE_Q5_K: return 256;
690
  case GGML_TYPE_Q6_K: return 256;
 
 
691
  default: return 1;
692
  }
693
  }
@@ -702,6 +729,8 @@ static inline int64_t ggml_type_bytes_per_block(GGMLType type)
702
  case GGML_TYPE_Q2_K: return sizeof(BlockQ2K); /* 84 */
703
  case GGML_TYPE_Q4_0: return 18; /* 2 + 16 */
704
  case GGML_TYPE_Q4_1: return 20; /* 2 + 2 + 16 */
 
 
705
  default: return 4;
706
  }
707
  }
 
168
 
169
  /* sizeof(BlockIQ1S) = 2 + 32 + 16 = 50 bytes for 256 weights */
170
 
171
+ /* IQ2_XS BLOCK (2.3125 bpw). Layout matches ggml block_iq2_xs:
172
+ * d(fp16) + qs[32] (per 8 weights: 9-bit codeword index | 7 sign bits << 9)
173
+ * + scales[8] (4-bit per 16-weight sub-block, low nibble first).
174
+ * Dequant: y_j = d * (ls + 0.5) * 0.25 * grid[j] * sign_j, grid ∈ {8,25,43}. */
175
+ typedef struct {
176
+ uint16_t d;
177
+ uint16_t qs[QK_K/8];
178
+ uint8_t scales[QK_K/32];
179
+ } BlockIQ2XS;
180
+
181
+ /* sizeof(BlockIQ2XS) = 2 + 64 + 8 = 74 bytes for 256 weights */
182
+
183
+ /* IQ2_S BLOCK (2.5625 bpw). Layout matches ggml block_iq2_s:
184
+ * qs[0..31] low 8 bits of the 1024-codeword index per 8 weights,
185
+ * qs[32..63] full 8 sign bits per group, qh[8] two high index bits per
186
+ * group (2 bits × 4 groups per byte), scales[8] as IQ2_XS. */
187
+ typedef struct {
188
+ uint16_t d;
189
+ uint8_t qs[QK_K/4];
190
+ uint8_t qh[QK_K/32];
191
+ uint8_t scales[QK_K/32];
192
+ } BlockIQ2S;
193
+
194
+ /* sizeof(BlockIQ2S) = 2 + 64 + 8 + 8 = 82 bytes for 256 weights */
195
+
196
  /* ═══════════════════════════════════════════════════════════════════════
197
  * FP16 ←→ FP32 CONVERSION
198
  *
 
713
  case GGML_TYPE_Q4_K: return 256;
714
  case GGML_TYPE_Q5_K: return 256;
715
  case GGML_TYPE_Q6_K: return 256;
716
+ case GGML_TYPE_IQ2_XS: return QK_K;
717
+ case GGML_TYPE_IQ2_S: return QK_K;
718
  default: return 1;
719
  }
720
  }
 
729
  case GGML_TYPE_Q2_K: return sizeof(BlockQ2K); /* 84 */
730
  case GGML_TYPE_Q4_0: return 18; /* 2 + 16 */
731
  case GGML_TYPE_Q4_1: return 20; /* 2 + 2 + 16 */
732
+ case GGML_TYPE_IQ2_XS: return sizeof(BlockIQ2XS); /* 74 */
733
+ case GGML_TYPE_IQ2_S: return sizeof(BlockIQ2S); /* 82 */
734
  default: return 4;
735
  }
736
  }
hexstate_quantize.c CHANGED
@@ -5561,33 +5561,51 @@ write_fail:
5561
 
5562
  #include "iq2xs_grid.h"
5563
 
5564
- #define IQ2XS_NGRID 512
5565
  #define IQ2XS_TOPK 8
5566
  #define IQ2XS_NGROUP (QK_K / 8) /* 32 */
5567
  #define IQ2XS_NSUB (QK_K / 16) /* 16 */
5568
 
5569
- static float g_iq2xs_gridf[IQ2XS_NGRID][8];
 
 
 
 
 
 
 
 
 
 
 
 
5570
  static int g_iq2xs_grid_ready = 0;
 
 
5571
 
5572
  static void iq2xs_prepare_grid(void)
5573
  {
5574
  if (g_iq2xs_grid_ready) return;
5575
- for (int k = 0; k < IQ2XS_NGRID; k++)
5576
  for (int j = 0; j < 8; j++)
5577
  g_iq2xs_gridf[k][j] = (float)((iq2xs_grid[k] >> (8 * j)) & 0xFF);
 
 
 
5578
  g_iq2xs_grid_ready = 1;
5579
  }
5580
 
 
5581
  typedef struct {
5582
- uint16_t code; /* grid | signs7 << 9 */
5583
  float sse; /* Σ w (x − deq)² */
5584
  float deq[8];
5585
  } IQ2Cand;
5586
 
5587
  /* Top-K codewords for 8 weights at magnitude scale db (db > 0).
5588
- * Sign parity is enforced by the cheapest flip *per codeword*. */
5589
- static int iq2xs_group_candidates(const float *x, const float *w, float db,
5590
- int K, IQ2Cand *out)
 
5591
  {
5592
  float ax[8], wa[8];
5593
  uint8_t s = 0; int par = 0;
@@ -5596,9 +5614,10 @@ static int iq2xs_group_candidates(const float *x, const float *w, float db,
5596
  wa[i] = w[i];
5597
  if (x[i] < 0.0f) { s |= (uint8_t)(1u << i); par ^= 1; }
5598
  }
 
5599
  int n = 0;
5600
- for (int k = 0; k < IQ2XS_NGRID; k++) {
5601
- const float *g = g_iq2xs_gridf[k];
5602
  float e0 = 0.0f;
5603
  for (int i = 0; i < 8; i++) {
5604
  float d = ax[i] - db * g[i];
@@ -5628,7 +5647,7 @@ static int iq2xs_group_candidates(const float *x, const float *w, float db,
5628
  while (pos > 0 && out[pos - 1].sse > e) { out[pos] = out[pos - 1]; pos--; }
5629
  uint8_t sf = s;
5630
  if (flips[f] >= 0) sf ^= (uint8_t)(1u << flips[f]);
5631
- out[pos].code = (uint16_t)(k | ((sf & 127) << 9));
5632
  out[pos].sse = e;
5633
  for (int i = 0; i < 8; i++)
5634
  out[pos].deq[i] = db * g[i] * ((sf >> i) & 1 ? -1.0f : 1.0f);
@@ -5638,18 +5657,18 @@ static int iq2xs_group_candidates(const float *x, const float *w, float db,
5638
  return n;
5639
  }
5640
 
5641
- /* Decode one group from its 16-bit code at scale db. */
5642
- static inline void iq2xs_decode_group(uint16_t code, float db, float *deq)
5643
  {
5644
- const float *g = g_iq2xs_gridf[code & 511];
5645
- uint8_t signs = ksigns_iq2xs[code >> 9];
5646
  for (int j = 0; j < 8; j++)
5647
  deq[j] = db * g[j] * ((signs >> j) & 1 ? -1.0f : 1.0f);
5648
  }
5649
 
5650
  /* Best codes for a 16-weight sub-block at fixed db; returns weighted SSE. */
5651
- static float iq2xs_sub_pick(const float *x, const float *w, float db,
5652
- uint16_t code[2], float deq[16])
5653
  {
5654
  if (db <= 0.0f) {
5655
  code[0] = code[1] = 0;
@@ -5660,7 +5679,7 @@ static float iq2xs_sub_pick(const float *x, const float *w, float db,
5660
  IQ2Cand c;
5661
  float e = 0.0f;
5662
  for (int k = 0; k < 2; k++) {
5663
- iq2xs_group_candidates(x + 8 * k, w + 8 * k, db, 1, &c);
5664
  code[k] = c.code;
5665
  memcpy(deq + 8 * k, c.deq, sizeof(c.deq));
5666
  e += c.sse;
@@ -5669,18 +5688,18 @@ static float iq2xs_sub_pick(const float *x, const float *w, float db,
5669
  }
5670
 
5671
  /* Float scale search for one sub-block: candidate db grid + LS refit. */
5672
- static float iq2xs_sub_fit(const float *x, const float *w, float *db_out)
5673
  {
5674
  float amax = 0.0f;
5675
  for (int i = 0; i < 16; i++) amax = fmaxf(amax, fabsf(x[i]));
5676
  if (amax < 1e-12f) { *db_out = 0.0f; return 0.0f; }
5677
 
5678
  float best_e = 1e30f, best_db = amax / 43.0f;
5679
- uint16_t code[2]; float deq[16];
5680
  for (int is = -10; is <= 10; is++) {
5681
  /* amax lands on grid value 43·(1+0.035·is): includes clipped maxima */
5682
  float db = amax / (43.0f * (1.0f + 0.035f * (float)is));
5683
- float e = iq2xs_sub_pick(x, w, db, code, deq);
5684
  /* LS refit of db with codes fixed: deq = db·ĝ */
5685
  double num = 0.0, den = 0.0;
5686
  for (int i = 0; i < 16; i++) {
@@ -5690,7 +5709,7 @@ static float iq2xs_sub_fit(const float *x, const float *w, float *db_out)
5690
  }
5691
  if (den > 0.0 && num > 0.0) {
5692
  float db2 = (float)(num / den);
5693
- float e2 = iq2xs_sub_pick(x, w, db2, code, deq);
5694
  if (e2 < e) { e = e2; db = db2; }
5695
  }
5696
  if (e < best_e) { best_e = e; best_db = db; }
@@ -5701,32 +5720,59 @@ static float iq2xs_sub_fit(const float *x, const float *w, float *db_out)
5701
 
5702
  /* Whole-block encode at a given d: ls from float sub-scales, re-pick codes.
5703
  * Returns weighted SSE. */
5704
- static float iq2xs_block_at_d(const float *x, const float *w, float d,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5705
  const float *db_f, uint8_t ls[IQ2XS_NSUB],
5706
- uint16_t code[IQ2XS_NGROUP], float deq[QK_K])
5707
  {
5708
  float e = 0.0f;
 
5709
  for (int ib = 0; ib < IQ2XS_NSUB; ib++) {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5710
  int l = (d > 0.0f) ? gguf_nearest_int(db_f[ib] * 4.0f / d - 0.5f) : 0;
5711
  if (l < 0) l = 0; if (l > 15) l = 15;
5712
  ls[ib] = (uint8_t)l;
5713
  float db = d * ((float)l + 0.5f) * 0.25f;
5714
- e += iq2xs_sub_pick(x + 16 * ib, w + 16 * ib, db, code + 2 * ib, deq + 16 * ib);
5715
  }
5716
  return e;
5717
  }
5718
 
5719
- static inline float iq2xs_sub_db(float d, uint8_t ls)
5720
- {
5721
- return d * ((float)ls + 0.5f) * 0.25f;
5722
- }
5723
-
5724
  /* Greedy re-selection among top-K codewords per group on
5725
  * SSE + (λ_dc/n)(Σe + carry)² + (λ_vw/n) Σ_p (e_p + e_{p+128})²
5726
  * with a block SSE cap. This is the fold-through-codebook step. */
5727
- static void iq2xs_shape_block(const float *x, const float *w, float d,
5728
  const uint8_t ls[IQ2XS_NSUB],
5729
- uint16_t code[IQ2XS_NGROUP], float dc_carry)
5730
  {
5731
  if (HEX_DC_LAMBDA == 0.0f && HEX_VW_LAMBDA == 0.0f) return;
5732
 
@@ -5740,7 +5786,7 @@ static void iq2xs_shape_block(const float *x, const float *w, float d,
5740
  if (db <= 0.0f) { ncand[g] = 0; cur[g] = -1;
5741
  for (int j = 0; j < 8; j++) { e[8*g+j] = x[8*g+j]; sse += w[8*g+j]*x[8*g+j]*x[8*g+j]; dc += e[8*g+j]; }
5742
  continue; }
5743
- ncand[g] = iq2xs_group_candidates(x + 8*g, w + 8*g, db, IQ2XS_TOPK, cands[g]);
5744
  cur[g] = 0;
5745
  for (int c = 0; c < ncand[g]; c++)
5746
  if (cands[g][c].code == code[g]) { cur[g] = c; break; }
@@ -5795,29 +5841,70 @@ static void iq2xs_shape_block(const float *x, const float *w, float d,
5795
  }
5796
  }
5797
 
5798
- static void iq2xs_pack(BlockIQ2XS *b, float d, const uint8_t ls[IQ2XS_NSUB],
5799
- const uint16_t code[IQ2XS_NGROUP])
 
 
 
5800
  {
5801
- b->d = gguf_fp32_to_fp16(d);
5802
- for (int ib = 0; ib < IQ2XS_NSUB; ib += 2)
5803
- b->scales[ib / 2] = (uint8_t)(ls[ib] | (ls[ib + 1] << 4));
5804
- memcpy(b->qs, code, sizeof(uint16_t) * IQ2XS_NGROUP);
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5805
  }
5806
 
5807
- static void iq2xs_dequant_block(const BlockIQ2XS *b, float *out)
 
5808
  {
5809
- iq2xs_prepare_grid();
5810
- float d = gguf_fp16_to_fp32(b->d);
5811
- for (int g = 0; g < IQ2XS_NGROUP; g++) {
5812
- uint8_t ls = (g & 2) ? (b->scales[g >> 2] >> 4) : (b->scales[g >> 2] & 0xF);
5813
- iq2xs_decode_group(b->qs[g], iq2xs_sub_db(d, ls), out + 8 * g);
 
 
 
 
 
 
 
 
 
 
 
5814
  }
5815
  }
5816
 
5817
- static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5818
- BlockIQ2XS *output, float *out_total_error,
5819
- const float *imat_importance, int verbose,
5820
- int64_t row_width)
 
 
 
 
 
 
 
 
 
 
5821
  {
5822
  if (!weights || !output || n_elements <= 0 || n_elements % QK_K != 0) {
5823
  if (out_total_error) *out_total_error = -1.0f;
@@ -5825,6 +5912,7 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5825
  }
5826
  iq2xs_prepare_grid();
5827
  const int64_t n_blocks = n_elements / QK_K;
 
5828
  static const float d_mult[] = { 1.0f, 0.97f, 1.03f, 0.94f, 1.06f, 0.90f, 1.10f, 0.85f, 1.15f };
5829
  const int n_dm = (int)(sizeof(d_mult) / sizeof(d_mult[0]));
5830
 
@@ -5853,21 +5941,24 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5853
  /* 1. float sub-block scales */
5854
  float db_f[IQ2XS_NSUB], db_max = 0.0f;
5855
  for (int ib = 0; ib < IQ2XS_NSUB; ib++) {
5856
- iq2xs_sub_fit(x + 16 * ib, w + 16 * ib, &db_f[ib]);
5857
  db_max = fmaxf(db_max, db_f[ib]);
5858
  }
5859
- if (db_max <= 0.0f) { memset(&output[blk], 0, sizeof(BlockIQ2XS)); continue; }
5860
 
5861
  /* 2. d candidate search (fp16-exact), codes re-picked at quantised db */
5862
  float d0 = db_max * 4.0f / 15.5f;
5863
  float best_e = 1e30f, best_d = d0;
5864
  uint8_t ls[IQ2XS_NSUB], ls_t[IQ2XS_NSUB];
5865
- uint16_t code[IQ2XS_NGROUP], code_t[IQ2XS_NGROUP];
5866
  float deq[QK_K];
5867
- for (int c = 0; c < n_dm; c++) {
5868
- float d = gguf_fp16_to_fp32(gguf_fp32_to_fp16(d0 * d_mult[c]));
 
 
 
5869
  if (d <= 0.0f) continue;
5870
- float e = iq2xs_block_at_d(x, w, d, db_f, ls_t, code_t, deq);
5871
  if (e < best_e) { best_e = e; best_d = d;
5872
  memcpy(ls, ls_t, sizeof(ls)); memcpy(code, code_t, sizeof(code)); }
5873
  }
@@ -5875,7 +5966,7 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5875
  /* 3. per-sub-block ls ±1 coordinate descent at fixed d */
5876
  float sub_e[IQ2XS_NSUB];
5877
  for (int ib = 0; ib < IQ2XS_NSUB; ib++)
5878
- sub_e[ib] = iq2xs_sub_pick(x + 16*ib, w + 16*ib, iq2xs_sub_db(best_d, ls[ib]),
5879
  code + 2*ib, deq + 16*ib);
5880
  for (int it = 0; it < 3; it++) {
5881
  int moved = 0;
@@ -5883,8 +5974,8 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5883
  for (int dl = -1; dl <= 1; dl += 2) {
5884
  int l = (int)ls[ib] + dl;
5885
  if (l < 0 || l > 15) continue;
5886
- uint16_t ct[2]; float dq[16];
5887
- float e = iq2xs_sub_pick(x + 16*ib, w + 16*ib,
5888
  iq2xs_sub_db(best_d, (uint8_t)l), ct, dq);
5889
  if (e < sub_e[ib]) {
5890
  sub_e[ib] = e; ls[ib] = (uint8_t)l;
@@ -5899,24 +5990,24 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5899
 
5900
  /* 4. fold/DC shaping among near-equivalent codewords (carry = 0 here;
5901
  * the sequential pass below applies the true residual carry). */
5902
- iq2xs_shape_block(x, w, best_d, ls, code, 0.0f);
5903
 
5904
  float d_out = best_d;
5905
  if (inflate != 0.0f)
5906
  d_out = gguf_fp16_to_fp32(gguf_fp32_to_fp16(best_d * (1.0f + inflate)));
5907
- iq2xs_pack(&output[blk], d_out, ls, code);
5908
  }
5909
 
5910
  /* 5. sequential TRUE residual carry along each row */
5911
  if (HEX_DC_LAMBDA > 0.0f && g_hex_dc_decay > 0.0f) {
5912
  int64_t bpr = (row_width > 0 && row_width % QK_K == 0) ? row_width / QK_K : 0;
5913
  float rolling = 0.0f, w[QK_K], deq[QK_K];
5914
- uint8_t ls[IQ2XS_NSUB]; uint16_t code[IQ2XS_NGROUP];
5915
  for (int64_t blk = 0; blk < n_blocks; blk++) {
5916
  if (bpr > 0 && (blk % bpr) == 0) rolling = 0.0f;
5917
  const float *x = weights + blk * QK_K;
5918
- BlockIQ2XS *b = &output[blk];
5919
- float d = gguf_fp16_to_fp32(b->d);
5920
  if (d > 0.0f) {
5921
  float sigma2 = 0.0f;
5922
  for (int i = 0; i < QK_K; i++) sigma2 += x[i] * x[i];
@@ -5925,13 +6016,10 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5925
  float base = imat_importance ? imat_importance[blk * QK_K + i] : 1.0f;
5926
  w[i] = (wmode == 1) ? base * sqrtf(sigma2 + x[i] * x[i]) : base;
5927
  }
5928
- for (int ib = 0; ib < IQ2XS_NSUB; ib++)
5929
- ls[ib] = (ib & 1) ? (b->scales[ib >> 1] >> 4) : (b->scales[ib >> 1] & 0xF);
5930
- memcpy(code, b->qs, sizeof(code));
5931
- iq2xs_shape_block(x, w, d, ls, code, g_hex_dc_decay * rolling);
5932
- memcpy(b->qs, code, sizeof(code));
5933
  }
5934
- iq2xs_dequant_block(b, deq);
5935
  float r = 0.0f;
5936
  for (int i = 0; i < QK_K; i++) r += x[i] - deq[i];
5937
  rolling = r;
@@ -5943,13 +6031,13 @@ static void quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_elements,
5943
  #pragma omp parallel for reduction(+:tot)
5944
  for (int64_t blk = 0; blk < n_blocks; blk++) {
5945
  float deq[QK_K];
5946
- iq2xs_dequant_block(&output[blk], deq);
5947
  const float *x = weights + blk * QK_K;
5948
  for (int i = 0; i < QK_K; i++) { double e = x[i] - deq[i]; tot += e * e; }
5949
  }
5950
  if (out_total_error) *out_total_error = (float)tot;
5951
  if (verbose)
5952
- printf(" [IQ2_XS·Sieve] blocks=%lld rmse=%.4e\n", (long long)n_blocks,
5953
  sqrt(tot / (double)n_elements));
5954
  }
5955
 
@@ -6069,15 +6157,35 @@ void hexstate_quantize_tensor_iq2_xs_hpc(const float *weights, int64_t n_element
6069
  int64_t row_width)
6070
  {
6071
  hexstate_init();
6072
- quantize_tensor_iq2_xs_hpc(weights, n_elements, (BlockIQ2XS *)output,
6073
- out_error, imat_importance, verbose, row_width);
6074
  }
6075
 
6076
  void hexstate_dequant_iq2_xs(const void *blocks, int64_t n_blocks, float *out)
6077
  {
6078
- const BlockIQ2XS *b = (const BlockIQ2XS *)blocks;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6079
  for (int64_t i = 0; i < n_blocks; i++)
6080
- iq2xs_dequant_block(&b[i], out + i * QK_K);
6081
  }
6082
 
6083
  #ifndef HEXSTATE_LIBRARY
 
5561
 
5562
  #include "iq2xs_grid.h"
5563
 
 
5564
  #define IQ2XS_TOPK 8
5565
  #define IQ2XS_NGROUP (QK_K / 8) /* 32 */
5566
  #define IQ2XS_NSUB (QK_K / 16) /* 16 */
5567
 
5568
+ /* Codebook spec. IQ2_XS: 512 codewords, 7 sign bits + parity (even number
5569
+ * of negatives). IQ2_S: 1024 codewords, 8 free sign bits. Same alphabet
5570
+ * {8,25,43}, same sub-block scale law, so one search serves both. */
5571
+ typedef struct {
5572
+ int ngrid;
5573
+ const float (*g)[8];
5574
+ int parity;
5575
+ int block_bytes;
5576
+ const char *name;
5577
+ } IQ2Book;
5578
+
5579
+ static float g_iq2xs_gridf[512][8];
5580
+ static float g_iq2s_gridf[1024][8];
5581
  static int g_iq2xs_grid_ready = 0;
5582
+ static const IQ2Book g_iq2_book_xs = { 512, g_iq2xs_gridf, 1, 74, "IQ2_XS" };
5583
+ static const IQ2Book g_iq2_book_s = { 1024, g_iq2s_gridf, 0, 82, "IQ2_S" };
5584
 
5585
  static void iq2xs_prepare_grid(void)
5586
  {
5587
  if (g_iq2xs_grid_ready) return;
5588
+ for (int k = 0; k < 512; k++)
5589
  for (int j = 0; j < 8; j++)
5590
  g_iq2xs_gridf[k][j] = (float)((iq2xs_grid[k] >> (8 * j)) & 0xFF);
5591
+ for (int k = 0; k < 1024; k++)
5592
+ for (int j = 0; j < 8; j++)
5593
+ g_iq2s_gridf[k][j] = (float)((iq2s_grid[k] >> (8 * j)) & 0xFF);
5594
  g_iq2xs_grid_ready = 1;
5595
  }
5596
 
5597
+ /* Internal code: grid index | signs8 << 16 (all 8 sign bits explicit). */
5598
  typedef struct {
5599
+ uint32_t code;
5600
  float sse; /* Σ w (x − deq)² */
5601
  float deq[8];
5602
  } IQ2Cand;
5603
 
5604
  /* Top-K codewords for 8 weights at magnitude scale db (db > 0).
5605
+ * For parity codebooks the parity is enforced by the cheapest flip *per
5606
+ * codeword*. */
5607
+ static int iq2xs_group_candidates(const IQ2Book *bk, const float *x, const float *w,
5608
+ float db, int K, IQ2Cand *out)
5609
  {
5610
  float ax[8], wa[8];
5611
  uint8_t s = 0; int par = 0;
 
5614
  wa[i] = w[i];
5615
  if (x[i] < 0.0f) { s |= (uint8_t)(1u << i); par ^= 1; }
5616
  }
5617
+ if (!bk->parity) par = 0;
5618
  int n = 0;
5619
+ for (int k = 0; k < bk->ngrid; k++) {
5620
+ const float *g = bk->g[k];
5621
  float e0 = 0.0f;
5622
  for (int i = 0; i < 8; i++) {
5623
  float d = ax[i] - db * g[i];
 
5647
  while (pos > 0 && out[pos - 1].sse > e) { out[pos] = out[pos - 1]; pos--; }
5648
  uint8_t sf = s;
5649
  if (flips[f] >= 0) sf ^= (uint8_t)(1u << flips[f]);
5650
+ out[pos].code = (uint32_t)k | ((uint32_t)sf << 16);
5651
  out[pos].sse = e;
5652
  for (int i = 0; i < 8; i++)
5653
  out[pos].deq[i] = db * g[i] * ((sf >> i) & 1 ? -1.0f : 1.0f);
 
5657
  return n;
5658
  }
5659
 
5660
+ /* Decode one group from its internal code at scale db. */
5661
+ static inline void iq2xs_decode_group(const IQ2Book *bk, uint32_t code, float db, float *deq)
5662
  {
5663
+ const float *g = bk->g[code & 0xFFFF];
5664
+ uint8_t signs = (uint8_t)(code >> 16);
5665
  for (int j = 0; j < 8; j++)
5666
  deq[j] = db * g[j] * ((signs >> j) & 1 ? -1.0f : 1.0f);
5667
  }
5668
 
5669
  /* Best codes for a 16-weight sub-block at fixed db; returns weighted SSE. */
5670
+ static float iq2xs_sub_pick(const IQ2Book *bk, const float *x, const float *w, float db,
5671
+ uint32_t code[2], float deq[16])
5672
  {
5673
  if (db <= 0.0f) {
5674
  code[0] = code[1] = 0;
 
5679
  IQ2Cand c;
5680
  float e = 0.0f;
5681
  for (int k = 0; k < 2; k++) {
5682
+ iq2xs_group_candidates(bk, x + 8 * k, w + 8 * k, db, 1, &c);
5683
  code[k] = c.code;
5684
  memcpy(deq + 8 * k, c.deq, sizeof(c.deq));
5685
  e += c.sse;
 
5688
  }
5689
 
5690
  /* Float scale search for one sub-block: candidate db grid + LS refit. */
5691
+ static float iq2xs_sub_fit(const IQ2Book *bk, const float *x, const float *w, float *db_out)
5692
  {
5693
  float amax = 0.0f;
5694
  for (int i = 0; i < 16; i++) amax = fmaxf(amax, fabsf(x[i]));
5695
  if (amax < 1e-12f) { *db_out = 0.0f; return 0.0f; }
5696
 
5697
  float best_e = 1e30f, best_db = amax / 43.0f;
5698
+ uint32_t code[2]; float deq[16];
5699
  for (int is = -10; is <= 10; is++) {
5700
  /* amax lands on grid value 43·(1+0.035·is): includes clipped maxima */
5701
  float db = amax / (43.0f * (1.0f + 0.035f * (float)is));
5702
+ float e = iq2xs_sub_pick(bk, x, w, db, code, deq);
5703
  /* LS refit of db with codes fixed: deq = db·ĝ */
5704
  double num = 0.0, den = 0.0;
5705
  for (int i = 0; i < 16; i++) {
 
5709
  }
5710
  if (den > 0.0 && num > 0.0) {
5711
  float db2 = (float)(num / den);
5712
+ float e2 = iq2xs_sub_pick(bk, x, w, db2, code, deq);
5713
  if (e2 < e) { e = e2; db = db2; }
5714
  }
5715
  if (e < best_e) { best_e = e; best_db = db; }
 
5720
 
5721
  /* Whole-block encode at a given d: ls from float sub-scales, re-pick codes.
5722
  * Returns weighted SSE. */
5723
+ static inline float iq2xs_sub_db(float d, uint8_t ls)
5724
+ {
5725
+ return d * ((float)ls + 0.5f) * 0.25f;
5726
+ }
5727
+
5728
+ /* HEX_IQ2_EXACT=1: exhaustive ls ∈ 0..15 per sub-block and a dense d scan.
5729
+ * Given d the sub-blocks separate, and given ls the two groups separate and
5730
+ * are already solved exactly, so this is the true optimum of the format for
5731
+ * the weighted-SSE objective — used to measure how far the fast path sits
5732
+ * from the floor. ~40× slower. */
5733
+ static int iq2xs_exact_mode(void)
5734
+ {
5735
+ static int mode = -1;
5736
+ if (mode < 0) { const char *s = getenv("HEX_IQ2_EXACT"); mode = (s && atoi(s)) ? 1 : 0; }
5737
+ return mode;
5738
+ }
5739
+
5740
+ static float iq2xs_block_at_d(const IQ2Book *bk, const float *x, const float *w, float d,
5741
  const float *db_f, uint8_t ls[IQ2XS_NSUB],
5742
+ uint32_t code[IQ2XS_NGROUP], float deq[QK_K])
5743
  {
5744
  float e = 0.0f;
5745
+ const int exact = iq2xs_exact_mode();
5746
  for (int ib = 0; ib < IQ2XS_NSUB; ib++) {
5747
+ if (exact && d > 0.0f) {
5748
+ float best = 1e30f; uint32_t ct[2]; float dq[16];
5749
+ for (int l = 0; l < 16; l++) {
5750
+ float el = iq2xs_sub_pick(bk, x + 16 * ib, w + 16 * ib,
5751
+ iq2xs_sub_db(d, (uint8_t)l), ct, dq);
5752
+ if (el < best) {
5753
+ best = el; ls[ib] = (uint8_t)l;
5754
+ code[2*ib] = ct[0]; code[2*ib+1] = ct[1];
5755
+ memcpy(deq + 16 * ib, dq, sizeof(dq));
5756
+ }
5757
+ }
5758
+ e += best;
5759
+ continue;
5760
+ }
5761
  int l = (d > 0.0f) ? gguf_nearest_int(db_f[ib] * 4.0f / d - 0.5f) : 0;
5762
  if (l < 0) l = 0; if (l > 15) l = 15;
5763
  ls[ib] = (uint8_t)l;
5764
  float db = d * ((float)l + 0.5f) * 0.25f;
5765
+ e += iq2xs_sub_pick(bk, x + 16 * ib, w + 16 * ib, db, code + 2 * ib, deq + 16 * ib);
5766
  }
5767
  return e;
5768
  }
5769
 
 
 
 
 
 
5770
  /* Greedy re-selection among top-K codewords per group on
5771
  * SSE + (λ_dc/n)(Σe + carry)² + (λ_vw/n) Σ_p (e_p + e_{p+128})²
5772
  * with a block SSE cap. This is the fold-through-codebook step. */
5773
+ static void iq2xs_shape_block(const IQ2Book *bk, const float *x, const float *w, float d,
5774
  const uint8_t ls[IQ2XS_NSUB],
5775
+ uint32_t code[IQ2XS_NGROUP], float dc_carry)
5776
  {
5777
  if (HEX_DC_LAMBDA == 0.0f && HEX_VW_LAMBDA == 0.0f) return;
5778
 
 
5786
  if (db <= 0.0f) { ncand[g] = 0; cur[g] = -1;
5787
  for (int j = 0; j < 8; j++) { e[8*g+j] = x[8*g+j]; sse += w[8*g+j]*x[8*g+j]*x[8*g+j]; dc += e[8*g+j]; }
5788
  continue; }
5789
+ ncand[g] = iq2xs_group_candidates(bk, x + 8*g, w + 8*g, db, IQ2XS_TOPK, cands[g]);
5790
  cur[g] = 0;
5791
  for (int c = 0; c < ncand[g]; c++)
5792
  if (cands[g][c].code == code[g]) { cur[g] = c; break; }
 
5841
  }
5842
  }
5843
 
5844
+ /* Pack / unpack to the ggml block layouts. Scales are identical in both;
5845
+ * XS stores idx(9) | signs7(7) per uint16 with the 8th sign as parity,
5846
+ * S stores idx low byte, a separate sign byte, and 2 high idx bits in qh. */
5847
+ static void iq2_pack(const IQ2Book *bk, void *blk, float d, const uint8_t ls[IQ2XS_NSUB],
5848
+ const uint32_t code[IQ2XS_NGROUP])
5849
  {
5850
+ if (bk->parity) {
5851
+ BlockIQ2XS *b = (BlockIQ2XS *)blk;
5852
+ b->d = gguf_fp32_to_fp16(d);
5853
+ for (int ib = 0; ib < IQ2XS_NSUB; ib += 2)
5854
+ b->scales[ib / 2] = (uint8_t)(ls[ib] | (ls[ib + 1] << 4));
5855
+ for (int g = 0; g < IQ2XS_NGROUP; g++)
5856
+ b->qs[g] = (uint16_t)((code[g] & 511) | (((code[g] >> 16) & 127) << 9));
5857
+ } else {
5858
+ BlockIQ2S *b = (BlockIQ2S *)blk;
5859
+ b->d = gguf_fp32_to_fp16(d);
5860
+ memset(b->qh, 0, sizeof(b->qh));
5861
+ for (int ib = 0; ib < IQ2XS_NSUB; ib += 2)
5862
+ b->scales[ib / 2] = (uint8_t)(ls[ib] | (ls[ib + 1] << 4));
5863
+ for (int g = 0; g < IQ2XS_NGROUP; g++) {
5864
+ uint32_t idx = code[g] & 1023;
5865
+ b->qs[g] = (uint8_t)(idx & 0xFF);
5866
+ b->qs[IQ2XS_NGROUP + g] = (uint8_t)(code[g] >> 16);
5867
+ b->qh[g >> 2] |= (uint8_t)(((idx >> 8) & 3) << (2 * (g & 3)));
5868
+ }
5869
+ }
5870
  }
5871
 
5872
+ static float iq2_unpack(const IQ2Book *bk, const void *blk, uint8_t ls[IQ2XS_NSUB],
5873
+ uint32_t code[IQ2XS_NGROUP])
5874
  {
5875
+ if (bk->parity) {
5876
+ const BlockIQ2XS *b = (const BlockIQ2XS *)blk;
5877
+ for (int ib = 0; ib < IQ2XS_NSUB; ib++)
5878
+ ls[ib] = (ib & 1) ? (b->scales[ib >> 1] >> 4) : (b->scales[ib >> 1] & 0xF);
5879
+ for (int g = 0; g < IQ2XS_NGROUP; g++)
5880
+ code[g] = (uint32_t)(b->qs[g] & 511) | ((uint32_t)ksigns_iq2xs[b->qs[g] >> 9] << 16);
5881
+ return gguf_fp16_to_fp32(b->d);
5882
+ } else {
5883
+ const BlockIQ2S *b = (const BlockIQ2S *)blk;
5884
+ for (int ib = 0; ib < IQ2XS_NSUB; ib++)
5885
+ ls[ib] = (ib & 1) ? (b->scales[ib >> 1] >> 4) : (b->scales[ib >> 1] & 0xF);
5886
+ for (int g = 0; g < IQ2XS_NGROUP; g++) {
5887
+ uint32_t idx = (uint32_t)b->qs[g] | ((((uint32_t)b->qh[g >> 2] >> (2 * (g & 3))) & 3) << 8);
5888
+ code[g] = idx | ((uint32_t)b->qs[IQ2XS_NGROUP + g] << 16);
5889
+ }
5890
+ return gguf_fp16_to_fp32(b->d);
5891
  }
5892
  }
5893
 
5894
+ static void iq2_dequant_block(const IQ2Book *bk, const void *blk, float *out)
5895
+ {
5896
+ iq2xs_prepare_grid();
5897
+ uint8_t ls[IQ2XS_NSUB]; uint32_t code[IQ2XS_NGROUP];
5898
+ float d = iq2_unpack(bk, blk, ls, code);
5899
+ for (int g = 0; g < IQ2XS_NGROUP; g++)
5900
+ iq2xs_decode_group(bk, code[g], iq2xs_sub_db(d, ls[g >> 1]), out + 8 * g);
5901
+ }
5902
+
5903
+ static void quantize_tensor_iq2_hpc(const IQ2Book *bk,
5904
+ const float *weights, int64_t n_elements,
5905
+ uint8_t *output, float *out_total_error,
5906
+ const float *imat_importance, int verbose,
5907
+ int64_t row_width)
5908
  {
5909
  if (!weights || !output || n_elements <= 0 || n_elements % QK_K != 0) {
5910
  if (out_total_error) *out_total_error = -1.0f;
 
5912
  }
5913
  iq2xs_prepare_grid();
5914
  const int64_t n_blocks = n_elements / QK_K;
5915
+ const int BB = bk->block_bytes;
5916
  static const float d_mult[] = { 1.0f, 0.97f, 1.03f, 0.94f, 1.06f, 0.90f, 1.10f, 0.85f, 1.15f };
5917
  const int n_dm = (int)(sizeof(d_mult) / sizeof(d_mult[0]));
5918
 
 
5941
  /* 1. float sub-block scales */
5942
  float db_f[IQ2XS_NSUB], db_max = 0.0f;
5943
  for (int ib = 0; ib < IQ2XS_NSUB; ib++) {
5944
+ iq2xs_sub_fit(bk, x + 16 * ib, w + 16 * ib, &db_f[ib]);
5945
  db_max = fmaxf(db_max, db_f[ib]);
5946
  }
5947
+ if (db_max <= 0.0f) { memset(output + blk * BB, 0, (size_t)BB); continue; }
5948
 
5949
  /* 2. d candidate search (fp16-exact), codes re-picked at quantised db */
5950
  float d0 = db_max * 4.0f / 15.5f;
5951
  float best_e = 1e30f, best_d = d0;
5952
  uint8_t ls[IQ2XS_NSUB], ls_t[IQ2XS_NSUB];
5953
+ uint32_t code[IQ2XS_NGROUP], code_t[IQ2XS_NGROUP];
5954
  float deq[QK_K];
5955
+ const int exact = iq2xs_exact_mode();
5956
+ const int n_dc = exact ? 61 : n_dm; /* exact: d0·[0.70..1.30] step 0.01 */
5957
+ for (int c = 0; c < n_dc; c++) {
5958
+ float mult = exact ? (0.70f + 0.01f * (float)c) : d_mult[c];
5959
+ float d = gguf_fp16_to_fp32(gguf_fp32_to_fp16(d0 * mult));
5960
  if (d <= 0.0f) continue;
5961
+ float e = iq2xs_block_at_d(bk, x, w, d, db_f, ls_t, code_t, deq);
5962
  if (e < best_e) { best_e = e; best_d = d;
5963
  memcpy(ls, ls_t, sizeof(ls)); memcpy(code, code_t, sizeof(code)); }
5964
  }
 
5966
  /* 3. per-sub-block ls ±1 coordinate descent at fixed d */
5967
  float sub_e[IQ2XS_NSUB];
5968
  for (int ib = 0; ib < IQ2XS_NSUB; ib++)
5969
+ sub_e[ib] = iq2xs_sub_pick(bk, x + 16*ib, w + 16*ib, iq2xs_sub_db(best_d, ls[ib]),
5970
  code + 2*ib, deq + 16*ib);
5971
  for (int it = 0; it < 3; it++) {
5972
  int moved = 0;
 
5974
  for (int dl = -1; dl <= 1; dl += 2) {
5975
  int l = (int)ls[ib] + dl;
5976
  if (l < 0 || l > 15) continue;
5977
+ uint32_t ct[2]; float dq[16];
5978
+ float e = iq2xs_sub_pick(bk, x + 16*ib, w + 16*ib,
5979
  iq2xs_sub_db(best_d, (uint8_t)l), ct, dq);
5980
  if (e < sub_e[ib]) {
5981
  sub_e[ib] = e; ls[ib] = (uint8_t)l;
 
5990
 
5991
  /* 4. fold/DC shaping among near-equivalent codewords (carry = 0 here;
5992
  * the sequential pass below applies the true residual carry). */
5993
+ iq2xs_shape_block(bk, x, w, best_d, ls, code, 0.0f);
5994
 
5995
  float d_out = best_d;
5996
  if (inflate != 0.0f)
5997
  d_out = gguf_fp16_to_fp32(gguf_fp32_to_fp16(best_d * (1.0f + inflate)));
5998
+ iq2_pack(bk, output + blk * BB, d_out, ls, code);
5999
  }
6000
 
6001
  /* 5. sequential TRUE residual carry along each row */
6002
  if (HEX_DC_LAMBDA > 0.0f && g_hex_dc_decay > 0.0f) {
6003
  int64_t bpr = (row_width > 0 && row_width % QK_K == 0) ? row_width / QK_K : 0;
6004
  float rolling = 0.0f, w[QK_K], deq[QK_K];
6005
+ uint8_t ls[IQ2XS_NSUB]; uint32_t code[IQ2XS_NGROUP];
6006
  for (int64_t blk = 0; blk < n_blocks; blk++) {
6007
  if (bpr > 0 && (blk % bpr) == 0) rolling = 0.0f;
6008
  const float *x = weights + blk * QK_K;
6009
+ uint8_t *b = output + blk * BB;
6010
+ float d = iq2_unpack(bk, b, ls, code);
6011
  if (d > 0.0f) {
6012
  float sigma2 = 0.0f;
6013
  for (int i = 0; i < QK_K; i++) sigma2 += x[i] * x[i];
 
6016
  float base = imat_importance ? imat_importance[blk * QK_K + i] : 1.0f;
6017
  w[i] = (wmode == 1) ? base * sqrtf(sigma2 + x[i] * x[i]) : base;
6018
  }
6019
+ iq2xs_shape_block(bk, x, w, d, ls, code, g_hex_dc_decay * rolling);
6020
+ iq2_pack(bk, b, d, ls, code);
 
 
 
6021
  }
6022
+ iq2_dequant_block(bk, b, deq);
6023
  float r = 0.0f;
6024
  for (int i = 0; i < QK_K; i++) r += x[i] - deq[i];
6025
  rolling = r;
 
6031
  #pragma omp parallel for reduction(+:tot)
6032
  for (int64_t blk = 0; blk < n_blocks; blk++) {
6033
  float deq[QK_K];
6034
+ iq2_dequant_block(bk, output + blk * BB, deq);
6035
  const float *x = weights + blk * QK_K;
6036
  for (int i = 0; i < QK_K; i++) { double e = x[i] - deq[i]; tot += e * e; }
6037
  }
6038
  if (out_total_error) *out_total_error = (float)tot;
6039
  if (verbose)
6040
+ printf(" [%s·Sieve] blocks=%lld rmse=%.4e\n", bk->name, (long long)n_blocks,
6041
  sqrt(tot / (double)n_elements));
6042
  }
6043
 
 
6157
  int64_t row_width)
6158
  {
6159
  hexstate_init();
6160
+ quantize_tensor_iq2_hpc(&g_iq2_book_xs, weights, n_elements, (uint8_t *)output,
6161
+ out_error, imat_importance, verbose, row_width);
6162
  }
6163
 
6164
  void hexstate_dequant_iq2_xs(const void *blocks, int64_t n_blocks, float *out)
6165
  {
6166
+ const uint8_t *b = (const uint8_t *)blocks;
6167
+ for (int64_t i = 0; i < n_blocks; i++)
6168
+ iq2_dequant_block(&g_iq2_book_xs, b + i * sizeof(BlockIQ2XS), out + i * QK_K);
6169
+ }
6170
+
6171
+ /* IQ2_S (82 bytes / 256 weights): 1024 codewords, 8 free sign bits. */
6172
+ int hexstate_iq2s_block_bytes(void) { return (int)sizeof(BlockIQ2S); }
6173
+
6174
+ void hexstate_quantize_tensor_iq2_s_hpc(const float *weights, int64_t n_elements,
6175
+ void *output, float *out_error,
6176
+ const float *imat_importance, int verbose,
6177
+ int64_t row_width)
6178
+ {
6179
+ hexstate_init();
6180
+ quantize_tensor_iq2_hpc(&g_iq2_book_s, weights, n_elements, (uint8_t *)output,
6181
+ out_error, imat_importance, verbose, row_width);
6182
+ }
6183
+
6184
+ void hexstate_dequant_iq2_s(const void *blocks, int64_t n_blocks, float *out)
6185
+ {
6186
+ const uint8_t *b = (const uint8_t *)blocks;
6187
  for (int64_t i = 0; i < n_blocks; i++)
6188
+ iq2_dequant_block(&g_iq2_book_s, b + i * sizeof(BlockIQ2S), out + i * QK_K);
6189
  }
6190
 
6191
  #ifndef HEXSTATE_LIBRARY
hexstate_requantize.py CHANGED
@@ -118,10 +118,13 @@ def _load_hexstate_lib():
118
  lib.hexstate_set_sse_budget.restype = None
119
  lib.hexstate_set_sse_budget.argtypes = [ctypes.c_float]
120
 
121
- # IQ2_XS (E8 codebook) quantizer + dequant
122
- if hasattr(lib, 'hexstate_quantize_tensor_iq2_xs_hpc'):
123
- lib.hexstate_quantize_tensor_iq2_xs_hpc.restype = None
124
- lib.hexstate_quantize_tensor_iq2_xs_hpc.argtypes = [
 
 
 
125
  ctypes.POINTER(ctypes.c_float), # weights
126
  ctypes.c_int64, # n_elements
127
  ctypes.c_void_p, # output
@@ -130,9 +133,9 @@ def _load_hexstate_lib():
130
  ctypes.c_int, # verbose
131
  ctypes.c_int64, # row_width
132
  ]
133
- lib.hexstate_dequant_iq2_xs.restype = None
134
- lib.hexstate_dequant_iq2_xs.argtypes = [
135
- ctypes.c_void_p, ctypes.c_int64, ctypes.POINTER(ctypes.c_float)]
136
 
137
  lib.hexstate_init()
138
  dc_l = os.environ.get('HEX_DC_LAMBDA')
@@ -148,7 +151,7 @@ def _load_hexstate_lib():
148
  # IQ2_XS has no per-sub-block offset, so cancelling DC means swapping
149
  # codewords — it needs ~2e-2 (≈ +0.2% RMSE) to be effective.
150
  budget = os.environ.get('HEX_SSE_BUDGET')
151
- if budget is None and LOWBIT_FORMAT == 'iq2xs':
152
  budget = '0.02'
153
  if budget is not None and hasattr(lib, 'hexstate_set_sse_budget'):
154
  lib.hexstate_set_sse_budget(ctypes.c_float(float(budget)))
@@ -349,30 +352,44 @@ GGML_TYPE_Q4_0 = 2
349
  GGML_TYPE_Q8_0 = 8
350
  GGML_TYPE_Q2_K = 10
351
  GGML_TYPE_IQ2_XS = 17
 
352
  GGML_TYPE_BF16 = 30
353
 
354
  IQ2_XS_BLOCK_BYTES = 74 # d(fp16) + 32×u16 codes + 8 scale bytes
355
-
356
- # Low-bit target for the "Q2_K plan" tensors: 'q2k' (default) or 'iq2xs'
357
- # (--iq2xs: E8 codebook, 2.3125 bpw, native llama.cpp decode).
 
 
 
 
 
 
358
  LOWBIT_FORMAT = 'q2k'
359
 
 
 
 
 
 
 
 
360
  TYPE_NAME = {
361
  0: "F32", 1: "F16", 2: "Q4_0", 3: "Q4_1", 6: "Q5_0", 7: "Q5_1",
362
  8: "Q8_0", 9: "Q8_1", 10: "Q2_K", 11: "Q3_K", 12: "Q4_K",
363
- 13: "Q5_K", 14: "Q6_K", 15: "Q8_K", 17: "IQ2_XS", 30: "BF16",
364
  }
365
 
366
  # Block sizes and byte sizes for each type
367
  TYPE_BLOCK_SIZE = {
368
  0: 1, 1: 1, 2: 32, 3: 32, 6: 32, 7: 32,
369
  8: 32, 9: 32, 10: 256, 11: 256, 12: 256,
370
- 13: 256, 14: 256, 15: 256, 17: 256, 30: 1,
371
  }
372
  TYPE_BLOCK_BYTES = {
373
  0: 4, 1: 2, 2: 18, 3: 20, 6: 20, 7: 22,
374
  8: 34, 9: 36, 10: 84, 11: 110, 12: 144,
375
- 13: 176, 14: 210, 15: 292, 17: IQ2_XS_BLOCK_BYTES, 30: 2,
376
  }
377
 
378
 
@@ -835,25 +852,27 @@ def _copy_bytes(fin, fout, abs_offset, n_bytes):
835
  return written
836
 
837
 
838
- def quantize_tensor_iq2xs_hpc(f32_data, importance=None, row_width=0):
839
- """IQ2_XS (E8 codebook, 74 B / 256 w) via the HPC C library.
840
  Returns (bytes, n_blocks, dequant_f32)."""
 
 
841
  lib = _load_hexstate_lib()
842
- if lib is None or not hasattr(lib, 'hexstate_quantize_tensor_iq2_xs_hpc'):
843
- raise RuntimeError('libhexstate_q2k.so lacks IQ2_XS support — rebuild')
844
  f32 = np.ascontiguousarray(f32_data, dtype=np.float32).reshape(-1)
845
  n = int(f32.size)
846
  if n % QK_K != 0:
847
- raise ValueError(f'IQ2_XS needs a multiple of {QK_K} elements, got {n}')
848
  n_blocks = n // QK_K
849
- out = np.zeros(n_blocks * IQ2_XS_BLOCK_BYTES, dtype=np.uint8)
850
  err = ctypes.c_float(0.0)
851
  imat_ptr = None
852
  if importance is not None:
853
  imat_c = np.ascontiguousarray(importance, dtype=np.float32).reshape(-1)
854
  if imat_c.size == n:
855
  imat_ptr = imat_c.ctypes.data_as(ctypes.POINTER(ctypes.c_float))
856
- lib.hexstate_quantize_tensor_iq2_xs_hpc(
857
  f32.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
858
  ctypes.c_int64(n),
859
  out.ctypes.data_as(ctypes.c_void_p),
@@ -863,12 +882,68 @@ def quantize_tensor_iq2xs_hpc(f32_data, importance=None, row_width=0):
863
  ctypes.c_int64(int(row_width)),
864
  )
865
  deq = np.zeros(n, dtype=np.float32)
866
- lib.hexstate_dequant_iq2_xs(
867
  out.ctypes.data_as(ctypes.c_void_p), ctypes.c_int64(n_blocks),
868
  deq.ctypes.data_as(ctypes.POINTER(ctypes.c_float)))
869
  return out.tobytes(), n_blocks, deq
870
 
871
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
872
  def _stream_quantize(fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc):
873
  """Quantize one tensor in row chunks. kind: 'q2k' | 'iq2xs' | 'q4' | 'q8'.
874
  Returns (n_out_bytes, rmse_or_None, sigma_or_None).
@@ -883,7 +958,7 @@ def _stream_quantize(fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc):
883
 
884
  d0 = int(ti['dims'][0])
885
  n_rows = int(ti['n_elements']) // d0
886
- align = QK_K if kind in ('q2k', 'iq2xs') else 32
887
  if d0 % align != 0:
888
  raise ValueError(f'{ti["name"]} dim0={d0} not aligned to {align}')
889
 
@@ -901,9 +976,9 @@ def _stream_quantize(fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc):
901
  total_ss += float(np.vdot(f32, f32))
902
  total_n += n_valid
903
 
904
- if kind == 'iq2xs':
905
- qbytes, n_blocks, deq = quantize_tensor_iq2xs_hpc(
906
- f32, importance=imp, row_width=d0)
907
  fout.write(qbytes)
908
  written += len(qbytes)
909
  diff = f32.reshape(-1)[:n_valid] - deq[:n_valid]
@@ -1074,13 +1149,18 @@ def should_quantize(name, n_dims, dims, tied_embeddings=False):
1074
  def main():
1075
  if len(sys.argv) < 3:
1076
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
1077
- " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all] [--iq2xs]")
 
1078
  print(" --iq2xs low-bit tensors → IQ2_XS (E8 codebook, 2.3125 bpw) instead of Q2_K")
 
 
 
 
1079
  print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
1080
  print(" HEX_DC_LAMBDA DC residual weight (default 1)")
1081
  print(" HEX_VW_LAMBDA vesica weight (default 1)")
1082
  print(" HEX_DC_DECAY rolling residual carry 0..1 (default 0.85)")
1083
- print(" HEX_SSE_BUDGET relative SSE the shaper may spend (Q2_K 5e-4, IQ2_XS 2e-2)")
1084
  sys.exit(1)
1085
 
1086
  global LOWBIT_FORMAT
@@ -1092,11 +1172,17 @@ def main():
1092
  keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
1093
  if '--iq2xs' in sys.argv:
1094
  LOWBIT_FORMAT = 'iq2xs'
1095
- lowbit_kind = LOWBIT_FORMAT # 'q2k' | 'iq2xs'
1096
- lowbit_name = 'IQ2_XS' if lowbit_kind == 'iq2xs' else 'Q2_K'
1097
- lowbit_type = GGML_TYPE_IQ2_XS if lowbit_kind == 'iq2xs' else GGML_TYPE_Q2_K
1098
- lowbit_bytes = IQ2_XS_BLOCK_BYTES if lowbit_kind == 'iq2xs' else 84
1099
- lowbit_file_type = 20 if lowbit_kind == 'iq2xs' else 10 # LLAMA_FTYPE_MOSTLY_*
 
 
 
 
 
 
1100
 
1101
  # Check for imatrix
1102
  imatrix_data = None
@@ -1112,19 +1198,25 @@ def main():
1112
 
1113
  # Check for HPC C library
1114
  use_hpc = _load_hexstate_lib() is not None
1115
- if lowbit_kind == 'iq2xs':
1116
  lib = _load_hexstate_lib()
1117
- if lib is None or not hasattr(lib, 'hexstate_quantize_tensor_iq2_xs_hpc'):
1118
- print(" ERROR: --iq2xs needs libhexstate_q2k.so with IQ2_XS support "
1119
- "(make -f makefile.quantize.c)")
 
 
1120
  sys.exit(1)
1121
 
1122
  print()
1123
  print(" ╔════════════════════════════════════════════════════════════════╗")
1124
  print(" ║ HExState GGUF Re-Quantizer ║")
1125
- print(f" ║ GGUF → {lowbit_name:6s} GGUF with metadata passthrough ║")
1126
  if lowbit_kind == 'iq2xs':
1127
  print(" ║ Low-bit: IQ2_XS E8 codebook · fold/DC shaping · 2.3125 bpw ║")
 
 
 
 
1128
  if q2all:
1129
  print(" ║ Mode: --q2all ALL eligible tensors → Q2_K (test mode) ║")
1130
  if use_hpc and imatrix_data:
@@ -1268,6 +1360,10 @@ def main():
1268
  # ── Compute output tensor sizes and offsets ──
1269
  out_tensor_infos = []
1270
  out_data_offset = 0
 
 
 
 
1271
 
1272
  for i, ti in enumerate(tensor_infos):
1273
  if quant_plan[i]:
@@ -1292,9 +1388,18 @@ def main():
1292
  out_size = n_blocks * 18
1293
  print(f" Q4_0: {ti['name']} (dims[0]={dim0})")
1294
  elif quant_plan[i] is True and q2k_row_compatible(ti['n_dims'], ti['dims']):
1295
- out_type = lowbit_type
 
 
 
 
 
 
 
 
 
1296
  n_blocks = ti['n_elements'] // QK_K
1297
- out_size = n_blocks * lowbit_bytes
1298
  else:
1299
  out_type = ti['type']
1300
  out_size = ti['data_size']
@@ -1482,15 +1587,17 @@ def main():
1482
  total_quant_bytes += nbytes
1483
 
1484
  elif plan:
 
 
1485
  nbytes, rmse, sigma = _stream_quantize(
1486
- fin, fout, ti, abs_offset, lowbit_kind, imatrix_data, use_hpc)
1487
  if rmse is not None:
1488
  q2k_rmse_sum += rmse
1489
  q2k_tensor_count += 1
1490
- print(f"\n [{lowbit_name}] {ti['name'][:50]} RMSE={rmse:.6e}"
1491
  f" σ={sigma:.4f} rel={rmse / max(sigma, 1e-30):.4f}")
1492
  else:
1493
- print(f"\n [{lowbit_name}] {ti['name'][:55]} RMSE=n/a")
1494
  quant_count += 1
1495
  total_quant_bytes += nbytes
1496
 
@@ -1516,16 +1623,18 @@ def main():
1516
  print(" ╔════════════════════════════════════════════════════════════════╗")
1517
  print(" ║ RE-QUANTIZATION SUMMARY ║")
1518
  print(" ╠════════════════════════════════════════════════════════════════╣")
1519
- print(f" ║ Tensors quantized ({lowbit_name:6s}): {quant_count:<31d} ║")
 
 
1520
  print(f" ║ Tensors kept as-is: {total_keep:<33d} ║")
1521
- print(f" ║ {lowbit_name:6s} data: {total_quant_bytes:>12,} bytes ({total_quant_bytes/1024**2:>7.1f} MB) ║")
1522
  print(f" ║ Kept data: {total_keep_bytes:>12,} bytes ({total_keep_bytes/1024**2:>7.1f} MB) ║")
1523
  print(f" ║ Original size: {file_size:>12,} bytes ({file_size/1024**3:>7.2f} GB) ║")
1524
  print(f" ║ Output size: {final_size:>12,} bytes ({final_size/1024**3:>7.2f} GB) ║")
1525
  print(f" ║ Compression: {compression:>42.1f}x ║")
1526
  if q2k_tensor_count > 0:
1527
  mean_rmse = q2k_rmse_sum / q2k_tensor_count
1528
- print(f" ║ Mean {lowbit_name:6s} RMSE: {mean_rmse:>12.6e} ║")
1529
  print(f" ║ Total time: {elapsed:>39.1f} sec ║")
1530
  print(" ╚════════════════════════════════════════════════════════════════╝")
1531
  print()
 
118
  lib.hexstate_set_sse_budget.restype = None
119
  lib.hexstate_set_sse_budget.argtypes = [ctypes.c_float]
120
 
121
+ # IQ2_XS / IQ2_S (E8 codebook) quantizers + dequant
122
+ for suffix in ('iq2_xs', 'iq2_s'):
123
+ qfn = f'hexstate_quantize_tensor_{suffix}_hpc'
124
+ if not hasattr(lib, qfn):
125
+ continue
126
+ getattr(lib, qfn).restype = None
127
+ getattr(lib, qfn).argtypes = [
128
  ctypes.POINTER(ctypes.c_float), # weights
129
  ctypes.c_int64, # n_elements
130
  ctypes.c_void_p, # output
 
133
  ctypes.c_int, # verbose
134
  ctypes.c_int64, # row_width
135
  ]
136
+ dfn = getattr(lib, f'hexstate_dequant_{suffix}')
137
+ dfn.restype = None
138
+ dfn.argtypes = [ctypes.c_void_p, ctypes.c_int64, ctypes.POINTER(ctypes.c_float)]
139
 
140
  lib.hexstate_init()
141
  dc_l = os.environ.get('HEX_DC_LAMBDA')
 
151
  # IQ2_XS has no per-sub-block offset, so cancelling DC means swapping
152
  # codewords — it needs ~2e-2 (≈ +0.2% RMSE) to be effective.
153
  budget = os.environ.get('HEX_SSE_BUDGET')
154
+ if budget is None and LOWBIT_FORMAT in ('iq2xs', 'iq2s', 'auto'):
155
  budget = '0.02'
156
  if budget is not None and hasattr(lib, 'hexstate_set_sse_budget'):
157
  lib.hexstate_set_sse_budget(ctypes.c_float(float(budget)))
 
352
  GGML_TYPE_Q8_0 = 8
353
  GGML_TYPE_Q2_K = 10
354
  GGML_TYPE_IQ2_XS = 17
355
+ GGML_TYPE_IQ2_S = 22
356
  GGML_TYPE_BF16 = 30
357
 
358
  IQ2_XS_BLOCK_BYTES = 74 # d(fp16) + 32×u16 codes + 8 scale bytes
359
+ IQ2_S_BLOCK_BYTES = 82 # d(fp16) + 32 idx + 32 sign bytes + 8 qh + 8 scales
360
+
361
+ # Low-bit target for the "Q2_K plan" tensors:
362
+ # 'q2k' (default) Q2_K, 2.625 bpw
363
+ # 'iq2xs' (--iq2xs) IQ2_XS E8 codebook, 2.3125 bpw
364
+ # 'iq2s' (--iq2s) IQ2_S E8 codebook, 2.5625 bpw
365
+ # 'auto' (--iq2auto) per tensor: IQ2_S where its imatrix-weighted RMSE on a
366
+ # row sample beats Q2_K (Gaussian-ish tensors), Q2_K where the
367
+ # weights are outlier-heavy and Q2_K's min/scale offset wins.
368
  LOWBIT_FORMAT = 'q2k'
369
 
370
+ # kind -> (ggml type, block bytes, display name, LLAMA_FTYPE_MOSTLY_*)
371
+ LOWBIT_KINDS = {
372
+ 'q2k': (GGML_TYPE_Q2_K, 84, 'Q2_K', 10),
373
+ 'iq2xs': (GGML_TYPE_IQ2_XS, IQ2_XS_BLOCK_BYTES, 'IQ2_XS', 20),
374
+ 'iq2s': (GGML_TYPE_IQ2_S, IQ2_S_BLOCK_BYTES, 'IQ2_S', 28),
375
+ }
376
+
377
  TYPE_NAME = {
378
  0: "F32", 1: "F16", 2: "Q4_0", 3: "Q4_1", 6: "Q5_0", 7: "Q5_1",
379
  8: "Q8_0", 9: "Q8_1", 10: "Q2_K", 11: "Q3_K", 12: "Q4_K",
380
+ 13: "Q5_K", 14: "Q6_K", 15: "Q8_K", 17: "IQ2_XS", 22: "IQ2_S", 30: "BF16",
381
  }
382
 
383
  # Block sizes and byte sizes for each type
384
  TYPE_BLOCK_SIZE = {
385
  0: 1, 1: 1, 2: 32, 3: 32, 6: 32, 7: 32,
386
  8: 32, 9: 32, 10: 256, 11: 256, 12: 256,
387
+ 13: 256, 14: 256, 15: 256, 17: 256, 22: 256, 30: 1,
388
  }
389
  TYPE_BLOCK_BYTES = {
390
  0: 4, 1: 2, 2: 18, 3: 20, 6: 20, 7: 22,
391
  8: 34, 9: 36, 10: 84, 11: 110, 12: 144,
392
+ 13: 176, 14: 210, 15: 292, 17: IQ2_XS_BLOCK_BYTES, 22: IQ2_S_BLOCK_BYTES, 30: 2,
393
  }
394
 
395
 
 
852
  return written
853
 
854
 
855
+ def quantize_tensor_iq2_hpc(f32_data, importance=None, row_width=0, kind='iq2xs'):
856
+ """IQ2_XS (74 B) or IQ2_S (82 B) per 256 weights via the HPC C library.
857
  Returns (bytes, n_blocks, dequant_f32)."""
858
+ suffix = {'iq2xs': 'iq2_xs', 'iq2s': 'iq2_s'}[kind]
859
+ block_bytes = LOWBIT_KINDS[kind][1]
860
  lib = _load_hexstate_lib()
861
+ if lib is None or not hasattr(lib, f'hexstate_quantize_tensor_{suffix}_hpc'):
862
+ raise RuntimeError(f'libhexstate_q2k.so lacks {LOWBIT_KINDS[kind][2]} support — rebuild')
863
  f32 = np.ascontiguousarray(f32_data, dtype=np.float32).reshape(-1)
864
  n = int(f32.size)
865
  if n % QK_K != 0:
866
+ raise ValueError(f'{LOWBIT_KINDS[kind][2]} needs a multiple of {QK_K} elements, got {n}')
867
  n_blocks = n // QK_K
868
+ out = np.zeros(n_blocks * block_bytes, dtype=np.uint8)
869
  err = ctypes.c_float(0.0)
870
  imat_ptr = None
871
  if importance is not None:
872
  imat_c = np.ascontiguousarray(importance, dtype=np.float32).reshape(-1)
873
  if imat_c.size == n:
874
  imat_ptr = imat_c.ctypes.data_as(ctypes.POINTER(ctypes.c_float))
875
+ getattr(lib, f'hexstate_quantize_tensor_{suffix}_hpc')(
876
  f32.ctypes.data_as(ctypes.POINTER(ctypes.c_float)),
877
  ctypes.c_int64(n),
878
  out.ctypes.data_as(ctypes.c_void_p),
 
882
  ctypes.c_int64(int(row_width)),
883
  )
884
  deq = np.zeros(n, dtype=np.float32)
885
+ getattr(lib, f'hexstate_dequant_{suffix}')(
886
  out.ctypes.data_as(ctypes.c_void_p), ctypes.c_int64(n_blocks),
887
  deq.ctypes.data_as(ctypes.POINTER(ctypes.c_float)))
888
  return out.tobytes(), n_blocks, deq
889
 
890
 
891
+ def quantize_tensor_iq2xs_hpc(f32_data, importance=None, row_width=0):
892
+ return quantize_tensor_iq2_hpc(f32_data, importance, row_width, 'iq2xs')
893
+
894
+
895
+ def _lowbit_dequant(kind, qbytes, n_blocks, n):
896
+ """Dequantize a low-bit payload of any supported kind to f32[n]."""
897
+ if kind == 'q2k':
898
+ return dequant_q2k_fast(qbytes, n_blocks)[:n]
899
+ lib = _load_hexstate_lib()
900
+ suffix = {'iq2xs': 'iq2_xs', 'iq2s': 'iq2_s'}[kind]
901
+ buf = np.frombuffer(qbytes, dtype=np.uint8)
902
+ deq = np.zeros(n_blocks * QK_K, dtype=np.float32)
903
+ getattr(lib, f'hexstate_dequant_{suffix}')(
904
+ buf.ctypes.data_as(ctypes.c_void_p), ctypes.c_int64(n_blocks),
905
+ deq.ctypes.data_as(ctypes.POINTER(ctypes.c_float)))
906
+ return deq[:n]
907
+
908
+
909
+ def auto_select_lowbit(fin, ti, abs_offset, imatrix_data, n_sample_rows=32,
910
+ margin=None):
911
+ """Pick 'iq2s' or 'q2k' for one tensor from a row sample.
912
+
913
+ Encodes evenly spaced rows both ways and compares imatrix-weighted RMSE
914
+ (the quantity that tracked PPL in the splice A/Bs). IQ2_S is 2.4%
915
+ smaller, so it wins ties; HEX_AUTO_MARGIN (default 0) lets it win while
916
+ up to that relative fraction *worse* if you want to bias toward size.
917
+ Returns (kind, wrmse_q2k, wrmse_iq2s).
918
+ """
919
+ if margin is None:
920
+ margin = float(os.environ.get('HEX_AUTO_MARGIN', '0'))
921
+ d0 = int(ti['dims'][0])
922
+ n_rows = int(ti['n_elements']) // d0
923
+ k = min(n_sample_rows, n_rows)
924
+ rows = np.unique(np.linspace(0, n_rows - 1, k).astype(np.int64))
925
+ parts, imps = [], []
926
+ for r in rows:
927
+ parts.append(_load_rows_f32(fin, abs_offset, ti['type'], d0, int(r), int(r) + 1))
928
+ imps.append(_imat_chunk(ti, imatrix_data, int(r), int(r) + 1, d0))
929
+ X = np.concatenate(parts).astype(np.float32)
930
+ W = np.concatenate(imps).astype(np.float32) if imps[0] is not None else np.ones_like(X)
931
+ if W.size != X.size:
932
+ W = np.ones_like(X)
933
+ n = X.size
934
+ qb, nb = quantize_tensor_q2k_hpc(X, opt_mode=2, importance=W if imps[0] is not None else None,
935
+ row_width=d0)
936
+ e_q = _lowbit_dequant('q2k', qb, nb, n) - X
937
+ sb, nb2, deq_s = quantize_tensor_iq2_hpc(X, importance=W if imps[0] is not None else None,
938
+ row_width=d0, kind='iq2s')
939
+ e_s = deq_s[:n] - X
940
+ den = float(np.sum(W * X * X)) or 1.0
941
+ wr_q = float(np.sqrt(np.sum(W * e_q * e_q) / den))
942
+ wr_s = float(np.sqrt(np.sum(W * e_s * e_s) / den))
943
+ kind = 'iq2s' if wr_s <= wr_q * (1.0 + margin) else 'q2k'
944
+ return kind, wr_q, wr_s
945
+
946
+
947
  def _stream_quantize(fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc):
948
  """Quantize one tensor in row chunks. kind: 'q2k' | 'iq2xs' | 'q4' | 'q8'.
949
  Returns (n_out_bytes, rmse_or_None, sigma_or_None).
 
958
 
959
  d0 = int(ti['dims'][0])
960
  n_rows = int(ti['n_elements']) // d0
961
+ align = QK_K if kind in ('q2k', 'iq2xs', 'iq2s') else 32
962
  if d0 % align != 0:
963
  raise ValueError(f'{ti["name"]} dim0={d0} not aligned to {align}')
964
 
 
976
  total_ss += float(np.vdot(f32, f32))
977
  total_n += n_valid
978
 
979
+ if kind in ('iq2xs', 'iq2s'):
980
+ qbytes, n_blocks, deq = quantize_tensor_iq2_hpc(
981
+ f32, importance=imp, row_width=d0, kind=kind)
982
  fout.write(qbytes)
983
  written += len(qbytes)
984
  diff = f32.reshape(-1)[:n_valid] - deq[:n_valid]
 
1149
  def main():
1150
  if len(sys.argv) < 3:
1151
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
1152
+ " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
1153
+ " [--iq2xs | --iq2s | --iq2auto]")
1154
  print(" --iq2xs low-bit tensors → IQ2_XS (E8 codebook, 2.3125 bpw) instead of Q2_K")
1155
+ print(" --iq2s low-bit tensors → IQ2_S (E8 codebook, 2.5625 bpw) instead of Q2_K")
1156
+ print(" --iq2auto per tensor: IQ2_S or Q2_K, whichever has lower imatrix-weighted")
1157
+ print(" RMSE on a row sample (IQ2_S wins Gaussian-ish tensors, Q2_K wins")
1158
+ print(" outlier-heavy ones). HEX_AUTO_MARGIN biases toward IQ2_S.")
1159
  print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
1160
  print(" HEX_DC_LAMBDA DC residual weight (default 1)")
1161
  print(" HEX_VW_LAMBDA vesica weight (default 1)")
1162
  print(" HEX_DC_DECAY rolling residual carry 0..1 (default 0.85)")
1163
+ print(" HEX_SSE_BUDGET relative SSE the shaper may spend (Q2_K 5e-4, IQ2 2e-2)")
1164
  sys.exit(1)
1165
 
1166
  global LOWBIT_FORMAT
 
1172
  keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
1173
  if '--iq2xs' in sys.argv:
1174
  LOWBIT_FORMAT = 'iq2xs'
1175
+ elif '--iq2s' in sys.argv:
1176
+ LOWBIT_FORMAT = 'iq2s'
1177
+ elif '--iq2auto' in sys.argv:
1178
+ LOWBIT_FORMAT = 'auto'
1179
+ lowbit_kind = LOWBIT_FORMAT # 'q2k' | 'iq2xs' | 'iq2s' | 'auto'
1180
+ lowbit_auto = lowbit_kind == 'auto'
1181
+ # For 'auto' the per-tensor kind is decided at plan time (lowbit_choice);
1182
+ # file-level labels use IQ2_S since that is the intended majority type.
1183
+ _label_kind = 'iq2s' if lowbit_auto else lowbit_kind
1184
+ lowbit_name = 'IQ2_S/Q2_K' if lowbit_auto else LOWBIT_KINDS[_label_kind][2]
1185
+ lowbit_file_type = LOWBIT_KINDS[_label_kind][3] # LLAMA_FTYPE_MOSTLY_*
1186
 
1187
  # Check for imatrix
1188
  imatrix_data = None
 
1198
 
1199
  # Check for HPC C library
1200
  use_hpc = _load_hexstate_lib() is not None
1201
+ if lowbit_kind != 'q2k':
1202
  lib = _load_hexstate_lib()
1203
+ need = 'hexstate_quantize_tensor_iq2_xs_hpc' if lowbit_kind == 'iq2xs' \
1204
+ else 'hexstate_quantize_tensor_iq2_s_hpc'
1205
+ if lib is None or not hasattr(lib, need):
1206
+ print(f" ERROR: --{'iq2auto' if lowbit_auto else lowbit_kind} needs libhexstate_q2k.so "
1207
+ "with IQ2 support (make -f makefile.quantize.c)")
1208
  sys.exit(1)
1209
 
1210
  print()
1211
  print(" ╔════════════════════════════════════════════════════════════════╗")
1212
  print(" ║ HExState GGUF Re-Quantizer ║")
1213
+ print(f" ║ GGUF → {lowbit_name:10s} GGUF with metadata passthrough ║")
1214
  if lowbit_kind == 'iq2xs':
1215
  print(" ║ Low-bit: IQ2_XS E8 codebook · fold/DC shaping · 2.3125 bpw ║")
1216
+ elif lowbit_kind == 'iq2s':
1217
+ print(" ║ Low-bit: IQ2_S E8 codebook · fold/DC shaping · 2.5625 bpw ║")
1218
+ elif lowbit_auto:
1219
+ print(" ║ Low-bit: AUTO IQ2_S vs Q2_K per tensor by weighted RMSE ║")
1220
  if q2all:
1221
  print(" ║ Mode: --q2all ALL eligible tensors → Q2_K (test mode) ║")
1222
  if use_hpc and imatrix_data:
 
1360
  # ── Compute output tensor sizes and offsets ──
1361
  out_tensor_infos = []
1362
  out_data_offset = 0
1363
+ lowbit_choice = {} # tensor index -> 'q2k' | 'iq2xs' | 'iq2s'
1364
+ auto_counts = {}
1365
+ if lowbit_auto:
1366
+ print(" Auto-selecting IQ2_S vs Q2_K per tensor (32-row samples)...")
1367
 
1368
  for i, ti in enumerate(tensor_infos):
1369
  if quant_plan[i]:
 
1388
  out_size = n_blocks * 18
1389
  print(f" Q4_0: {ti['name']} (dims[0]={dim0})")
1390
  elif quant_plan[i] is True and q2k_row_compatible(ti['n_dims'], ti['dims']):
1391
+ if lowbit_auto:
1392
+ kind, wr_q, wr_s = auto_select_lowbit(
1393
+ fin, ti, data_section_start + ti['offset'], imatrix_data)
1394
+ auto_counts[kind] = auto_counts.get(kind, 0) + 1
1395
+ print(f" [AUTO→{LOWBIT_KINDS[kind][2]:6s}] {ti['name'][:48]:48s} "
1396
+ f"wRMSE Q2_K={wr_q:.4f} IQ2_S={wr_s:.4f}")
1397
+ else:
1398
+ kind = lowbit_kind
1399
+ lowbit_choice[i] = kind
1400
+ out_type = LOWBIT_KINDS[kind][0]
1401
  n_blocks = ti['n_elements'] // QK_K
1402
+ out_size = n_blocks * LOWBIT_KINDS[kind][1]
1403
  else:
1404
  out_type = ti['type']
1405
  out_size = ti['data_size']
 
1587
  total_quant_bytes += nbytes
1588
 
1589
  elif plan:
1590
+ kind = lowbit_choice.get(i, 'q2k')
1591
+ kname = LOWBIT_KINDS[kind][2]
1592
  nbytes, rmse, sigma = _stream_quantize(
1593
+ fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc)
1594
  if rmse is not None:
1595
  q2k_rmse_sum += rmse
1596
  q2k_tensor_count += 1
1597
+ print(f"\n [{kname}] {ti['name'][:50]} RMSE={rmse:.6e}"
1598
  f" σ={sigma:.4f} rel={rmse / max(sigma, 1e-30):.4f}")
1599
  else:
1600
+ print(f"\n [{kname}] {ti['name'][:55]} RMSE=n/a")
1601
  quant_count += 1
1602
  total_quant_bytes += nbytes
1603
 
 
1623
  print(" ╔════════════════════════════════════════════════════════════════╗")
1624
  print(" ║ RE-QUANTIZATION SUMMARY ║")
1625
  print(" ╠════════════════════════════════════════════════════════════════╣")
1626
+ print(f" ║ Tensors quantized ({lowbit_name[:10]:10s}): {quant_count:<27d} ║")
1627
+ if lowbit_auto:
1628
+ print(f" ║ Auto split: IQ2_S {auto_counts.get('iq2s', 0):<5d} Q2_K {auto_counts.get('q2k', 0):<27d} ║")
1629
  print(f" ║ Tensors kept as-is: {total_keep:<33d} ║")
1630
+ print(f" ║ {lowbit_name[:10]:10s} data: {total_quant_bytes:>12,} bytes ({total_quant_bytes/1024**2:>7.1f} MB) ║")
1631
  print(f" ║ Kept data: {total_keep_bytes:>12,} bytes ({total_keep_bytes/1024**2:>7.1f} MB) ║")
1632
  print(f" ║ Original size: {file_size:>12,} bytes ({file_size/1024**3:>7.2f} GB) ║")
1633
  print(f" ║ Output size: {final_size:>12,} bytes ({final_size/1024**3:>7.2f} GB) ║")
1634
  print(f" ║ Compression: {compression:>42.1f}x ║")
1635
  if q2k_tensor_count > 0:
1636
  mean_rmse = q2k_rmse_sum / q2k_tensor_count
1637
+ print(f" ║ Mean {lowbit_name[:10]:10s} RMSE: {mean_rmse:>12.6e} ║")
1638
  print(f" ║ Total time: {elapsed:>39.1f} sec ║")
1639
  print(" ╚════════════════════════════════════════════════════════════════╝")
1640
  print()
iq2xs_grid.h ADDED
@@ -0,0 +1,410 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ /* iq2xs_grid.h — IQ2_XS codebook, copied verbatim from ggml-common.h (MIT).
2
+ * 512 codewords x 8 bytes; byte j = magnitude {0x08,0x19,0x2b} of weight j.
3
+ * ksigns_iq2xs[i] = i | (parity(i) << 7): 7 stored sign bits + parity bit. */
4
+ #ifndef IQ2XS_GRID_H
5
+ #define IQ2XS_GRID_H
6
+ #include <stdint.h>
7
+
8
+ static const uint8_t ksigns_iq2xs[128] = {
9
+ 0, 129, 130, 3, 132, 5, 6, 135, 136, 9, 10, 139, 12, 141, 142, 15,
10
+ 144, 17, 18, 147, 20, 149, 150, 23, 24, 153, 154, 27, 156, 29, 30, 159,
11
+ 160, 33, 34, 163, 36, 165, 166, 39, 40, 169, 170, 43, 172, 45, 46, 175,
12
+ 48, 177, 178, 51, 180, 53, 54, 183, 184, 57, 58, 187, 60, 189, 190, 63,
13
+ 192, 65, 66, 195, 68, 197, 198, 71, 72, 201, 202, 75, 204, 77, 78, 207,
14
+ 80, 209, 210, 83, 212, 85, 86, 215, 216, 89, 90, 219, 92, 221, 222, 95,
15
+ 96, 225, 226, 99, 228, 101, 102, 231, 232, 105, 106, 235, 108, 237, 238, 111,
16
+ 240, 113, 114, 243, 116, 245, 246, 119, 120, 249, 250, 123, 252, 125, 126, 255,
17
+ };
18
+
19
+ static const uint64_t iq2xs_grid[512] = {
20
+ 0x0808080808080808, 0x080808080808082b, 0x0808080808081919, 0x0808080808082b08,
21
+ 0x0808080808082b2b, 0x0808080808190819, 0x0808080808191908, 0x080808080819192b,
22
+ 0x0808080808192b19, 0x08080808082b0808, 0x08080808082b082b, 0x08080808082b1919,
23
+ 0x08080808082b2b08, 0x0808080819080819, 0x0808080819081908, 0x080808081908192b,
24
+ 0x0808080819082b19, 0x0808080819190808, 0x080808081919082b, 0x0808080819191919,
25
+ 0x0808080819192b08, 0x08080808192b0819, 0x08080808192b1908, 0x080808082b080808,
26
+ 0x080808082b08082b, 0x080808082b081919, 0x080808082b082b08, 0x080808082b190819,
27
+ 0x080808082b191908, 0x080808082b192b19, 0x080808082b2b0808, 0x0808081908080819,
28
+ 0x0808081908081908, 0x080808190808192b, 0x0808081908082b19, 0x0808081908190808,
29
+ 0x080808190819082b, 0x0808081908191919, 0x0808081908192b08, 0x0808081908192b2b,
30
+ 0x08080819082b0819, 0x08080819082b1908, 0x0808081919080808, 0x080808191908082b,
31
+ 0x0808081919081919, 0x0808081919082b08, 0x0808081919190819, 0x0808081919191908,
32
+ 0x08080819192b0808, 0x08080819192b2b08, 0x080808192b080819, 0x080808192b081908,
33
+ 0x080808192b190808, 0x0808082b08080808, 0x0808082b0808082b, 0x0808082b08081919,
34
+ 0x0808082b08082b08, 0x0808082b08190819, 0x0808082b08191908, 0x0808082b082b0808,
35
+ 0x0808082b19080819, 0x0808082b19081908, 0x0808082b19190808, 0x0808082b19191919,
36
+ 0x0808082b2b080808, 0x0808082b2b082b2b, 0x0808190808080819, 0x0808190808081908,
37
+ 0x080819080808192b, 0x0808190808082b19, 0x0808190808190808, 0x080819080819082b,
38
+ 0x0808190808191919, 0x0808190808192b08, 0x08081908082b0819, 0x08081908082b1908,
39
+ 0x0808190819080808, 0x080819081908082b, 0x0808190819081919, 0x0808190819082b08,
40
+ 0x0808190819190819, 0x0808190819191908, 0x080819081919192b, 0x08081908192b0808,
41
+ 0x080819082b080819, 0x080819082b081908, 0x080819082b190808, 0x0808191908080808,
42
+ 0x080819190808082b, 0x0808191908081919, 0x0808191908082b08, 0x0808191908190819,
43
+ 0x0808191908191908, 0x08081919082b0808, 0x0808191919080819, 0x0808191919081908,
44
+ 0x0808191919190808, 0x08081919192b0819, 0x080819192b080808, 0x0808192b08080819,
45
+ 0x0808192b08081908, 0x0808192b08190808, 0x0808192b082b192b, 0x0808192b19080808,
46
+ 0x0808192b1908082b, 0x0808192b2b081908, 0x08082b0808080808, 0x08082b080808082b,
47
+ 0x08082b0808081919, 0x08082b0808082b08, 0x08082b0808082b2b, 0x08082b0808190819,
48
+ 0x08082b0808191908, 0x08082b08082b0808, 0x08082b08082b1919, 0x08082b0819080819,
49
+ 0x08082b0819081908, 0x08082b0819190808, 0x08082b0819192b08, 0x08082b082b080808,
50
+ 0x08082b082b2b0808, 0x08082b082b2b2b2b, 0x08082b1908080819, 0x08082b1908081908,
51
+ 0x08082b1908190808, 0x08082b1919080808, 0x08082b192b080819, 0x08082b192b082b19,
52
+ 0x08082b2b08080808, 0x08082b2b082b0808, 0x08082b2b082b2b08, 0x08082b2b2b19192b,
53
+ 0x08082b2b2b2b0808, 0x0819080808080819, 0x0819080808081908, 0x081908080808192b,
54
+ 0x0819080808082b19, 0x0819080808190808, 0x081908080819082b, 0x0819080808191919,
55
+ 0x0819080808192b08, 0x08190808082b0819, 0x08190808082b1908, 0x0819080819080808,
56
+ 0x081908081908082b, 0x0819080819081919, 0x0819080819082b08, 0x0819080819190819,
57
+ 0x0819080819191908, 0x08190808192b0808, 0x08190808192b2b2b, 0x081908082b080819,
58
+ 0x081908082b081908, 0x081908082b190808, 0x0819081908080808, 0x081908190808082b,
59
+ 0x0819081908081919, 0x0819081908082b08, 0x0819081908190819, 0x0819081908191908,
60
+ 0x08190819082b0808, 0x0819081919080819, 0x0819081919081908, 0x0819081919190808,
61
+ 0x081908192b080808, 0x081908192b191908, 0x081908192b19192b, 0x0819082b08080819,
62
+ 0x0819082b08081908, 0x0819082b0808192b, 0x0819082b08190808, 0x0819082b19080808,
63
+ 0x0819082b192b0808, 0x0819190808080808, 0x081919080808082b, 0x0819190808081919,
64
+ 0x0819190808082b08, 0x0819190808190819, 0x0819190808191908, 0x08191908082b0808,
65
+ 0x0819190819080819, 0x0819190819081908, 0x0819190819082b19, 0x0819190819190808,
66
+ 0x08191908192b1908, 0x081919082b080808, 0x0819191908080819, 0x0819191908081908,
67
+ 0x0819191908190808, 0x0819191919080808, 0x0819192b08080808, 0x0819192b08191908,
68
+ 0x0819192b19082b19, 0x08192b0808080819, 0x08192b0808081908, 0x08192b0808190808,
69
+ 0x08192b080819082b, 0x08192b0819080808, 0x08192b0819191908, 0x08192b082b08192b,
70
+ 0x08192b1908080808, 0x08192b1908081919, 0x08192b19192b192b, 0x08192b2b19190819,
71
+ 0x08192b2b2b2b2b19, 0x082b080808080808, 0x082b08080808082b, 0x082b080808081919,
72
+ 0x082b080808082b08, 0x082b080808082b2b, 0x082b080808190819, 0x082b080808191908,
73
+ 0x082b0808082b0808, 0x082b080819080819, 0x082b080819081908, 0x082b080819190808,
74
+ 0x082b08082b080808, 0x082b08082b2b0808, 0x082b081908080819, 0x082b081908081908,
75
+ 0x082b081908190808, 0x082b081919080808, 0x082b081919082b08, 0x082b0819192b1919,
76
+ 0x082b082b08080808, 0x082b082b082b082b, 0x082b082b2b080808, 0x082b082b2b2b2b08,
77
+ 0x082b190808080819, 0x082b190808081908, 0x082b190808190808, 0x082b1908082b2b19,
78
+ 0x082b190819080808, 0x082b191908080808, 0x082b191919080819, 0x082b19191919082b,
79
+ 0x082b19192b192b19, 0x082b192b08080819, 0x082b192b08192b2b, 0x082b192b2b2b192b,
80
+ 0x082b2b0808080808, 0x082b2b0808082b08, 0x082b2b0808082b2b, 0x082b2b08082b0808,
81
+ 0x082b2b0819191919, 0x082b2b082b082b08, 0x082b2b082b2b082b, 0x082b2b19192b2b08,
82
+ 0x082b2b192b190808, 0x082b2b2b08082b08, 0x082b2b2b082b0808, 0x082b2b2b2b08082b,
83
+ 0x082b2b2b2b082b08, 0x082b2b2b2b082b2b, 0x1908080808080819, 0x1908080808081908,
84
+ 0x190808080808192b, 0x1908080808082b19, 0x1908080808190808, 0x190808080819082b,
85
+ 0x1908080808191919, 0x1908080808192b08, 0x19080808082b0819, 0x19080808082b1908,
86
+ 0x1908080819080808, 0x190808081908082b, 0x1908080819081919, 0x1908080819082b08,
87
+ 0x1908080819082b2b, 0x1908080819190819, 0x1908080819191908, 0x19080808192b0808,
88
+ 0x19080808192b1919, 0x190808082b080819, 0x190808082b081908, 0x190808082b190808,
89
+ 0x1908081908080808, 0x190808190808082b, 0x1908081908081919, 0x1908081908082b08,
90
+ 0x1908081908190819, 0x1908081908191908, 0x19080819082b0808, 0x1908081919080819,
91
+ 0x1908081919081908, 0x1908081919190808, 0x190808192b080808, 0x190808192b081919,
92
+ 0x190808192b2b082b, 0x1908082b08080819, 0x1908082b08081908, 0x1908082b08190808,
93
+ 0x1908082b0819082b, 0x1908082b082b2b19, 0x1908082b19080808, 0x1908190808080808,
94
+ 0x190819080808082b, 0x1908190808081919, 0x1908190808082b08, 0x1908190808190819,
95
+ 0x1908190808191908, 0x1908190808192b19, 0x19081908082b0808, 0x1908190819080819,
96
+ 0x1908190819081908, 0x1908190819190808, 0x190819082b080808, 0x190819082b191908,
97
+ 0x1908191908080819, 0x1908191908081908, 0x1908191908190808, 0x19081919082b1908,
98
+ 0x1908191919080808, 0x190819192b192b2b, 0x1908192b08080808, 0x1908192b08082b2b,
99
+ 0x1908192b19081908, 0x1908192b19190808, 0x19082b0808080819, 0x19082b0808081908,
100
+ 0x19082b0808190808, 0x19082b0819080808, 0x19082b0819081919, 0x19082b0819191908,
101
+ 0x19082b08192b082b, 0x19082b1908080808, 0x19082b1908190819, 0x19082b1919081908,
102
+ 0x19082b1919190808, 0x19082b19192b2b19, 0x19082b2b08081908, 0x1919080808080808,
103
+ 0x191908080808082b, 0x1919080808081919, 0x1919080808082b08, 0x1919080808190819,
104
+ 0x1919080808191908, 0x19190808082b0808, 0x19190808082b2b08, 0x1919080819080819,
105
+ 0x1919080819081908, 0x1919080819190808, 0x191908082b080808, 0x1919081908080819,
106
+ 0x1919081908081908, 0x1919081908190808, 0x1919081908191919, 0x1919081919080808,
107
+ 0x191908191908082b, 0x1919082b08080808, 0x1919082b19081908, 0x1919082b2b2b2b2b,
108
+ 0x1919190808080819, 0x1919190808081908, 0x1919190808190808, 0x19191908082b0819,
109
+ 0x1919190819080808, 0x19191908192b0808, 0x191919082b080819, 0x191919082b2b0819,
110
+ 0x1919191908080808, 0x1919191908082b08, 0x191919192b080808, 0x191919192b082b08,
111
+ 0x1919192b082b0819, 0x1919192b192b2b08, 0x1919192b2b2b0819, 0x19192b0808080808,
112
+ 0x19192b0808191908, 0x19192b0819080819, 0x19192b0819190808, 0x19192b082b192b19,
113
+ 0x19192b1908192b2b, 0x19192b1919080808, 0x19192b191908082b, 0x19192b2b2b081919,
114
+ 0x192b080808080819, 0x192b080808081908, 0x192b080808190808, 0x192b080819080808,
115
+ 0x192b080819191908, 0x192b0808192b082b, 0x192b08082b08192b, 0x192b08082b2b2b19,
116
+ 0x192b081908080808, 0x192b082b082b1908, 0x192b082b19082b2b, 0x192b082b2b19082b,
117
+ 0x192b190808080808, 0x192b19080819192b, 0x192b191908190808, 0x192b191919080808,
118
+ 0x192b191919081919, 0x192b19192b2b1908, 0x192b2b0808080819, 0x192b2b08192b2b2b,
119
+ 0x192b2b19082b1919, 0x192b2b2b0808192b, 0x192b2b2b19191908, 0x192b2b2b192b082b,
120
+ 0x2b08080808080808, 0x2b0808080808082b, 0x2b08080808081919, 0x2b08080808082b08,
121
+ 0x2b08080808190819, 0x2b08080808191908, 0x2b080808082b0808, 0x2b080808082b2b2b,
122
+ 0x2b08080819080819, 0x2b08080819081908, 0x2b08080819190808, 0x2b0808082b080808,
123
+ 0x2b0808082b08082b, 0x2b0808082b2b2b08, 0x2b0808082b2b2b2b, 0x2b08081908080819,
124
+ 0x2b08081908081908, 0x2b0808190808192b, 0x2b08081908190808, 0x2b08081919080808,
125
+ 0x2b08081919190819, 0x2b08081919192b19, 0x2b08082b08080808, 0x2b08082b082b0808,
126
+ 0x2b08082b2b080808, 0x2b08082b2b08082b, 0x2b08082b2b2b0808, 0x2b08082b2b2b2b08,
127
+ 0x2b08190808080819, 0x2b08190808081908, 0x2b08190808190808, 0x2b0819080819082b,
128
+ 0x2b08190808191919, 0x2b08190819080808, 0x2b081908192b0808, 0x2b0819082b082b19,
129
+ 0x2b08191908080808, 0x2b08191919081908, 0x2b0819192b2b1919, 0x2b08192b08192b08,
130
+ 0x2b08192b192b2b2b, 0x2b082b0808080808, 0x2b082b0808082b08, 0x2b082b08082b1919,
131
+ 0x2b082b0819192b2b, 0x2b082b082b080808, 0x2b082b082b08082b, 0x2b082b082b2b2b08,
132
+ 0x2b082b190808192b, 0x2b082b2b082b082b, 0x2b082b2b2b080808, 0x2b082b2b2b082b08,
133
+ 0x2b082b2b2b19192b, 0x2b082b2b2b2b2b08, 0x2b19080808080819, 0x2b19080808081908,
134
+ 0x2b19080808190808, 0x2b19080819080808, 0x2b1908081919192b, 0x2b1908082b081908,
135
+ 0x2b19081908080808, 0x2b190819082b082b, 0x2b190819192b1908, 0x2b19082b1919192b,
136
+ 0x2b19082b2b082b19, 0x2b19190808080808, 0x2b19190808081919, 0x2b19190819081908,
137
+ 0x2b19190819190808, 0x2b19190819192b08, 0x2b191919082b2b19, 0x2b1919192b190808,
138
+ 0x2b1919192b19082b, 0x2b19192b19080819, 0x2b192b0819190819, 0x2b192b082b2b192b,
139
+ 0x2b192b1919082b19, 0x2b192b2b08191919, 0x2b192b2b192b0808, 0x2b2b080808080808,
140
+ 0x2b2b08080808082b, 0x2b2b080808082b08, 0x2b2b080808082b2b, 0x2b2b0808082b0808,
141
+ 0x2b2b0808082b2b2b, 0x2b2b08082b2b0808, 0x2b2b081919190819, 0x2b2b081919192b19,
142
+ 0x2b2b08192b2b192b, 0x2b2b082b08080808, 0x2b2b082b0808082b, 0x2b2b082b08082b08,
143
+ 0x2b2b082b082b2b2b, 0x2b2b082b2b080808, 0x2b2b082b2b2b0808, 0x2b2b190819080808,
144
+ 0x2b2b19082b191919, 0x2b2b192b192b1919, 0x2b2b192b2b192b08, 0x2b2b2b0808082b2b,
145
+ 0x2b2b2b08082b0808, 0x2b2b2b08082b082b, 0x2b2b2b08082b2b08, 0x2b2b2b082b2b0808,
146
+ 0x2b2b2b082b2b2b08, 0x2b2b2b1908081908, 0x2b2b2b192b081908, 0x2b2b2b192b08192b,
147
+ 0x2b2b2b2b082b2b08, 0x2b2b2b2b082b2b2b, 0x2b2b2b2b2b190819, 0x2b2b2b2b2b2b2b2b,
148
+ };
149
+
150
+ /* IQ2_S codebook (1024 codewords, same alphabet, full 8 sign bits). */
151
+ static const uint64_t iq2s_grid[1024] = {
152
+ 0x0808080808080808, 0x080808080808082b, 0x0808080808081919, 0x0808080808082b08,
153
+ 0x0808080808082b2b, 0x0808080808190819, 0x0808080808191908, 0x080808080819192b,
154
+ 0x0808080808192b19, 0x08080808082b0808, 0x08080808082b082b, 0x08080808082b1919,
155
+ 0x08080808082b2b08, 0x0808080819080819, 0x0808080819081908, 0x080808081908192b,
156
+ 0x0808080819082b19, 0x0808080819190808, 0x080808081919082b, 0x0808080819191919,
157
+ 0x0808080819192b08, 0x08080808192b0819, 0x08080808192b1908, 0x08080808192b192b,
158
+ 0x08080808192b2b19, 0x080808082b080808, 0x080808082b08082b, 0x080808082b081919,
159
+ 0x080808082b082b08, 0x080808082b190819, 0x080808082b191908, 0x080808082b2b0808,
160
+ 0x080808082b2b1919, 0x080808082b2b2b2b, 0x0808081908080819, 0x0808081908081908,
161
+ 0x080808190808192b, 0x0808081908082b19, 0x0808081908190808, 0x080808190819082b,
162
+ 0x0808081908191919, 0x0808081908192b08, 0x08080819082b0819, 0x08080819082b1908,
163
+ 0x0808081919080808, 0x080808191908082b, 0x0808081919081919, 0x0808081919082b08,
164
+ 0x0808081919190819, 0x0808081919191908, 0x080808191919192b, 0x0808081919192b19,
165
+ 0x08080819192b0808, 0x08080819192b1919, 0x08080819192b2b08, 0x080808192b080819,
166
+ 0x080808192b081908, 0x080808192b190808, 0x080808192b19082b, 0x080808192b191919,
167
+ 0x080808192b2b0819, 0x080808192b2b1908, 0x0808082b08080808, 0x0808082b0808082b,
168
+ 0x0808082b08081919, 0x0808082b08082b08, 0x0808082b08190819, 0x0808082b08191908,
169
+ 0x0808082b082b0808, 0x0808082b082b2b2b, 0x0808082b19080819, 0x0808082b19081908,
170
+ 0x0808082b1908192b, 0x0808082b19082b19, 0x0808082b19190808, 0x0808082b19191919,
171
+ 0x0808082b2b080808, 0x0808082b2b081919, 0x0808082b2b082b2b, 0x0808082b2b191908,
172
+ 0x0808082b2b2b082b, 0x0808190808080819, 0x0808190808081908, 0x080819080808192b,
173
+ 0x0808190808082b19, 0x0808190808190808, 0x080819080819082b, 0x0808190808191919,
174
+ 0x0808190808192b08, 0x08081908082b0819, 0x08081908082b1908, 0x08081908082b192b,
175
+ 0x08081908082b2b19, 0x0808190819080808, 0x080819081908082b, 0x0808190819081919,
176
+ 0x0808190819082b08, 0x0808190819082b2b, 0x0808190819190819, 0x0808190819191908,
177
+ 0x080819081919192b, 0x0808190819192b19, 0x08081908192b0808, 0x08081908192b082b,
178
+ 0x08081908192b1919, 0x080819082b080819, 0x080819082b081908, 0x080819082b08192b,
179
+ 0x080819082b082b19, 0x080819082b190808, 0x080819082b191919, 0x080819082b192b08,
180
+ 0x080819082b2b0819, 0x080819082b2b1908, 0x0808191908080808, 0x080819190808082b,
181
+ 0x0808191908081919, 0x0808191908082b08, 0x0808191908082b2b, 0x0808191908190819,
182
+ 0x0808191908191908, 0x080819190819192b, 0x0808191908192b19, 0x08081919082b0808,
183
+ 0x08081919082b1919, 0x08081919082b2b08, 0x0808191919080819, 0x0808191919081908,
184
+ 0x080819191908192b, 0x0808191919082b19, 0x0808191919190808, 0x080819191919082b,
185
+ 0x0808191919191919, 0x0808191919192b08, 0x08081919192b0819, 0x08081919192b1908,
186
+ 0x080819192b080808, 0x080819192b08082b, 0x080819192b081919, 0x080819192b082b08,
187
+ 0x080819192b190819, 0x080819192b191908, 0x080819192b2b0808, 0x0808192b08080819,
188
+ 0x0808192b08081908, 0x0808192b0808192b, 0x0808192b08082b19, 0x0808192b08190808,
189
+ 0x0808192b08191919, 0x0808192b19080808, 0x0808192b19081919, 0x0808192b19082b08,
190
+ 0x0808192b19190819, 0x0808192b19191908, 0x0808192b192b0808, 0x0808192b2b080819,
191
+ 0x0808192b2b081908, 0x0808192b2b190808, 0x08082b0808080808, 0x08082b080808082b,
192
+ 0x08082b0808081919, 0x08082b0808082b08, 0x08082b0808190819, 0x08082b0808191908,
193
+ 0x08082b080819192b, 0x08082b0808192b19, 0x08082b08082b0808, 0x08082b08082b1919,
194
+ 0x08082b08082b2b2b, 0x08082b0819080819, 0x08082b0819081908, 0x08082b081908192b,
195
+ 0x08082b0819082b19, 0x08082b0819190808, 0x08082b081919082b, 0x08082b0819191919,
196
+ 0x08082b0819192b08, 0x08082b08192b0819, 0x08082b08192b1908, 0x08082b082b080808,
197
+ 0x08082b082b081919, 0x08082b082b191908, 0x08082b082b2b2b2b, 0x08082b1908080819,
198
+ 0x08082b1908081908, 0x08082b1908190808, 0x08082b190819082b, 0x08082b1908191919,
199
+ 0x08082b1908192b08, 0x08082b19082b0819, 0x08082b1919080808, 0x08082b1919081919,
200
+ 0x08082b1919082b08, 0x08082b1919190819, 0x08082b1919191908, 0x08082b19192b0808,
201
+ 0x08082b192b080819, 0x08082b192b190808, 0x08082b2b08080808, 0x08082b2b08190819,
202
+ 0x08082b2b08191908, 0x08082b2b082b082b, 0x08082b2b082b2b08, 0x08082b2b082b2b2b,
203
+ 0x08082b2b19190808, 0x08082b2b2b192b19, 0x0819080808080819, 0x0819080808081908,
204
+ 0x081908080808192b, 0x0819080808082b19, 0x0819080808190808, 0x081908080819082b,
205
+ 0x0819080808191919, 0x0819080808192b08, 0x08190808082b0819, 0x08190808082b1908,
206
+ 0x08190808082b192b, 0x0819080819080808, 0x081908081908082b, 0x0819080819081919,
207
+ 0x0819080819082b08, 0x0819080819190819, 0x0819080819191908, 0x081908081919192b,
208
+ 0x0819080819192b19, 0x08190808192b0808, 0x08190808192b082b, 0x08190808192b1919,
209
+ 0x08190808192b2b08, 0x081908082b080819, 0x081908082b081908, 0x081908082b08192b,
210
+ 0x081908082b190808, 0x081908082b191919, 0x081908082b192b08, 0x081908082b2b0819,
211
+ 0x081908082b2b1908, 0x0819081908080808, 0x081908190808082b, 0x0819081908081919,
212
+ 0x0819081908082b08, 0x0819081908082b2b, 0x0819081908190819, 0x0819081908191908,
213
+ 0x081908190819192b, 0x0819081908192b19, 0x08190819082b0808, 0x08190819082b082b,
214
+ 0x08190819082b1919, 0x08190819082b2b08, 0x0819081919080819, 0x0819081919081908,
215
+ 0x081908191908192b, 0x0819081919082b19, 0x0819081919190808, 0x081908191919082b,
216
+ 0x0819081919191919, 0x0819081919192b08, 0x08190819192b0819, 0x08190819192b1908,
217
+ 0x081908192b080808, 0x081908192b08082b, 0x081908192b081919, 0x081908192b082b08,
218
+ 0x081908192b190819, 0x081908192b191908, 0x0819082b08080819, 0x0819082b08081908,
219
+ 0x0819082b08082b19, 0x0819082b08190808, 0x0819082b08191919, 0x0819082b082b0819,
220
+ 0x0819082b082b1908, 0x0819082b19080808, 0x0819082b19081919, 0x0819082b19190819,
221
+ 0x0819082b19191908, 0x0819082b2b080819, 0x0819082b2b081908, 0x0819082b2b190808,
222
+ 0x0819190808080808, 0x081919080808082b, 0x0819190808081919, 0x0819190808082b08,
223
+ 0x0819190808190819, 0x0819190808191908, 0x081919080819192b, 0x0819190808192b19,
224
+ 0x08191908082b0808, 0x08191908082b1919, 0x08191908082b2b08, 0x0819190819080819,
225
+ 0x0819190819081908, 0x081919081908192b, 0x0819190819082b19, 0x0819190819190808,
226
+ 0x081919081919082b, 0x0819190819191919, 0x0819190819192b08, 0x08191908192b0819,
227
+ 0x08191908192b1908, 0x081919082b080808, 0x081919082b08082b, 0x081919082b081919,
228
+ 0x081919082b082b08, 0x081919082b190819, 0x081919082b191908, 0x081919082b2b0808,
229
+ 0x0819191908080819, 0x0819191908081908, 0x081919190808192b, 0x0819191908082b19,
230
+ 0x0819191908190808, 0x081919190819082b, 0x0819191908191919, 0x0819191908192b08,
231
+ 0x08191919082b0819, 0x08191919082b1908, 0x0819191919080808, 0x081919191908082b,
232
+ 0x0819191919081919, 0x0819191919082b08, 0x0819191919190819, 0x0819191919191908,
233
+ 0x08191919192b0808, 0x081919192b080819, 0x081919192b081908, 0x081919192b190808,
234
+ 0x0819192b08080808, 0x0819192b08081919, 0x0819192b08082b08, 0x0819192b08190819,
235
+ 0x0819192b08191908, 0x0819192b082b0808, 0x0819192b19080819, 0x0819192b19081908,
236
+ 0x0819192b19190808, 0x0819192b2b080808, 0x0819192b2b2b2b2b, 0x08192b0808080819,
237
+ 0x08192b0808081908, 0x08192b080808192b, 0x08192b0808082b19, 0x08192b0808190808,
238
+ 0x08192b0808191919, 0x08192b0808192b08, 0x08192b08082b0819, 0x08192b0819080808,
239
+ 0x08192b081908082b, 0x08192b0819081919, 0x08192b0819082b08, 0x08192b0819190819,
240
+ 0x08192b0819191908, 0x08192b08192b0808, 0x08192b082b080819, 0x08192b082b081908,
241
+ 0x08192b1908080808, 0x08192b190808082b, 0x08192b1908081919, 0x08192b1908082b08,
242
+ 0x08192b1908190819, 0x08192b1908191908, 0x08192b19082b0808, 0x08192b1919080819,
243
+ 0x08192b1919081908, 0x08192b1919190808, 0x08192b19192b2b19, 0x08192b192b2b082b,
244
+ 0x08192b2b08081908, 0x08192b2b08190808, 0x08192b2b19080808, 0x08192b2b1919192b,
245
+ 0x082b080808080808, 0x082b08080808082b, 0x082b080808081919, 0x082b080808082b08,
246
+ 0x082b080808190819, 0x082b080808191908, 0x082b08080819192b, 0x082b080808192b19,
247
+ 0x082b0808082b0808, 0x082b0808082b1919, 0x082b0808082b2b2b, 0x082b080819080819,
248
+ 0x082b080819081908, 0x082b080819190808, 0x082b08081919082b, 0x082b080819191919,
249
+ 0x082b0808192b1908, 0x082b08082b080808, 0x082b08082b082b2b, 0x082b08082b191908,
250
+ 0x082b08082b2b2b2b, 0x082b081908080819, 0x082b081908081908, 0x082b081908190808,
251
+ 0x082b08190819082b, 0x082b081908191919, 0x082b0819082b0819, 0x082b081919080808,
252
+ 0x082b08191908082b, 0x082b081919081919, 0x082b081919190819, 0x082b081919191908,
253
+ 0x082b0819192b0808, 0x082b08192b080819, 0x082b08192b081908, 0x082b08192b190808,
254
+ 0x082b082b08080808, 0x082b082b08082b2b, 0x082b082b082b082b, 0x082b082b082b2b08,
255
+ 0x082b082b082b2b2b, 0x082b082b19081908, 0x082b082b19190808, 0x082b082b2b082b08,
256
+ 0x082b082b2b082b2b, 0x082b082b2b2b2b08, 0x082b190808080819, 0x082b190808081908,
257
+ 0x082b19080808192b, 0x082b190808082b19, 0x082b190808190808, 0x082b190808191919,
258
+ 0x082b190808192b08, 0x082b1908082b0819, 0x082b1908082b1908, 0x082b190819080808,
259
+ 0x082b19081908082b, 0x082b190819081919, 0x082b190819082b08, 0x082b190819190819,
260
+ 0x082b190819191908, 0x082b1908192b0808, 0x082b19082b080819, 0x082b19082b081908,
261
+ 0x082b19082b190808, 0x082b191908080808, 0x082b191908081919, 0x082b191908082b08,
262
+ 0x082b191908190819, 0x082b191908191908, 0x082b1919082b0808, 0x082b191919080819,
263
+ 0x082b191919081908, 0x082b191919190808, 0x082b1919192b192b, 0x082b19192b080808,
264
+ 0x082b192b08080819, 0x082b192b08081908, 0x082b192b08190808, 0x082b192b19080808,
265
+ 0x082b192b19192b19, 0x082b2b0808080808, 0x082b2b0808081919, 0x082b2b0808190819,
266
+ 0x082b2b0808191908, 0x082b2b0819080819, 0x082b2b0819081908, 0x082b2b0819190808,
267
+ 0x082b2b082b082b2b, 0x082b2b082b2b2b2b, 0x082b2b1908080819, 0x082b2b1908081908,
268
+ 0x082b2b1908190808, 0x082b2b192b191919, 0x082b2b2b08082b2b, 0x082b2b2b082b082b,
269
+ 0x082b2b2b192b1908, 0x082b2b2b2b082b08, 0x082b2b2b2b082b2b, 0x1908080808080819,
270
+ 0x1908080808081908, 0x190808080808192b, 0x1908080808082b19, 0x1908080808190808,
271
+ 0x190808080819082b, 0x1908080808191919, 0x1908080808192b08, 0x1908080808192b2b,
272
+ 0x19080808082b0819, 0x19080808082b1908, 0x19080808082b192b, 0x1908080819080808,
273
+ 0x190808081908082b, 0x1908080819081919, 0x1908080819082b08, 0x1908080819082b2b,
274
+ 0x1908080819190819, 0x1908080819191908, 0x190808081919192b, 0x1908080819192b19,
275
+ 0x19080808192b0808, 0x19080808192b082b, 0x19080808192b1919, 0x190808082b080819,
276
+ 0x190808082b081908, 0x190808082b190808, 0x190808082b191919, 0x190808082b192b08,
277
+ 0x190808082b2b0819, 0x190808082b2b1908, 0x1908081908080808, 0x190808190808082b,
278
+ 0x1908081908081919, 0x1908081908082b08, 0x1908081908190819, 0x1908081908191908,
279
+ 0x190808190819192b, 0x1908081908192b19, 0x19080819082b0808, 0x19080819082b082b,
280
+ 0x19080819082b1919, 0x1908081919080819, 0x1908081919081908, 0x190808191908192b,
281
+ 0x1908081919082b19, 0x1908081919190808, 0x190808191919082b, 0x1908081919191919,
282
+ 0x1908081919192b08, 0x19080819192b0819, 0x19080819192b1908, 0x190808192b080808,
283
+ 0x190808192b08082b, 0x190808192b081919, 0x190808192b082b08, 0x190808192b190819,
284
+ 0x190808192b191908, 0x190808192b2b0808, 0x1908082b08080819, 0x1908082b08081908,
285
+ 0x1908082b08190808, 0x1908082b0819082b, 0x1908082b08191919, 0x1908082b08192b08,
286
+ 0x1908082b082b1908, 0x1908082b19080808, 0x1908082b19081919, 0x1908082b19082b08,
287
+ 0x1908082b19190819, 0x1908082b19191908, 0x1908082b192b0808, 0x1908082b2b080819,
288
+ 0x1908082b2b081908, 0x1908190808080808, 0x190819080808082b, 0x1908190808081919,
289
+ 0x1908190808082b08, 0x1908190808082b2b, 0x1908190808190819, 0x1908190808191908,
290
+ 0x190819080819192b, 0x1908190808192b19, 0x19081908082b0808, 0x19081908082b082b,
291
+ 0x19081908082b1919, 0x19081908082b2b08, 0x1908190819080819, 0x1908190819081908,
292
+ 0x190819081908192b, 0x1908190819082b19, 0x1908190819190808, 0x190819081919082b,
293
+ 0x1908190819191919, 0x1908190819192b08, 0x19081908192b0819, 0x19081908192b1908,
294
+ 0x190819082b080808, 0x190819082b08082b, 0x190819082b081919, 0x190819082b082b08,
295
+ 0x190819082b190819, 0x190819082b191908, 0x190819082b2b0808, 0x1908191908080819,
296
+ 0x1908191908081908, 0x190819190808192b, 0x1908191908082b19, 0x1908191908190808,
297
+ 0x190819190819082b, 0x1908191908191919, 0x1908191908192b08, 0x19081919082b0819,
298
+ 0x19081919082b1908, 0x1908191919080808, 0x190819191908082b, 0x1908191919081919,
299
+ 0x1908191919082b08, 0x1908191919190819, 0x1908191919191908, 0x19081919192b0808,
300
+ 0x19081919192b2b2b, 0x190819192b080819, 0x190819192b081908, 0x190819192b190808,
301
+ 0x1908192b08080808, 0x1908192b0808082b, 0x1908192b08081919, 0x1908192b08082b08,
302
+ 0x1908192b08190819, 0x1908192b08191908, 0x1908192b082b0808, 0x1908192b19080819,
303
+ 0x1908192b19081908, 0x1908192b19190808, 0x1908192b2b080808, 0x1908192b2b2b1919,
304
+ 0x19082b0808080819, 0x19082b0808081908, 0x19082b0808082b19, 0x19082b0808190808,
305
+ 0x19082b080819082b, 0x19082b0808191919, 0x19082b0808192b08, 0x19082b08082b0819,
306
+ 0x19082b08082b1908, 0x19082b0819080808, 0x19082b081908082b, 0x19082b0819081919,
307
+ 0x19082b0819082b08, 0x19082b0819190819, 0x19082b0819191908, 0x19082b08192b0808,
308
+ 0x19082b082b081908, 0x19082b082b190808, 0x19082b1908080808, 0x19082b190808082b,
309
+ 0x19082b1908081919, 0x19082b1908082b08, 0x19082b1908190819, 0x19082b1908191908,
310
+ 0x19082b19082b0808, 0x19082b1919080819, 0x19082b1919081908, 0x19082b1919190808,
311
+ 0x19082b192b080808, 0x19082b192b19192b, 0x19082b2b08080819, 0x19082b2b08081908,
312
+ 0x19082b2b08190808, 0x19082b2b19080808, 0x1919080808080808, 0x191908080808082b,
313
+ 0x1919080808081919, 0x1919080808082b08, 0x1919080808190819, 0x1919080808191908,
314
+ 0x191908080819192b, 0x1919080808192b19, 0x19190808082b0808, 0x19190808082b082b,
315
+ 0x19190808082b1919, 0x19190808082b2b08, 0x1919080819080819, 0x1919080819081908,
316
+ 0x191908081908192b, 0x1919080819082b19, 0x1919080819190808, 0x191908081919082b,
317
+ 0x1919080819191919, 0x1919080819192b08, 0x19190808192b0819, 0x19190808192b1908,
318
+ 0x191908082b080808, 0x191908082b08082b, 0x191908082b081919, 0x191908082b082b08,
319
+ 0x191908082b190819, 0x191908082b191908, 0x1919081908080819, 0x1919081908081908,
320
+ 0x191908190808192b, 0x1919081908082b19, 0x1919081908190808, 0x191908190819082b,
321
+ 0x1919081908191919, 0x1919081908192b08, 0x19190819082b0819, 0x19190819082b1908,
322
+ 0x1919081919080808, 0x191908191908082b, 0x1919081919081919, 0x1919081919082b08,
323
+ 0x1919081919190819, 0x1919081919191908, 0x19190819192b0808, 0x191908192b080819,
324
+ 0x191908192b081908, 0x191908192b190808, 0x1919082b08080808, 0x1919082b08081919,
325
+ 0x1919082b08082b08, 0x1919082b08190819, 0x1919082b08191908, 0x1919082b082b0808,
326
+ 0x1919082b19080819, 0x1919082b19081908, 0x1919082b19190808, 0x1919082b192b2b19,
327
+ 0x1919082b2b080808, 0x1919190808080819, 0x1919190808081908, 0x191919080808192b,
328
+ 0x1919190808082b19, 0x1919190808190808, 0x191919080819082b, 0x1919190808191919,
329
+ 0x1919190808192b08, 0x19191908082b0819, 0x19191908082b1908, 0x1919190819080808,
330
+ 0x191919081908082b, 0x1919190819081919, 0x1919190819082b08, 0x1919190819190819,
331
+ 0x1919190819191908, 0x19191908192b0808, 0x191919082b080819, 0x191919082b081908,
332
+ 0x191919082b190808, 0x1919191908080808, 0x191919190808082b, 0x1919191908081919,
333
+ 0x1919191908082b08, 0x1919191908190819, 0x1919191908191908, 0x19191919082b0808,
334
+ 0x1919191919080819, 0x1919191919081908, 0x1919191919190808, 0x191919192b080808,
335
+ 0x1919192b08080819, 0x1919192b08081908, 0x1919192b08190808, 0x1919192b082b192b,
336
+ 0x1919192b19080808, 0x19192b0808080808, 0x19192b080808082b, 0x19192b0808081919,
337
+ 0x19192b0808082b08, 0x19192b0808190819, 0x19192b0808191908, 0x19192b08082b0808,
338
+ 0x19192b0819080819, 0x19192b0819081908, 0x19192b0819190808, 0x19192b0819192b2b,
339
+ 0x19192b082b080808, 0x19192b1908080819, 0x19192b1908081908, 0x19192b1908190808,
340
+ 0x19192b1919080808, 0x19192b2b08080808, 0x19192b2b08192b19, 0x19192b2b2b081919,
341
+ 0x19192b2b2b2b2b08, 0x192b080808080819, 0x192b080808081908, 0x192b08080808192b,
342
+ 0x192b080808190808, 0x192b08080819082b, 0x192b080808191919, 0x192b080808192b08,
343
+ 0x192b0808082b0819, 0x192b0808082b1908, 0x192b080819080808, 0x192b080819081919,
344
+ 0x192b080819082b08, 0x192b080819190819, 0x192b080819191908, 0x192b0808192b0808,
345
+ 0x192b08082b081908, 0x192b08082b190808, 0x192b081908080808, 0x192b08190808082b,
346
+ 0x192b081908081919, 0x192b081908082b08, 0x192b081908190819, 0x192b081908191908,
347
+ 0x192b0819082b0808, 0x192b081919080819, 0x192b081919081908, 0x192b081919190808,
348
+ 0x192b08192b080808, 0x192b08192b192b19, 0x192b082b08081908, 0x192b082b08190808,
349
+ 0x192b082b19080808, 0x192b082b1919192b, 0x192b082b2b2b0819, 0x192b190808080808,
350
+ 0x192b190808081919, 0x192b190808082b08, 0x192b190808190819, 0x192b190808191908,
351
+ 0x192b1908082b0808, 0x192b190819080819, 0x192b190819081908, 0x192b190819190808,
352
+ 0x192b19082b080808, 0x192b191908080819, 0x192b191908081908, 0x192b191908190808,
353
+ 0x192b191919080808, 0x192b191919082b2b, 0x192b1919192b2b08, 0x192b19192b19082b,
354
+ 0x192b192b08080808, 0x192b192b2b191908, 0x192b2b0808080819, 0x192b2b0808081908,
355
+ 0x192b2b0808190808, 0x192b2b08192b1919, 0x192b2b082b192b08, 0x192b2b1908080808,
356
+ 0x192b2b19082b2b2b, 0x192b2b2b1908082b, 0x192b2b2b2b2b0819, 0x2b08080808080808,
357
+ 0x2b0808080808082b, 0x2b08080808081919, 0x2b08080808082b08, 0x2b08080808190819,
358
+ 0x2b08080808191908, 0x2b08080808192b19, 0x2b080808082b0808, 0x2b080808082b1919,
359
+ 0x2b08080819080819, 0x2b08080819081908, 0x2b08080819190808, 0x2b0808081919082b,
360
+ 0x2b08080819191919, 0x2b08080819192b08, 0x2b080808192b0819, 0x2b0808082b080808,
361
+ 0x2b0808082b081919, 0x2b0808082b190819, 0x2b0808082b191908, 0x2b08081908080819,
362
+ 0x2b08081908081908, 0x2b08081908082b19, 0x2b08081908190808, 0x2b0808190819082b,
363
+ 0x2b08081908191919, 0x2b08081908192b08, 0x2b080819082b0819, 0x2b080819082b1908,
364
+ 0x2b08081919080808, 0x2b0808191908082b, 0x2b08081919081919, 0x2b08081919082b08,
365
+ 0x2b08081919190819, 0x2b08081919191908, 0x2b0808192b080819, 0x2b0808192b081908,
366
+ 0x2b0808192b190808, 0x2b0808192b2b2b19, 0x2b08082b08080808, 0x2b08082b08081919,
367
+ 0x2b08082b08082b2b, 0x2b08082b08190819, 0x2b08082b08191908, 0x2b08082b19080819,
368
+ 0x2b08082b19081908, 0x2b08082b19190808, 0x2b08190808080819, 0x2b08190808081908,
369
+ 0x2b0819080808192b, 0x2b08190808082b19, 0x2b08190808190808, 0x2b0819080819082b,
370
+ 0x2b08190808191919, 0x2b08190808192b08, 0x2b081908082b0819, 0x2b08190819080808,
371
+ 0x2b0819081908082b, 0x2b08190819081919, 0x2b08190819082b08, 0x2b08190819190819,
372
+ 0x2b08190819191908, 0x2b081908192b0808, 0x2b0819082b080819, 0x2b0819082b081908,
373
+ 0x2b0819082b190808, 0x2b08191908080808, 0x2b0819190808082b, 0x2b08191908081919,
374
+ 0x2b08191908082b08, 0x2b08191908190819, 0x2b08191908191908, 0x2b081919082b0808,
375
+ 0x2b08191919080819, 0x2b08191919081908, 0x2b08191919190808, 0x2b0819192b080808,
376
+ 0x2b0819192b082b2b, 0x2b08192b08080819, 0x2b08192b08081908, 0x2b08192b08190808,
377
+ 0x2b08192b082b2b19, 0x2b08192b19080808, 0x2b082b0808080808, 0x2b082b0808081919,
378
+ 0x2b082b0808190819, 0x2b082b0808191908, 0x2b082b0819080819, 0x2b082b0819081908,
379
+ 0x2b082b0819190808, 0x2b082b082b2b082b, 0x2b082b1908080819, 0x2b082b1908081908,
380
+ 0x2b082b1919080808, 0x2b082b19192b1919, 0x2b082b2b082b082b, 0x2b082b2b19192b08,
381
+ 0x2b082b2b19192b2b, 0x2b082b2b2b08082b, 0x2b082b2b2b2b082b, 0x2b19080808080819,
382
+ 0x2b19080808081908, 0x2b19080808082b19, 0x2b19080808190808, 0x2b1908080819082b,
383
+ 0x2b19080808191919, 0x2b19080808192b08, 0x2b190808082b1908, 0x2b19080819080808,
384
+ 0x2b1908081908082b, 0x2b19080819081919, 0x2b19080819082b08, 0x2b19080819190819,
385
+ 0x2b19080819191908, 0x2b190808192b0808, 0x2b1908082b080819, 0x2b1908082b081908,
386
+ 0x2b1908082b190808, 0x2b19081908080808, 0x2b19081908081919, 0x2b19081908190819,
387
+ 0x2b19081908191908, 0x2b19081919080819, 0x2b19081919081908, 0x2b19081919190808,
388
+ 0x2b19081919192b2b, 0x2b19082b08080819, 0x2b19082b08081908, 0x2b19082b08190808,
389
+ 0x2b19082b19080808, 0x2b19082b2b2b192b, 0x2b19190808080808, 0x2b1919080808082b,
390
+ 0x2b19190808081919, 0x2b19190808082b08, 0x2b19190808190819, 0x2b19190808191908,
391
+ 0x2b191908082b0808, 0x2b19190819080819, 0x2b19190819081908, 0x2b19190819190808,
392
+ 0x2b1919082b080808, 0x2b1919082b19192b, 0x2b19191908080819, 0x2b19191908081908,
393
+ 0x2b19191908190808, 0x2b19191919080808, 0x2b1919192b192b08, 0x2b1919192b2b0819,
394
+ 0x2b19192b08080808, 0x2b19192b1908192b, 0x2b19192b192b1908, 0x2b192b0808080819,
395
+ 0x2b192b0808081908, 0x2b192b0808190808, 0x2b192b08082b192b, 0x2b192b0819080808,
396
+ 0x2b192b082b2b2b19, 0x2b192b1908080808, 0x2b192b1919082b19, 0x2b192b191919082b,
397
+ 0x2b192b2b2b190808, 0x2b2b080808080808, 0x2b2b080808081919, 0x2b2b080808082b2b,
398
+ 0x2b2b080808191908, 0x2b2b0808082b082b, 0x2b2b0808082b2b2b, 0x2b2b080819080819,
399
+ 0x2b2b080819081908, 0x2b2b080819190808, 0x2b2b08082b2b082b, 0x2b2b08082b2b2b2b,
400
+ 0x2b2b081919080808, 0x2b2b0819192b1919, 0x2b2b082b0808082b, 0x2b2b082b08082b2b,
401
+ 0x2b2b082b082b082b, 0x2b2b082b082b2b08, 0x2b2b082b082b2b2b, 0x2b2b082b2b08082b,
402
+ 0x2b2b082b2b082b08, 0x2b2b082b2b082b2b, 0x2b2b082b2b2b2b08, 0x2b2b190808080819,
403
+ 0x2b2b190808081908, 0x2b2b190808190808, 0x2b2b190819080808, 0x2b2b19082b082b19,
404
+ 0x2b2b19082b2b1908, 0x2b2b191908080808, 0x2b2b191908192b19, 0x2b2b192b19190819,
405
+ 0x2b2b2b0808082b2b, 0x2b2b2b08082b2b08, 0x2b2b2b082b2b082b, 0x2b2b2b1919191908,
406
+ 0x2b2b2b192b08192b, 0x2b2b2b2b08082b08, 0x2b2b2b2b08082b2b, 0x2b2b2b2b082b0808,
407
+ 0x2b2b2b2b082b082b, 0x2b2b2b2b082b2b08, 0x2b2b2b2b2b082b08, 0x2b2b2b2b2b2b2b2b,
408
+ };
409
+
410
+ #endif /* IQ2XS_GRID_H */