CompressedGemma commited on
Commit
33b246c
Β·
verified Β·
1 Parent(s): 05e9fe6

Upload 2 files

Browse files
Files changed (1) hide show
  1. hexstate_requantize.py +7 -45
hexstate_requantize.py CHANGED
@@ -5,10 +5,9 @@ HexState GGUF Re-Quantizer β€” GGUF-to-GGUF Q2_K quantization.
5
  Reads a source GGUF (F16/BF16/F32), copies all metadata verbatim,
6
  and re-quantizes eligible weight tensors to Q2_K. Fold-interleave is ON
7
  by default: each Q2_K block is order-stat paired (k-th smallest at k,
8
- k-th largest at k+128) and fold_perm.* I16 tensors are stored for
9
- FoldLlama, which inverts the perm at matmul (no requant). Pass
10
- --no-fold-interleave for a plain Q2_K GGUF. Optional --fold-basis is a
11
- shared dimension perm (stock llama, does not hit Fold RMSE).
12
 
13
  This bypasses the tokenizer parsing problem entirely β€” the source GGUF
14
  (from llama.cpp's convert_hf_to_gguf.py) has correct metadata.
@@ -20,7 +19,6 @@ Usage:
20
  import struct
21
  import sys
22
  import time
23
- import tempfile
24
  import os
25
  import io
26
  import ctypes
@@ -757,7 +755,7 @@ def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K):
757
 
758
 
759
  def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
760
- use_hpc, fold_interleave=False, perm_file=None):
761
  """Quantize one tensor in row chunks. kind: 'q2k' | 'q4' | 'q8'.
762
  Returns (n_out_bytes, rmse_or_None). Copies raw if source is already quant.
763
  """
@@ -801,8 +799,6 @@ def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
801
 
802
  if kind == 'q2k' and fold_interleave:
803
  perm_u8 = apply_fold_interleave(f32)
804
- if perm_file is not None:
805
- perm_file.write(np.ascontiguousarray(perm_u8, dtype='<i2').tobytes())
806
  if imp is not None:
807
  imp = apply_fold_interleave_importance(imp, perm_u8)
808
 
@@ -1351,7 +1347,7 @@ def main():
1351
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
1352
  " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
1353
  " [--fold-basis] [--no-fold-interleave]")
1354
- print(" Fold interleave is ON by default (Q2_K + fold_perm.* for FoldLlama).")
1355
  print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
1356
  sys.exit(1)
1357
 
@@ -1386,7 +1382,7 @@ def main():
1386
  if q2all:
1387
  print(" β•‘ Mode: --q2all ALL eligible tensors β†’ Q2_K (test mode) β•‘")
1388
  if fold_interleave:
1389
- print(" β•‘ Fold-interleave: ON (Q2_K + perm tensors, FoldLlama) β•‘")
1390
  if fold_basis:
1391
  print(" β•‘ Fold-basis: ON (shared dim perm, optional) β•‘")
1392
  print(f" β•‘ Chunk: {_chunk_max_elems():<7d} f32 elems/tensor (HEX_CHUNK_ELEMS) β•‘")
@@ -1581,18 +1577,6 @@ def main():
1581
  })
1582
  out_data_offset += out_size
1583
  out_data_offset = align_offset(out_data_offset)
1584
- if fold_interleave and out_type == GGML_TYPE_Q2_K:
1585
- psize = int(ti['n_elements']) * 2
1586
- out_tensor_infos.append({
1587
- 'name': 'fold_perm.' + ti['name'],
1588
- 'n_dims': 1,
1589
- 'dims': [int(ti['n_elements'])],
1590
- 'type': GGML_TYPE_I16,
1591
- 'offset': out_data_offset,
1592
- 'data_size': psize,
1593
- })
1594
- out_data_offset += psize
1595
- out_data_offset = align_offset(out_data_offset)
1596
 
1597
  # ── Update KV pairs ──
1598
  updated_kv = []
@@ -1687,14 +1671,6 @@ def main():
1687
  else:
1688
  updated_kv.append((key, vtype, raw_value))
1689
 
1690
- if fold_interleave:
1691
- val = b'true'
1692
- updated_kv.append((
1693
- 'hexstate.fold_interleave', 8,
1694
- struct.pack('<Q', len(val)) + val,
1695
- ))
1696
- print(" KV hexstate.fold_interleave = true")
1697
-
1698
  # ── Write output GGUF ──
1699
  print(" Writing output GGUF...")
1700
  with open(output_path, 'wb') as fout:
@@ -1768,13 +1744,10 @@ def main():
1768
  total_quant_bytes += nbytes
1769
 
1770
  elif plan:
1771
- perm_file = None
1772
- if fold_interleave:
1773
- perm_file = tempfile.SpooledTemporaryFile(max_size=64 * 1024 * 1024)
1774
  nbytes, rmse, sigma = _stream_quantize(
1775
  fin, fout, ti, abs_offset, 'q2k', fb_perms,
1776
  imatrix_data, use_hpc,
1777
- fold_interleave=fold_interleave, perm_file=perm_file)
1778
  if rmse is not None:
1779
  q2k_rmse_sum += rmse
1780
  q2k_tensor_count += 1
@@ -1785,17 +1758,6 @@ def main():
1785
  quant_count += 1
1786
  total_quant_bytes += nbytes
1787
  pad = align_offset(fout.tell()) - fout.tell()
1788
- if pad > 0:
1789
- fout.write(b'\x00' * pad)
1790
- if perm_file is not None:
1791
- perm_file.seek(0)
1792
- while True:
1793
- chunk = perm_file.read(16 * 1024 * 1024)
1794
- if not chunk:
1795
- break
1796
- fout.write(chunk)
1797
- perm_file.close()
1798
- pad = align_offset(fout.tell()) - fout.tell()
1799
  if pad > 0:
1800
  fout.write(b'\x00' * pad)
1801
  continue
 
5
  Reads a source GGUF (F16/BF16/F32), copies all metadata verbatim,
6
  and re-quantizes eligible weight tensors to Q2_K. Fold-interleave is ON
7
  by default: each Q2_K block is order-stat paired (k-th smallest at k,
8
+ k-th largest at k+128) then written as ordinary 84-byte Q2_K β€” stored
9
+ order is decode order, no sidecar. Pass --no-fold-interleave for
10
+ identity-layout Q2_K. Optional --fold-basis is a shared dimension perm.
 
11
 
12
  This bypasses the tokenizer parsing problem entirely β€” the source GGUF
13
  (from llama.cpp's convert_hf_to_gguf.py) has correct metadata.
 
19
  import struct
20
  import sys
21
  import time
 
22
  import os
23
  import io
24
  import ctypes
 
755
 
756
 
757
  def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
758
+ use_hpc, fold_interleave=False):
759
  """Quantize one tensor in row chunks. kind: 'q2k' | 'q4' | 'q8'.
760
  Returns (n_out_bytes, rmse_or_None). Copies raw if source is already quant.
761
  """
 
799
 
800
  if kind == 'q2k' and fold_interleave:
801
  perm_u8 = apply_fold_interleave(f32)
 
 
802
  if imp is not None:
803
  imp = apply_fold_interleave_importance(imp, perm_u8)
804
 
 
1347
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
1348
  " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
1349
  " [--fold-basis] [--no-fold-interleave]")
1350
+ print(" Fold interleave is ON by default (in-block Q2_K layout, no extra tensors).")
1351
  print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
1352
  sys.exit(1)
1353
 
 
1382
  if q2all:
1383
  print(" β•‘ Mode: --q2all ALL eligible tensors β†’ Q2_K (test mode) β•‘")
1384
  if fold_interleave:
1385
+ print(" β•‘ Fold-interleave: ON (in-block Q2_K, same file size) β•‘")
1386
  if fold_basis:
1387
  print(" β•‘ Fold-basis: ON (shared dim perm, optional) β•‘")
1388
  print(f" β•‘ Chunk: {_chunk_max_elems():<7d} f32 elems/tensor (HEX_CHUNK_ELEMS) β•‘")
 
1577
  })
1578
  out_data_offset += out_size
1579
  out_data_offset = align_offset(out_data_offset)
 
 
 
 
 
 
 
 
 
 
 
 
1580
 
1581
  # ── Update KV pairs ──
1582
  updated_kv = []
 
1671
  else:
1672
  updated_kv.append((key, vtype, raw_value))
1673
 
 
 
 
 
 
 
 
 
1674
  # ── Write output GGUF ──
1675
  print(" Writing output GGUF...")
1676
  with open(output_path, 'wb') as fout:
 
1744
  total_quant_bytes += nbytes
1745
 
1746
  elif plan:
 
 
 
1747
  nbytes, rmse, sigma = _stream_quantize(
1748
  fin, fout, ti, abs_offset, 'q2k', fb_perms,
1749
  imatrix_data, use_hpc,
1750
+ fold_interleave=fold_interleave)
1751
  if rmse is not None:
1752
  q2k_rmse_sum += rmse
1753
  q2k_tensor_count += 1
 
1758
  quant_count += 1
1759
  total_quant_bytes += nbytes
1760
  pad = align_offset(fout.tell()) - fout.tell()
 
 
 
 
 
 
 
 
 
 
 
1761
  if pad > 0:
1762
  fout.write(b'\x00' * pad)
1763
  continue