Upload 2 files
Browse files- hexstate_requantize.py +7 -45
hexstate_requantize.py
CHANGED
|
@@ -5,10 +5,9 @@ HexState GGUF Re-Quantizer β GGUF-to-GGUF Q2_K quantization.
|
|
| 5 |
Reads a source GGUF (F16/BF16/F32), copies all metadata verbatim,
|
| 6 |
and re-quantizes eligible weight tensors to Q2_K. Fold-interleave is ON
|
| 7 |
by default: each Q2_K block is order-stat paired (k-th smallest at k,
|
| 8 |
-
k-th largest at k+128)
|
| 9 |
-
|
| 10 |
-
-
|
| 11 |
-
shared dimension perm (stock llama, does not hit Fold RMSE).
|
| 12 |
|
| 13 |
This bypasses the tokenizer parsing problem entirely β the source GGUF
|
| 14 |
(from llama.cpp's convert_hf_to_gguf.py) has correct metadata.
|
|
@@ -20,7 +19,6 @@ Usage:
|
|
| 20 |
import struct
|
| 21 |
import sys
|
| 22 |
import time
|
| 23 |
-
import tempfile
|
| 24 |
import os
|
| 25 |
import io
|
| 26 |
import ctypes
|
|
@@ -757,7 +755,7 @@ def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K):
|
|
| 757 |
|
| 758 |
|
| 759 |
def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
|
| 760 |
-
use_hpc, fold_interleave=False
|
| 761 |
"""Quantize one tensor in row chunks. kind: 'q2k' | 'q4' | 'q8'.
|
| 762 |
Returns (n_out_bytes, rmse_or_None). Copies raw if source is already quant.
|
| 763 |
"""
|
|
@@ -801,8 +799,6 @@ def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
|
|
| 801 |
|
| 802 |
if kind == 'q2k' and fold_interleave:
|
| 803 |
perm_u8 = apply_fold_interleave(f32)
|
| 804 |
-
if perm_file is not None:
|
| 805 |
-
perm_file.write(np.ascontiguousarray(perm_u8, dtype='<i2').tobytes())
|
| 806 |
if imp is not None:
|
| 807 |
imp = apply_fold_interleave_importance(imp, perm_u8)
|
| 808 |
|
|
@@ -1351,7 +1347,7 @@ def main():
|
|
| 1351 |
print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
|
| 1352 |
" [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
|
| 1353 |
" [--fold-basis] [--no-fold-interleave]")
|
| 1354 |
-
print(" Fold interleave is ON by default (Q2_K
|
| 1355 |
print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
|
| 1356 |
sys.exit(1)
|
| 1357 |
|
|
@@ -1386,7 +1382,7 @@ def main():
|
|
| 1386 |
if q2all:
|
| 1387 |
print(" β Mode: --q2all ALL eligible tensors β Q2_K (test mode) β")
|
| 1388 |
if fold_interleave:
|
| 1389 |
-
print(" β Fold-interleave: ON (Q2_K
|
| 1390 |
if fold_basis:
|
| 1391 |
print(" β Fold-basis: ON (shared dim perm, optional) β")
|
| 1392 |
print(f" β Chunk: {_chunk_max_elems():<7d} f32 elems/tensor (HEX_CHUNK_ELEMS) β")
|
|
@@ -1581,18 +1577,6 @@ def main():
|
|
| 1581 |
})
|
| 1582 |
out_data_offset += out_size
|
| 1583 |
out_data_offset = align_offset(out_data_offset)
|
| 1584 |
-
if fold_interleave and out_type == GGML_TYPE_Q2_K:
|
| 1585 |
-
psize = int(ti['n_elements']) * 2
|
| 1586 |
-
out_tensor_infos.append({
|
| 1587 |
-
'name': 'fold_perm.' + ti['name'],
|
| 1588 |
-
'n_dims': 1,
|
| 1589 |
-
'dims': [int(ti['n_elements'])],
|
| 1590 |
-
'type': GGML_TYPE_I16,
|
| 1591 |
-
'offset': out_data_offset,
|
| 1592 |
-
'data_size': psize,
|
| 1593 |
-
})
|
| 1594 |
-
out_data_offset += psize
|
| 1595 |
-
out_data_offset = align_offset(out_data_offset)
|
| 1596 |
|
| 1597 |
# ββ Update KV pairs ββ
|
| 1598 |
updated_kv = []
|
|
@@ -1687,14 +1671,6 @@ def main():
|
|
| 1687 |
else:
|
| 1688 |
updated_kv.append((key, vtype, raw_value))
|
| 1689 |
|
| 1690 |
-
if fold_interleave:
|
| 1691 |
-
val = b'true'
|
| 1692 |
-
updated_kv.append((
|
| 1693 |
-
'hexstate.fold_interleave', 8,
|
| 1694 |
-
struct.pack('<Q', len(val)) + val,
|
| 1695 |
-
))
|
| 1696 |
-
print(" KV hexstate.fold_interleave = true")
|
| 1697 |
-
|
| 1698 |
# ββ Write output GGUF ββ
|
| 1699 |
print(" Writing output GGUF...")
|
| 1700 |
with open(output_path, 'wb') as fout:
|
|
@@ -1768,13 +1744,10 @@ def main():
|
|
| 1768 |
total_quant_bytes += nbytes
|
| 1769 |
|
| 1770 |
elif plan:
|
| 1771 |
-
perm_file = None
|
| 1772 |
-
if fold_interleave:
|
| 1773 |
-
perm_file = tempfile.SpooledTemporaryFile(max_size=64 * 1024 * 1024)
|
| 1774 |
nbytes, rmse, sigma = _stream_quantize(
|
| 1775 |
fin, fout, ti, abs_offset, 'q2k', fb_perms,
|
| 1776 |
imatrix_data, use_hpc,
|
| 1777 |
-
fold_interleave=fold_interleave
|
| 1778 |
if rmse is not None:
|
| 1779 |
q2k_rmse_sum += rmse
|
| 1780 |
q2k_tensor_count += 1
|
|
@@ -1785,17 +1758,6 @@ def main():
|
|
| 1785 |
quant_count += 1
|
| 1786 |
total_quant_bytes += nbytes
|
| 1787 |
pad = align_offset(fout.tell()) - fout.tell()
|
| 1788 |
-
if pad > 0:
|
| 1789 |
-
fout.write(b'\x00' * pad)
|
| 1790 |
-
if perm_file is not None:
|
| 1791 |
-
perm_file.seek(0)
|
| 1792 |
-
while True:
|
| 1793 |
-
chunk = perm_file.read(16 * 1024 * 1024)
|
| 1794 |
-
if not chunk:
|
| 1795 |
-
break
|
| 1796 |
-
fout.write(chunk)
|
| 1797 |
-
perm_file.close()
|
| 1798 |
-
pad = align_offset(fout.tell()) - fout.tell()
|
| 1799 |
if pad > 0:
|
| 1800 |
fout.write(b'\x00' * pad)
|
| 1801 |
continue
|
|
|
|
| 5 |
Reads a source GGUF (F16/BF16/F32), copies all metadata verbatim,
|
| 6 |
and re-quantizes eligible weight tensors to Q2_K. Fold-interleave is ON
|
| 7 |
by default: each Q2_K block is order-stat paired (k-th smallest at k,
|
| 8 |
+
k-th largest at k+128) then written as ordinary 84-byte Q2_K β stored
|
| 9 |
+
order is decode order, no sidecar. Pass --no-fold-interleave for
|
| 10 |
+
identity-layout Q2_K. Optional --fold-basis is a shared dimension perm.
|
|
|
|
| 11 |
|
| 12 |
This bypasses the tokenizer parsing problem entirely β the source GGUF
|
| 13 |
(from llama.cpp's convert_hf_to_gguf.py) has correct metadata.
|
|
|
|
| 19 |
import struct
|
| 20 |
import sys
|
| 21 |
import time
|
|
|
|
| 22 |
import os
|
| 23 |
import io
|
| 24 |
import ctypes
|
|
|
|
| 755 |
|
| 756 |
|
| 757 |
def _stream_quantize(fin, fout, ti, abs_offset, kind, fb_perms, imatrix_data,
|
| 758 |
+
use_hpc, fold_interleave=False):
|
| 759 |
"""Quantize one tensor in row chunks. kind: 'q2k' | 'q4' | 'q8'.
|
| 760 |
Returns (n_out_bytes, rmse_or_None). Copies raw if source is already quant.
|
| 761 |
"""
|
|
|
|
| 799 |
|
| 800 |
if kind == 'q2k' and fold_interleave:
|
| 801 |
perm_u8 = apply_fold_interleave(f32)
|
|
|
|
|
|
|
| 802 |
if imp is not None:
|
| 803 |
imp = apply_fold_interleave_importance(imp, perm_u8)
|
| 804 |
|
|
|
|
| 1347 |
print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
|
| 1348 |
" [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
|
| 1349 |
" [--fold-basis] [--no-fold-interleave]")
|
| 1350 |
+
print(" Fold interleave is ON by default (in-block Q2_K layout, no extra tensors).")
|
| 1351 |
print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)")
|
| 1352 |
sys.exit(1)
|
| 1353 |
|
|
|
|
| 1382 |
if q2all:
|
| 1383 |
print(" β Mode: --q2all ALL eligible tensors β Q2_K (test mode) β")
|
| 1384 |
if fold_interleave:
|
| 1385 |
+
print(" β Fold-interleave: ON (in-block Q2_K, same file size) β")
|
| 1386 |
if fold_basis:
|
| 1387 |
print(" β Fold-basis: ON (shared dim perm, optional) β")
|
| 1388 |
print(f" β Chunk: {_chunk_max_elems():<7d} f32 elems/tensor (HEX_CHUNK_ELEMS) β")
|
|
|
|
| 1577 |
})
|
| 1578 |
out_data_offset += out_size
|
| 1579 |
out_data_offset = align_offset(out_data_offset)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1580 |
|
| 1581 |
# ββ Update KV pairs ββ
|
| 1582 |
updated_kv = []
|
|
|
|
| 1671 |
else:
|
| 1672 |
updated_kv.append((key, vtype, raw_value))
|
| 1673 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1674 |
# ββ Write output GGUF ββ
|
| 1675 |
print(" Writing output GGUF...")
|
| 1676 |
with open(output_path, 'wb') as fout:
|
|
|
|
| 1744 |
total_quant_bytes += nbytes
|
| 1745 |
|
| 1746 |
elif plan:
|
|
|
|
|
|
|
|
|
|
| 1747 |
nbytes, rmse, sigma = _stream_quantize(
|
| 1748 |
fin, fout, ti, abs_offset, 'q2k', fb_perms,
|
| 1749 |
imatrix_data, use_hpc,
|
| 1750 |
+
fold_interleave=fold_interleave)
|
| 1751 |
if rmse is not None:
|
| 1752 |
q2k_rmse_sum += rmse
|
| 1753 |
q2k_tensor_count += 1
|
|
|
|
| 1758 |
quant_count += 1
|
| 1759 |
total_quant_bytes += nbytes
|
| 1760 |
pad = align_offset(fout.tell()) - fout.tell()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1761 |
if pad > 0:
|
| 1762 |
fout.write(b'\x00' * pad)
|
| 1763 |
continue
|