Fold prototype
Browse files- hexstate_requantize.py +111 -564
hexstate_requantize.py
CHANGED
|
@@ -659,436 +659,98 @@ def dequant_q2k_fast(q2k_bytes, n_blocks):
|
|
| 659 |
# across all fold pairs, so the spectral objective's vesica/DC terms
|
| 660 |
# can be driven to zero by the unchanged HPC pipeline.
|
| 661 |
#
|
| 662 |
-
#
|
| 663 |
-
#
|
| 664 |
-
#
|
| 665 |
-
#
|
| 666 |
-
#
|
| 667 |
-
|
| 668 |
-
|
| 669 |
-
|
| 670 |
-
|
| 671 |
-
|
| 672 |
-
|
| 673 |
-
|
| 674 |
-
|
| 675 |
-
|
| 676 |
-
|
| 677 |
-
|
| 678 |
-
|
| 679 |
-
for cb in range(0, n_blocks, _FI_CHUNK_BLOCKS):
|
| 680 |
-
ce = min(cb + _FI_CHUNK_BLOCKS, n_blocks)
|
| 681 |
-
s = cb * block_size
|
| 682 |
-
e = ce * block_size
|
| 683 |
-
chunk = f32_data[s:e].reshape(ce - cb, block_size)
|
| 684 |
-
|
| 685 |
-
sorted_idx = np.argsort(chunk, axis=1)
|
| 686 |
-
perm = np.empty_like(sorted_idx)
|
| 687 |
-
perm[:, :half] = sorted_idx[:, :half]
|
| 688 |
-
perm[:, half:] = sorted_idx[:, -1:half-1:-1]
|
| 689 |
-
del sorted_idx
|
| 690 |
-
|
| 691 |
-
f32_data[s:e] = np.take_along_axis(chunk, perm, axis=1).reshape(-1)
|
| 692 |
-
all_perms[s:e] = perm.astype(np.uint8).reshape(-1)
|
| 693 |
-
del perm, chunk
|
| 694 |
-
|
| 695 |
-
return all_perms
|
| 696 |
-
|
| 697 |
-
|
| 698 |
-
def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K):
|
| 699 |
-
"""Permute an importance vector with the same fold permutation.
|
| 700 |
-
Returns a new array (importance is typically much smaller than
|
| 701 |
-
weights for tiled imatrix, so the copy is fine)."""
|
| 702 |
-
n = len(importance)
|
| 703 |
-
n_blocks = n // block_size
|
| 704 |
-
blocks = importance.reshape(n_blocks, block_size)
|
| 705 |
-
perms = perm_flat[:n].reshape(n_blocks, block_size).astype(np.intp)
|
| 706 |
-
permuted = np.take_along_axis(blocks, perms, axis=1)
|
| 707 |
-
return permuted.reshape(-1)
|
| 708 |
-
|
| 709 |
-
|
| 710 |
-
def is_attention_tensor(name):
|
| 711 |
-
"""Detect attention Q/K/V/O projection tensors.
|
| 712 |
-
These are the most sensitive to quantization and get promoted to Q4_0."""
|
| 713 |
-
|
| 714 |
-
# βββ Fold-Basis Change ββββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 715 |
-
# Whole-dimension permutations that propagate losslessly through the graph.
|
| 716 |
-
# P_hidden: one global permutation of the residual-stream dimension.
|
| 717 |
-
# Permutes columns of everything that reads from it (Q/K/V,
|
| 718 |
-
# gate, up, embeddings) and rows of everything that writes
|
| 719 |
-
# to it (O, down). Layer norms are 1D β just permute entries.
|
| 720 |
-
# P_inter_N: one permutation per layer of the FFN intermediate dimension.
|
| 721 |
-
# Permutes columns of W_down (the dimension Q2_K blocks span)
|
| 722 |
-
# and rows of W_gate/W_up (output dim, no block effect).
|
| 723 |
-
#
|
| 724 |
-
# The permutations are chosen by stride-interleaving the columns sorted
|
| 725 |
-
# by L2 energy: the k-th lowest-energy column is paired with the k-th
|
| 726 |
-
# highest at fold-complementary positions (k, k+128) within each QK_K
|
| 727 |
-
# block. This gives every block balanced fold pairs for vesica/DC
|
| 728 |
-
# cancellation β the same spectral benefit as the per-block sort, but
|
| 729 |
-
# baked into the weight matrix so the output GGUF is self-contained.
|
| 730 |
-
|
| 731 |
-
import re as _re
|
| 732 |
-
|
| 733 |
-
def _fb_tensor_perms(name):
|
| 734 |
-
"""Return list of (perm_key, axis) for this tensor.
|
| 735 |
-
axis: 0 β permute dims[0] (columns/input, what blocks span)
|
| 736 |
-
1 β permute dims[1] (rows/output)
|
| 737 |
-
'1d' β 1D parameter (norms)
|
| 738 |
"""
|
| 739 |
-
|
| 740 |
-
|
| 741 |
-
|
| 742 |
-
|
| 743 |
-
|
| 744 |
-
|
| 745 |
-
# Hidden dim as output (rows = dims[1])
|
| 746 |
-
if _re.match(r'blk\.\d+\.attn_output\.weight', name) or \
|
| 747 |
-
_re.match(r'blk\.\d+\.ffn_down\.weight', name):
|
| 748 |
-
perms.append(('hidden', 1))
|
| 749 |
-
# Hidden dim as 1D
|
| 750 |
-
if _re.match(r'blk\.\d+\.(attn|ffn)_norm\.weight', name) or \
|
| 751 |
-
name == 'output_norm.weight':
|
| 752 |
-
perms.append(('hidden', '1d'))
|
| 753 |
-
# Intermediate dim as input (cols = dims[0]) β THIS is the big win
|
| 754 |
-
m = _re.match(r'blk\.(\d+)\.ffn_down\.weight', name)
|
| 755 |
-
if m:
|
| 756 |
-
perms.append((f'inter_{m.group(1)}', 0))
|
| 757 |
-
# Intermediate dim as output (rows = dims[1])
|
| 758 |
-
m = _re.match(r'blk\.(\d+)\.ffn_(gate|up)\.weight', name)
|
| 759 |
-
if m:
|
| 760 |
-
perms.append((f'inter_{m.group(1)}', 1))
|
| 761 |
-
return perms
|
| 762 |
-
|
| 763 |
-
|
| 764 |
-
def _fb_compute_perm(col_stats, block_size=QK_K):
|
| 765 |
-
"""Per-block-group fold-optimal permutation.
|
| 766 |
-
|
| 767 |
-
For each group of 256 consecutive columns, sorts by the column
|
| 768 |
-
statistic and interleaves: k-th smallest at position k, k-th
|
| 769 |
-
largest at position k+128. This IS the fold principle β each
|
| 770 |
-
block group gets its own sort-interleave β while remaining a
|
| 771 |
-
valid column permutation that propagates through the graph.
|
| 772 |
-
Stock llama.cpp compatible: no modified dequant, no sidecar."""
|
| 773 |
-
dim = len(col_stats)
|
| 774 |
-
n_blocks = dim // block_size
|
| 775 |
-
half = block_size // 2
|
| 776 |
-
remainder = dim % block_size
|
| 777 |
-
|
| 778 |
-
if n_blocks == 0:
|
| 779 |
-
return np.arange(dim, dtype=np.intp)
|
| 780 |
-
|
| 781 |
-
perm = np.arange(dim, dtype=np.intp) # identity for remainder
|
| 782 |
-
|
| 783 |
-
for b in range(n_blocks):
|
| 784 |
-
s = b * block_size
|
| 785 |
-
e = s + block_size
|
| 786 |
-
group_sorted = np.argsort(col_stats[s:e])
|
| 787 |
-
perm[s:s + half] = s + group_sorted[:half]
|
| 788 |
-
perm[s + half:e] = s + group_sorted[-1:half-1:-1]
|
| 789 |
-
|
| 790 |
return perm
|
| 791 |
|
| 792 |
|
| 793 |
-
def
|
| 794 |
-
"""Apply
|
| 795 |
-
axis: 0 β dims[0] columns, 1 β dims[1] rows, '1d' β 1D vector.
|
| 796 |
-
Returns original f32 unchanged if perm length doesn't match the dim."""
|
| 797 |
-
if axis == '1d':
|
| 798 |
-
if len(perm) != len(f32):
|
| 799 |
-
return f32
|
| 800 |
-
return f32[perm].copy()
|
| 801 |
-
d0 = int(dims[0])
|
| 802 |
-
d1 = int(dims[1]) if len(dims) > 1 else 1
|
| 803 |
-
# Dimension safety: skip if perm doesn't match the target axis
|
| 804 |
-
if axis == 0 and len(perm) != d0:
|
| 805 |
-
return f32
|
| 806 |
-
if axis == 1 and len(perm) != d1:
|
| 807 |
-
return f32
|
| 808 |
-
M = f32.reshape(d1, d0)
|
| 809 |
-
if axis == 0:
|
| 810 |
-
M = M[:, perm]
|
| 811 |
-
else:
|
| 812 |
-
M = M[perm, :]
|
| 813 |
-
return M.reshape(-1).copy()
|
| 814 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 815 |
|
| 816 |
-
|
| 817 |
-
|
| 818 |
-
|
| 819 |
-
|
| 820 |
-
|
| 821 |
-
|
| 822 |
-
# objective is satisfied structurally, not by chasing it.
|
| 823 |
-
#
|
| 824 |
-
# Key properties:
|
| 825 |
-
# H is its own inverse: HΒ·H = nΒ·I β Hβ»ΒΉ = H/n
|
| 826 |
-
# H preserves L2 norm: βHxββ = βnΒ·βxββ (with unnormalised H)
|
| 827 |
-
# H is deterministic: defined entirely by block size
|
| 828 |
-
# Zero storage overhead: no permutation indices, no sidecar
|
| 829 |
-
#
|
| 830 |
-
# Baked into the GGUF: a metadata key signals "Hadamard Q2_K".
|
| 831 |
-
# The inference dequant kernel applies the same transform after
|
| 832 |
-
# unpacking: deq_orig = (1/n) Β· H Β· deq_packed
|
| 833 |
-
#
|
| 834 |
-
# This is QuIP#'s insight applied per-block instead of per-row,
|
| 835 |
-
# composed with the full HPC sieve pipeline.
|
| 836 |
-
|
| 837 |
-
def _wht_block(block):
|
| 838 |
-
"""In-place unnormalised Walsh-Hadamard transform of a power-of-2 array."""
|
| 839 |
-
n = len(block)
|
| 840 |
-
h = 1
|
| 841 |
-
while h < n:
|
| 842 |
-
for i in range(0, n, h * 2):
|
| 843 |
-
for j in range(i, i + h):
|
| 844 |
-
x = block[j]
|
| 845 |
-
y = block[j + h]
|
| 846 |
-
block[j] = x + y
|
| 847 |
-
block[j + h] = x - y
|
| 848 |
-
h *= 2
|
| 849 |
-
|
| 850 |
-
|
| 851 |
-
def apply_fold_hadamard(f32_data, block_size=QK_K):
|
| 852 |
-
"""Apply normalised per-block WHT to a flat tensor IN-PLACE.
|
| 853 |
-
Each block is transformed independently. O(n log n) per block."""
|
| 854 |
n = len(f32_data)
|
| 855 |
n_blocks = n // block_size
|
| 856 |
-
|
| 857 |
-
|
| 858 |
-
|
| 859 |
-
|
| 860 |
-
|
| 861 |
-
|
| 862 |
-
|
| 863 |
-
|
| 864 |
-
|
| 865 |
-
|
| 866 |
-
|
| 867 |
-
|
| 868 |
-
|
| 869 |
-
|
| 870 |
-
|
| 871 |
-
|
| 872 |
-
|
| 873 |
-
|
| 874 |
-
|
| 875 |
-
|
| 876 |
-
|
| 877 |
-
|
| 878 |
-
# 2. Quantize sorted blocks with full HPC pipeline
|
| 879 |
-
# (sieve, Viterbi, greedy descent, Floyd-Steinberg β everything)
|
| 880 |
-
# 3. Unsort the 2-bit codes back to original positions
|
| 881 |
-
# 4. Refit sub-block scales (Ls, Lm) via least-squares
|
| 882 |
-
# 5. Repack as standard Q2_K β stock llama.cpp compatible
|
| 883 |
-
#
|
| 884 |
-
# The HPC pipeline sees sorted blocks with narrow sub-block ranges
|
| 885 |
-
# and balanced fold pairs. The code assignments it produces are
|
| 886 |
-
# globally optimal in sorted space. After unsorting, the codes
|
| 887 |
-
# retain their relative quality β element k still gets the code
|
| 888 |
-
# that the HPC pipeline chose for it. Only the sub-block scales
|
| 889 |
-
# need refitting because elements moved across sub-block boundaries.
|
| 890 |
-
#
|
| 891 |
-
# Combined with fold-basis (column permutation), the sort-refit
|
| 892 |
-
# captures both coarse structure (consistent across rows) and fine
|
| 893 |
-
# per-block structure (row-specific), all in standard Q2_K format.
|
| 894 |
-
|
| 895 |
-
# Sub-block index mapping for Q2_K:
|
| 896 |
-
# Element i maps to sub-block j as follows:
|
| 897 |
-
# half = i // 128
|
| 898 |
-
# sub = (i % 128) // 32
|
| 899 |
-
# k = (i % 128) % 32
|
| 900 |
-
# j = half * 8 + sub * 2 + (k >= 16)
|
| 901 |
-
# Sub-block j covers 16 elements starting at:
|
| 902 |
-
# base = (j//8)*128 + ((j%8)//2)*32 + 16*(j%2)
|
| 903 |
-
|
| 904 |
-
_Q2K_SUB_BASES = np.array([
|
| 905 |
-
(j // 8) * 128 + ((j % 8) // 2) * 32 + 16 * (j % 2)
|
| 906 |
-
for j in range(16)
|
| 907 |
-
], dtype=np.int32)
|
| 908 |
-
|
| 909 |
-
|
| 910 |
-
def _sort_refit_q2k(f32_orig, q2k_sorted_bytes, sort_perms, n_blocks):
|
| 911 |
-
"""Unsort HPC-quantized codes and refit sub-block scales.
|
| 912 |
-
|
| 913 |
-
f32_orig: [n_blocks, 256] original (unsorted) weight values
|
| 914 |
-
q2k_sorted_bytes: packed Q2_K data from HPC quantization of sorted blocks
|
| 915 |
-
sort_perms: [n_blocks, 256] argsort permutations used to sort
|
| 916 |
-
n_blocks: number of Q2_K blocks
|
| 917 |
-
|
| 918 |
-
Returns repacked Q2_K bytes with unsorted codes and refit scales.
|
| 919 |
-
Fully vectorized over blocks β only loops over 16 sub-blocks."""
|
| 920 |
-
|
| 921 |
-
BLOCK_BYTES = 84
|
| 922 |
-
|
| 923 |
-
q2k_arr = np.frombuffer(q2k_sorted_bytes, dtype=np.uint8).reshape(n_blocks, BLOCK_BYTES).copy()
|
| 924 |
-
|
| 925 |
-
# ββ Extract d, dmin ββ
|
| 926 |
-
d_arr = np.frombuffer(q2k_arr[:, 0:2].tobytes(),
|
| 927 |
-
dtype=np.float16).reshape(n_blocks).astype(np.float32)
|
| 928 |
-
dmin_arr = np.frombuffer(q2k_arr[:, 2:4].tobytes(),
|
| 929 |
-
dtype=np.float16).reshape(n_blocks).astype(np.float32)
|
| 930 |
-
|
| 931 |
-
# ββ Unpack 2-bit codes: [n_blocks, 256] ββ
|
| 932 |
-
qs_raw = q2k_arr[:, 20:84] # [n_blocks, 64]
|
| 933 |
-
q_sorted = np.zeros((n_blocks, 256), dtype=np.int32)
|
| 934 |
-
for h in range(2):
|
| 935 |
-
qs_half = qs_raw[:, h * 32:(h + 1) * 32]
|
| 936 |
-
for s in range(4):
|
| 937 |
-
q_sorted[:, h * 128 + s * 32: h * 128 + (s + 1) * 32] = (qs_half >> (s * 2)) & 3
|
| 938 |
-
|
| 939 |
-
# ββ Unsort codes ββ
|
| 940 |
-
inv_perms = np.argsort(sort_perms, axis=1)
|
| 941 |
-
q_orig = np.take_along_axis(q_sorted, inv_perms, axis=1) # [n_blocks, 256]
|
| 942 |
-
|
| 943 |
-
# ββ Refit sub-block scales via OLS ββ
|
| 944 |
-
Ls_new = np.zeros((n_blocks, 16), dtype=np.int32)
|
| 945 |
-
Lm_new = np.zeros((n_blocks, 16), dtype=np.int32)
|
| 946 |
-
|
| 947 |
-
for j in range(16):
|
| 948 |
-
base = int(_Q2K_SUB_BASES[j])
|
| 949 |
-
x_sub = f32_orig[:, base:base + 16].astype(np.float64)
|
| 950 |
-
q_sub = q_orig[:, base:base + 16].astype(np.float64)
|
| 951 |
-
|
| 952 |
-
Sq = np.sum(q_sub, axis=1)
|
| 953 |
-
Sqq = np.sum(q_sub ** 2, axis=1)
|
| 954 |
-
Sx = np.sum(x_sub, axis=1)
|
| 955 |
-
Sqx = np.sum(q_sub * x_sub, axis=1)
|
| 956 |
-
|
| 957 |
-
n_el = 16.0
|
| 958 |
-
denom = n_el * Sqq - Sq * Sq
|
| 959 |
-
safe_denom = np.where(np.abs(denom) > 1e-10, denom, 1.0)
|
| 960 |
-
|
| 961 |
-
alpha = (n_el * Sqx - Sq * Sx) / safe_denom
|
| 962 |
-
beta = (Sx - alpha * Sq) / n_el
|
| 963 |
-
|
| 964 |
-
safe_d = np.where(d_arr > 1e-20, d_arr, 1.0)
|
| 965 |
-
safe_dmin = np.where(dmin_arr > 1e-20, dmin_arr, 1.0)
|
| 966 |
-
|
| 967 |
-
Ls_f = alpha / safe_d
|
| 968 |
-
Lm_f = -beta / safe_dmin
|
| 969 |
-
|
| 970 |
-
valid = (np.abs(denom) > 1e-10) & (d_arr > 1e-20) & (dmin_arr > 1e-20)
|
| 971 |
-
Ls_new[:, j] = np.where(valid, np.clip(np.round(Ls_f), 0, 15), 0).astype(np.int32)
|
| 972 |
-
Lm_new[:, j] = np.where(valid, np.clip(np.round(Lm_f), 0, 15), 0).astype(np.int32)
|
| 973 |
-
|
| 974 |
-
# ββ Repack scales ββ
|
| 975 |
-
for j in range(16):
|
| 976 |
-
q2k_arr[:, 4 + j] = (Ls_new[:, j].astype(np.uint8) & 0x0F) | \
|
| 977 |
-
((Lm_new[:, j].astype(np.uint8) & 0x0F) << 4)
|
| 978 |
-
|
| 979 |
-
# ββ Repack qs with unsorted codes ββ
|
| 980 |
-
new_qs = np.zeros((n_blocks, 64), dtype=np.uint8)
|
| 981 |
-
for h in range(2):
|
| 982 |
-
for s in range(4):
|
| 983 |
-
codes = q_orig[:, h * 128 + s * 32: h * 128 + (s + 1) * 32].astype(np.uint8)
|
| 984 |
-
new_qs[:, h * 32:(h + 1) * 32] |= codes << (s * 2)
|
| 985 |
-
q2k_arr[:, 20:84] = new_qs
|
| 986 |
-
|
| 987 |
-
return q2k_arr.tobytes()
|
| 988 |
-
|
| 989 |
-
|
| 990 |
-
def _fb_prepass(fin, tensor_infos, data_section_start):
|
| 991 |
-
"""Pre-pass: compute fold-basis permutations from weight statistics.
|
| 992 |
-
|
| 993 |
-
Memory-lean: reads only a small sample of tensors.
|
| 994 |
-
- P_hidden: from the first 3 layers' W_q only (3 tensor reads)
|
| 995 |
-
- P_inter_N: from W_down[N] only (1 read per layer)
|
| 996 |
-
|
| 997 |
-
Each tensor is loaded, its column means computed, and freed before
|
| 998 |
-
the next is loaded. Peak memory: one tensor in F32 at a time.
|
| 999 |
-
|
| 1000 |
-
Returns dict: perm_key β permutation array."""
|
| 1001 |
-
import gc
|
| 1002 |
-
print(" βββ Fold-Basis pre-pass: computing dimension permutations βββ")
|
| 1003 |
-
|
| 1004 |
-
perms = {}
|
| 1005 |
-
hidden_stat = None
|
| 1006 |
-
hidden_dim = None
|
| 1007 |
-
hidden_count = 0
|
| 1008 |
-
|
| 1009 |
-
# ββ Pass 1: P_hidden from first 3 layers' W_q ββ
|
| 1010 |
-
_HIDDEN_SAMPLE = 3
|
| 1011 |
-
for ti in tensor_infos:
|
| 1012 |
-
if hidden_count >= _HIDDEN_SAMPLE:
|
| 1013 |
-
break
|
| 1014 |
-
if not _re.match(r'blk\.\d+\.attn_q\.weight', ti['name']):
|
| 1015 |
-
continue
|
| 1016 |
-
if ti['n_dims'] < 2:
|
| 1017 |
-
continue
|
| 1018 |
-
d0 = int(ti['dims'][0])
|
| 1019 |
-
if d0 % QK_K != 0:
|
| 1020 |
-
continue
|
| 1021 |
-
|
| 1022 |
-
abs_offset = data_section_start + ti['offset']
|
| 1023 |
-
fin.seek(abs_offset)
|
| 1024 |
-
raw = fin.read(ti['data_size'])
|
| 1025 |
-
if ti['type'] == GGML_TYPE_BF16:
|
| 1026 |
-
f32 = bf16_to_f32(raw, ti['n_elements'])
|
| 1027 |
-
elif ti['type'] == GGML_TYPE_F16:
|
| 1028 |
-
f32 = f16_to_f32(raw, ti['n_elements'])
|
| 1029 |
-
elif ti['type'] == GGML_TYPE_F32:
|
| 1030 |
-
f32 = np.frombuffer(raw, dtype=np.float32)
|
| 1031 |
-
else:
|
| 1032 |
-
continue
|
| 1033 |
-
|
| 1034 |
-
d1 = int(ti['dims'][1]) if ti['n_dims'] > 1 else 1
|
| 1035 |
-
M = f32.reshape(d1, d0)
|
| 1036 |
-
col_mean = np.mean(M.astype(np.float64), axis=0)
|
| 1037 |
-
del f32, M, raw
|
| 1038 |
-
gc.collect()
|
| 1039 |
-
|
| 1040 |
-
if hidden_stat is None:
|
| 1041 |
-
hidden_dim = d0
|
| 1042 |
-
hidden_stat = col_mean
|
| 1043 |
-
else:
|
| 1044 |
-
hidden_stat += col_mean
|
| 1045 |
-
hidden_count += 1
|
| 1046 |
-
print(f" β P_hidden sample: {ti['name']}")
|
| 1047 |
-
|
| 1048 |
-
if hidden_stat is not None:
|
| 1049 |
-
perms['hidden'] = _fb_compute_perm(hidden_stat)
|
| 1050 |
-
print(f" β P_hidden: dim={hidden_dim}, from {hidden_count} W_q tensors")
|
| 1051 |
-
|
| 1052 |
-
# ββ Pass 2: P_inter from each layer's W_down ββ
|
| 1053 |
-
n_inter = 0
|
| 1054 |
-
for ti in tensor_infos:
|
| 1055 |
-
m = _re.match(r'blk\.(\d+)\.ffn_down\.weight', ti['name'])
|
| 1056 |
-
if not m:
|
| 1057 |
-
continue
|
| 1058 |
-
if ti['n_dims'] < 2:
|
| 1059 |
-
continue
|
| 1060 |
-
d0 = int(ti['dims'][0])
|
| 1061 |
-
if d0 % QK_K != 0:
|
| 1062 |
-
continue
|
| 1063 |
-
|
| 1064 |
-
layer = m.group(1)
|
| 1065 |
-
abs_offset = data_section_start + ti['offset']
|
| 1066 |
-
fin.seek(abs_offset)
|
| 1067 |
-
raw = fin.read(ti['data_size'])
|
| 1068 |
-
if ti['type'] == GGML_TYPE_BF16:
|
| 1069 |
-
f32 = bf16_to_f32(raw, ti['n_elements'])
|
| 1070 |
-
elif ti['type'] == GGML_TYPE_F16:
|
| 1071 |
-
f32 = f16_to_f32(raw, ti['n_elements'])
|
| 1072 |
-
elif ti['type'] == GGML_TYPE_F32:
|
| 1073 |
-
f32 = np.frombuffer(raw, dtype=np.float32)
|
| 1074 |
-
else:
|
| 1075 |
-
continue
|
| 1076 |
-
|
| 1077 |
-
d1 = int(ti['dims'][1]) if ti['n_dims'] > 1 else 1
|
| 1078 |
-
M = f32.reshape(d1, d0)
|
| 1079 |
-
col_mean = np.mean(M.astype(np.float64), axis=0)
|
| 1080 |
-
del f32, M, raw
|
| 1081 |
-
gc.collect()
|
| 1082 |
-
|
| 1083 |
-
perms[f'inter_{layer}'] = _fb_compute_perm(col_mean)
|
| 1084 |
-
n_inter += 1
|
| 1085 |
-
|
| 1086 |
-
if n_inter > 0:
|
| 1087 |
-
print(f" β P_inter: {n_inter} layers (from W_down column means)")
|
| 1088 |
|
| 1089 |
-
|
| 1090 |
-
|
| 1091 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1092 |
|
| 1093 |
|
| 1094 |
def is_attention_tensor(name):
|
|
@@ -1232,7 +894,7 @@ def main():
|
|
| 1232 |
if len(sys.argv) < 3:
|
| 1233 |
print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
|
| 1234 |
" [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
|
| 1235 |
-
" [--fold-interleave]
|
| 1236 |
sys.exit(1)
|
| 1237 |
|
| 1238 |
input_path = sys.argv[1]
|
|
@@ -1242,9 +904,6 @@ def main():
|
|
| 1242 |
q2all = '--q2all' in sys.argv
|
| 1243 |
keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
|
| 1244 |
fold_interleave = '--fold-interleave' in sys.argv # fold-optimal weight interleave for DC/vesica cancellation
|
| 1245 |
-
fold_basis = '--fold-basis' in sys.argv # whole-dimension fold-optimal basis change (baked into weights)
|
| 1246 |
-
fold_hadamard = '--fold-hadamard' in sys.argv # per-block Walsh-Hadamard transform (baked, zero storage)
|
| 1247 |
-
fold_sort_refit = '--fold-sort-refit' in sys.argv # sortβHPCβunsortβrefit (baked, stock llama compatible)
|
| 1248 |
|
| 1249 |
# Check for imatrix
|
| 1250 |
imatrix_data = None
|
|
@@ -1275,12 +934,6 @@ def main():
|
|
| 1275 |
print(" β Engine: Python (numpy vectorized) β")
|
| 1276 |
if fold_interleave:
|
| 1277 |
print(" β Fold Interleave: ON (sort-pair DC/vesica cancellation) β")
|
| 1278 |
-
if fold_basis:
|
| 1279 |
-
print(" β Fold Basis: ON (whole-dim permutation, baked in) β")
|
| 1280 |
-
if fold_hadamard:
|
| 1281 |
-
print(" β Fold Hadamard: ON (per-block WHT, zero storage) β")
|
| 1282 |
-
if fold_sort_refit:
|
| 1283 |
-
print(" β Fold Sort-Refit: ON (sortβHPCβunsortβrefit, stock llama) β")
|
| 1284 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 1285 |
print()
|
| 1286 |
|
|
@@ -1412,16 +1065,6 @@ def main():
|
|
| 1412 |
print(f" Tensors to keep as-is: {total_keep}")
|
| 1413 |
print()
|
| 1414 |
|
| 1415 |
-
# ββ Fold-Basis pre-pass ββ
|
| 1416 |
-
# Compute whole-dimension permutations from weight statistics.
|
| 1417 |
-
# This reads relevant tensors from the input to gather column
|
| 1418 |
-
# energies, then computes the fold-optimal permutation for each
|
| 1419 |
-
# shared dimension. The permutations are applied during the
|
| 1420 |
-
# main quantization loop below.
|
| 1421 |
-
fb_perms = {}
|
| 1422 |
-
if fold_basis:
|
| 1423 |
-
fb_perms = _fb_prepass(fin, tensor_infos, data_section_start)
|
| 1424 |
-
|
| 1425 |
# ββ Compute output tensor sizes and offsets ββ
|
| 1426 |
out_tensor_infos = []
|
| 1427 |
out_data_offset = 0
|
|
@@ -1566,24 +1209,22 @@ def main():
|
|
| 1566 |
else:
|
| 1567 |
updated_kv.append((key, vtype, raw_value))
|
| 1568 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1569 |
# ββ Write output GGUF ββ
|
| 1570 |
print(" Writing output GGUF...")
|
| 1571 |
-
|
| 1572 |
-
# ββ Inject fold-hadamard metadata ββ
|
| 1573 |
-
n_kv_out = n_kv
|
| 1574 |
-
if fold_hadamard:
|
| 1575 |
-
# GGUF KV type 8 = BOOL (stored as int8: 1=true)
|
| 1576 |
-
updated_kv.append(('hexstate.fold_hadamard', 8,
|
| 1577 |
-
struct.pack('<b', 1)))
|
| 1578 |
-
n_kv_out += 1
|
| 1579 |
-
print(" Added hexstate.fold_hadamard=true to GGUF metadata")
|
| 1580 |
-
|
| 1581 |
with open(output_path, 'wb') as fout:
|
| 1582 |
# Header
|
| 1583 |
fout.write(struct.pack('<I', GGUF_MAGIC))
|
| 1584 |
fout.write(struct.pack('<I', GGUF_VERSION))
|
| 1585 |
fout.write(struct.pack('<Q', n_tensors))
|
| 1586 |
-
fout.write(struct.pack('<Q',
|
| 1587 |
|
| 1588 |
# KV pairs (passthrough)
|
| 1589 |
for key, vtype, raw_value in updated_kv:
|
|
@@ -1614,9 +1255,6 @@ def main():
|
|
| 1614 |
q2k_rmse_sum = 0.0
|
| 1615 |
q2k_tensor_count = 0
|
| 1616 |
|
| 1617 |
-
# Fold-interleave sidecar: tensor_name β int16 permutation indices
|
| 1618 |
-
fold_perm_sidecar = {} if fold_interleave else None
|
| 1619 |
-
|
| 1620 |
for i, ti in enumerate(tensor_infos):
|
| 1621 |
# Progress bar
|
| 1622 |
pct = (i + 1) / n_tensors * 100
|
|
@@ -1651,12 +1289,6 @@ def main():
|
|
| 1651 |
n_el = ti['n_elements']
|
| 1652 |
n_blocks_q8 = n_el // 32
|
| 1653 |
|
| 1654 |
-
# ββ Fold-Basis ββ
|
| 1655 |
-
if fold_basis and fb_perms:
|
| 1656 |
-
for perm_key, axis in _fb_tensor_perms(ti['name']):
|
| 1657 |
-
if perm_key in fb_perms:
|
| 1658 |
-
f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
|
| 1659 |
-
|
| 1660 |
if use_hpc and hasattr(_HEXSTATE_LIB, 'hexstate_quantize_tensor_q8_0_hpc'):
|
| 1661 |
output_buf = np.zeros(n_blocks_q8 * 34, dtype=np.uint8)
|
| 1662 |
error = ctypes.c_float(0.0)
|
|
@@ -1702,13 +1334,6 @@ def main():
|
|
| 1702 |
continue
|
| 1703 |
|
| 1704 |
n_el = len(f32)
|
| 1705 |
-
|
| 1706 |
-
# ββ Fold-Basis ββ
|
| 1707 |
-
if fold_basis and fb_perms:
|
| 1708 |
-
for perm_key, axis in _fb_tensor_perms(ti['name']):
|
| 1709 |
-
if perm_key in fb_perms:
|
| 1710 |
-
f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
|
| 1711 |
-
|
| 1712 |
if n_el % 32:
|
| 1713 |
raise ValueError(f"Q4_0 tensor {ti['name']} is not row/block aligned")
|
| 1714 |
n_blocks_q4 = n_el // 32
|
|
@@ -1772,46 +1397,15 @@ def main():
|
|
| 1772 |
fout.write(b'\x00' * pad)
|
| 1773 |
continue
|
| 1774 |
|
| 1775 |
-
# ββ Fold-
|
| 1776 |
-
# Apply whole-dimension permutations computed in the
|
| 1777 |
-
# pre-pass. These are baked into the weight matrix β
|
| 1778 |
-
# the output GGUF is self-contained, no sidecar.
|
| 1779 |
-
if fold_basis and fb_perms:
|
| 1780 |
-
for perm_key, axis in _fb_tensor_perms(ti['name']):
|
| 1781 |
-
if perm_key in fb_perms:
|
| 1782 |
-
f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
|
| 1783 |
-
|
| 1784 |
-
# ββ Fold-Hadamard pre-processing ββββββββββββββββββββ
|
| 1785 |
-
# Per-block Walsh-Hadamard transform. Decorrelates
|
| 1786 |
-
# weights so quantization errors are independent β
|
| 1787 |
-
# DC and vesica go to zero structurally. No storage
|
| 1788 |
-
# overhead β H is defined by block size alone.
|
| 1789 |
-
# The dequant kernel applies Hβ»ΒΉ = H/n after unpacking.
|
| 1790 |
-
if fold_hadamard:
|
| 1791 |
-
apply_fold_hadamard(f32)
|
| 1792 |
-
|
| 1793 |
-
# ββ Fold-Sort-Refit pre-processing ββββββββββββββββ
|
| 1794 |
-
# Sort each block β quantize sorted with HPC β unsort
|
| 1795 |
-
# codes β refit sub-block scales. Produces standard
|
| 1796 |
-
# Q2_K blocks compatible with stock llama.cpp.
|
| 1797 |
-
sort_perms_for_refit = None
|
| 1798 |
-
f32_presort = None
|
| 1799 |
-
if fold_sort_refit:
|
| 1800 |
-
n_blk_sr = len(f32) // QK_K
|
| 1801 |
-
f32_presort = f32.reshape(n_blk_sr, QK_K).copy()
|
| 1802 |
-
sort_perms_for_refit = np.argsort(f32_presort, axis=1)
|
| 1803 |
-
f32 = np.take_along_axis(f32_presort, sort_perms_for_refit, axis=1).reshape(-1).astype(np.float32)
|
| 1804 |
-
|
| 1805 |
-
# ββ Fold-Interleave pre-processing ββββββββββββββββββ
|
| 1806 |
# Sort-pair weights within each QK_K block so that the
|
| 1807 |
# k-th smallest and k-th largest are at fold-complementary
|
| 1808 |
# positions (k, k+128). The HPC pipeline then sees blocks
|
| 1809 |
# whose vesica/DC structure is pre-aligned for cancellation.
|
| 1810 |
-
# f32_orig
|
| 1811 |
-
|
| 1812 |
-
|
| 1813 |
if fold_interleave:
|
| 1814 |
-
f32_orig = f32.copy()
|
| 1815 |
f32, fold_perm_flat = apply_fold_interleave(f32)
|
| 1816 |
|
| 1817 |
# Quantize to Q2_K β always use HPC with chunked processing
|
|
@@ -1823,7 +1417,7 @@ def main():
|
|
| 1823 |
imat_full = _expand_imatrix(ti, imatrix_data)
|
| 1824 |
|
| 1825 |
# If fold-interleave is active, permute the imatrix the same way
|
| 1826 |
-
if fold_interleave and imat_full is not None
|
| 1827 |
imat_full = apply_fold_interleave_importance(imat_full, fold_perm_flat)
|
| 1828 |
|
| 1829 |
n_el = ti['n_elements']
|
|
@@ -1869,25 +1463,10 @@ def main():
|
|
| 1869 |
n_blocks = n_el // QK_K
|
| 1870 |
fout.write(q2k_data)
|
| 1871 |
|
| 1872 |
-
# ββ Sort-Refit post-processing ββββββββββββββββββββ
|
| 1873 |
-
# Unsort the HPC-optimized codes back to original
|
| 1874 |
-
# weight positions and refit sub-block scales.
|
| 1875 |
-
if fold_sort_refit and sort_perms_for_refit is not None:
|
| 1876 |
-
# Seek back β we need to overwrite what we just wrote
|
| 1877 |
-
fout.seek(fout.tell() - len(q2k_data))
|
| 1878 |
-
q2k_data = _sort_refit_q2k(
|
| 1879 |
-
f32_presort, q2k_data,
|
| 1880 |
-
sort_perms_for_refit, n_blocks)
|
| 1881 |
-
fout.write(q2k_data)
|
| 1882 |
-
|
| 1883 |
-
# ββ Store fold-interleave permutation in sidecar ββ
|
| 1884 |
-
if fold_interleave and fold_perm_flat is not None:
|
| 1885 |
-
fold_perm_sidecar[ti['name']] = fold_perm_flat
|
| 1886 |
-
|
| 1887 |
# ββ Compute and report exact per-tensor RMSE ββ
|
| 1888 |
-
#
|
| 1889 |
-
# in permuted order
|
| 1890 |
-
#
|
| 1891 |
try:
|
| 1892 |
CHUNK_BLK = 100_000 # blocks per chunk to bound memory
|
| 1893 |
total_se = 0.0
|
|
@@ -1896,10 +1475,7 @@ def main():
|
|
| 1896 |
ce = min(ci + CHUNK_BLK, n_blocks)
|
| 1897 |
chunk_q = q2k_data[ci*84:ce*84]
|
| 1898 |
deq_chunk = dequant_q2k_fast(chunk_q, ce - ci)
|
| 1899 |
-
|
| 1900 |
-
perm_slice = fold_perm_flat[ci*QK_K:ce*QK_K]
|
| 1901 |
-
deq_chunk = invert_fold_interleave(deq_chunk, perm_slice)
|
| 1902 |
-
orig_chunk = f32_orig[ci*QK_K:ce*QK_K]
|
| 1903 |
n_valid = min(len(orig_chunk), len(deq_chunk))
|
| 1904 |
diff = orig_chunk[:n_valid] - deq_chunk[:n_valid]
|
| 1905 |
total_se += np.sum(diff ** 2)
|
|
@@ -1907,7 +1483,7 @@ def main():
|
|
| 1907 |
tensor_rmse = np.sqrt(total_se / max(total_n, 1))
|
| 1908 |
q2k_rmse_sum += tensor_rmse
|
| 1909 |
q2k_tensor_count += 1
|
| 1910 |
-
fi_tag = "Β·
|
| 1911 |
print(f"\n [Q2_K{fi_tag}] {ti['name'][:52]} RMSE={tensor_rmse:.6e}")
|
| 1912 |
except Exception as e:
|
| 1913 |
print(f"\n [Q2_K] {ti['name'][:55]} RMSE=err({e})")
|
|
@@ -1915,31 +1491,7 @@ def main():
|
|
| 1915 |
quant_count += 1
|
| 1916 |
total_quant_bytes += len(q2k_data)
|
| 1917 |
else:
|
| 1918 |
-
# Keep as-is (passthrough)
|
| 1919 |
-
# 1D norm weights so the residual stream is consistent
|
| 1920 |
-
if fold_basis and fb_perms:
|
| 1921 |
-
tp = _fb_tensor_perms(ti['name'])
|
| 1922 |
-
if tp:
|
| 1923 |
-
if ti['type'] == GGML_TYPE_BF16:
|
| 1924 |
-
f32 = bf16_to_f32(raw_data, ti['n_elements'])
|
| 1925 |
-
elif ti['type'] == GGML_TYPE_F16:
|
| 1926 |
-
f32 = f16_to_f32(raw_data, ti['n_elements'])
|
| 1927 |
-
elif ti['type'] == GGML_TYPE_F32:
|
| 1928 |
-
f32 = np.frombuffer(raw_data, dtype=np.float32).copy()
|
| 1929 |
-
else:
|
| 1930 |
-
f32 = None
|
| 1931 |
-
if f32 is not None:
|
| 1932 |
-
for perm_key, axis in tp:
|
| 1933 |
-
if perm_key in fb_perms:
|
| 1934 |
-
f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
|
| 1935 |
-
# Write permuted data in original type
|
| 1936 |
-
if ti['type'] == GGML_TYPE_BF16:
|
| 1937 |
-
raw_data = f32.astype(np.float32).view(np.uint32)
|
| 1938 |
-
raw_data = (((raw_data >> 16) & 0xFFFF).astype(np.uint16)).tobytes()
|
| 1939 |
-
elif ti['type'] == GGML_TYPE_F16:
|
| 1940 |
-
raw_data = f32.astype(np.float16).tobytes()
|
| 1941 |
-
elif ti['type'] == GGML_TYPE_F32:
|
| 1942 |
-
raw_data = f32.astype(np.float32).tobytes()
|
| 1943 |
fout.write(raw_data)
|
| 1944 |
total_keep_bytes += len(raw_data)
|
| 1945 |
|
|
@@ -1978,20 +1530,15 @@ def main():
|
|
| 1978 |
print()
|
| 1979 |
print(f" Output: {output_path}")
|
| 1980 |
|
| 1981 |
-
|
| 1982 |
-
|
| 1983 |
-
sidecar_path = output_path + '.fold_perm.npz'
|
| 1984 |
-
np.savez_compressed(sidecar_path, **fold_perm_sidecar)
|
| 1985 |
-
sidecar_size = os.path.getsize(sidecar_path)
|
| 1986 |
-
n_fi_tensors = len(fold_perm_sidecar)
|
| 1987 |
-
print(f" Fold-interleave sidecar: {sidecar_path}")
|
| 1988 |
-
print(f" {n_fi_tensors} tensors, {sidecar_size/1024:.0f} KB compressed")
|
| 1989 |
print()
|
| 1990 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 1991 |
-
print(" β
|
| 1992 |
-
print(" β
|
| 1993 |
-
print(" β
|
| 1994 |
-
print(" β
|
|
|
|
| 1995 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 1996 |
print()
|
| 1997 |
|
|
|
|
| 659 |
# across all fold pairs, so the spectral objective's vesica/DC terms
|
| 660 |
# can be driven to zero by the unchanged HPC pipeline.
|
| 661 |
#
|
| 662 |
+
# The permutation is a pre-processing step β the quantization algorithm
|
| 663 |
+
# sees 256 floats and runs identically. A sidecar .npz stores the
|
| 664 |
+
# per-block permutation indices; the inference engine applies the
|
| 665 |
+
# inverse permutation after dequantization (or equivalently permutes
|
| 666 |
+
# the activation vector).
|
| 667 |
+
|
| 668 |
+
def compute_fold_permutation(block_weights):
|
| 669 |
+
"""Sort-interleave permutation for one QK_K block.
|
| 670 |
+
|
| 671 |
+
Returns perm such that permuted[k] = block[perm[k]], with
|
| 672 |
+
perm[0..127] = argsort ascending (smallest half)
|
| 673 |
+
perm[128..255] = argsort descending (largest half, reversed)
|
| 674 |
+
|
| 675 |
+
Every fold pair (k, k+128) therefore contains the k-th smallest
|
| 676 |
+
and the k-th largest weight, making their sum approximately
|
| 677 |
+
constant across all 128 pairs.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 678 |
"""
|
| 679 |
+
n = len(block_weights)
|
| 680 |
+
half = n // 2
|
| 681 |
+
s = np.argsort(block_weights)
|
| 682 |
+
perm = np.empty(n, dtype=np.intp)
|
| 683 |
+
perm[:half] = s[:half]
|
| 684 |
+
perm[half:] = s[-1:half-1:-1]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 685 |
return perm
|
| 686 |
|
| 687 |
|
| 688 |
+
def apply_fold_interleave(f32_data, block_size=QK_K, chunk_blocks=32768):
|
| 689 |
+
"""Apply fold-optimal permutation to every block in a flat tensor.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 690 |
|
| 691 |
+
Returns (permuted_data, perm_indices) where perm_indices is a flat
|
| 692 |
+
int16 array of the same length β block b's permutation lives at
|
| 693 |
+
[b*block_size : (b+1)*block_size]. int16 suffices since indices
|
| 694 |
+
are 0..255.
|
| 695 |
|
| 696 |
+
Memory-optimized: chunks the vectorised argsort (int64 8 bytes/idx)
|
| 697 |
+
to avoid a 400 MB full-tensor allocation on 50M-element tensors
|
| 698 |
+
(Qwen 27B attn_qkv), and permutes in-place so peak is ~1.7x
|
| 699 |
+
less. Squared error is permutation-invariant, so RMSE needs no
|
| 700 |
+
inverse permutation.
|
| 701 |
+
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 702 |
n = len(f32_data)
|
| 703 |
n_blocks = n // block_size
|
| 704 |
+
# Ensure contiguous + writeable for in-place chunk updates
|
| 705 |
+
if not f32_data.flags.writeable or not f32_data.flags.c_contiguous:
|
| 706 |
+
f32_data = np.ascontiguousarray(f32_data)
|
| 707 |
+
perm_flat = np.empty(n, dtype=np.int16)
|
| 708 |
+
blocks = f32_data.reshape(n_blocks, block_size)
|
| 709 |
+
perm_blocks = perm_flat.reshape(n_blocks, block_size)
|
| 710 |
+
half = block_size // 2
|
| 711 |
+
for start in range(0, n_blocks, chunk_blocks):
|
| 712 |
+
end = min(start + chunk_blocks, n_blocks)
|
| 713 |
+
chunk = blocks[start:end] # view into f32_data
|
| 714 |
+
sorted_idx = np.argsort(chunk, axis=1) # chunk-sized int64 only
|
| 715 |
+
perm_chunk = np.empty((end - start, block_size), dtype=np.intp)
|
| 716 |
+
perm_chunk[:, :half] = sorted_idx[:, :half]
|
| 717 |
+
perm_chunk[:, half:] = sorted_idx[:, -1:half-1:-1]
|
| 718 |
+
tmp = np.take_along_axis(chunk, perm_chunk, axis=1)
|
| 719 |
+
chunk[:, :] = tmp
|
| 720 |
+
perm_blocks[start:end] = perm_chunk.astype(np.int16, copy=False)
|
| 721 |
+
return f32_data, perm_flat
|
| 722 |
+
|
| 723 |
+
|
| 724 |
+
def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K, chunk_blocks=32768):
|
| 725 |
+
"""Permute an importance vector with the same fold permutation.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 726 |
|
| 727 |
+
Chunked + in-place: avoids a full-tensor int64 perm copy
|
| 728 |
+
(400 MB at 50M elements) and a duplicate float32 buffer.
|
| 729 |
+
"""
|
| 730 |
+
n = len(importance)
|
| 731 |
+
n_blocks = n // block_size
|
| 732 |
+
if not importance.flags.writeable or not importance.flags.c_contiguous:
|
| 733 |
+
importance = np.ascontiguousarray(importance)
|
| 734 |
+
blocks = importance.reshape(n_blocks, block_size)
|
| 735 |
+
perm_blocks = perm_flat.reshape(n_blocks, block_size)
|
| 736 |
+
for start in range(0, n_blocks, chunk_blocks):
|
| 737 |
+
end = min(start + chunk_blocks, n_blocks)
|
| 738 |
+
chunk = blocks[start:end]
|
| 739 |
+
perm_chunk = perm_blocks[start:end].astype(np.intp, copy=False)
|
| 740 |
+
tmp = np.take_along_axis(chunk, perm_chunk, axis=1)
|
| 741 |
+
chunk[:, :] = tmp
|
| 742 |
+
return importance
|
| 743 |
+
|
| 744 |
+
|
| 745 |
+
def invert_fold_interleave(data, perm_flat, block_size=QK_K):
|
| 746 |
+
"""Apply the inverse fold permutation to restore original order."""
|
| 747 |
+
n = len(data)
|
| 748 |
+
n_blocks = n // block_size
|
| 749 |
+
blocks = data.reshape(n_blocks, block_size)
|
| 750 |
+
perms = perm_flat.reshape(n_blocks, block_size).astype(np.intp)
|
| 751 |
+
inv_perms = np.argsort(perms, axis=1)
|
| 752 |
+
result = np.take_along_axis(blocks, inv_perms, axis=1)
|
| 753 |
+
return result.reshape(-1)
|
| 754 |
|
| 755 |
|
| 756 |
def is_attention_tensor(name):
|
|
|
|
| 894 |
if len(sys.argv) < 3:
|
| 895 |
print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
|
| 896 |
" [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
|
| 897 |
+
" [--fold-interleave]")
|
| 898 |
sys.exit(1)
|
| 899 |
|
| 900 |
input_path = sys.argv[1]
|
|
|
|
| 904 |
q2all = '--q2all' in sys.argv
|
| 905 |
keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
|
| 906 |
fold_interleave = '--fold-interleave' in sys.argv # fold-optimal weight interleave for DC/vesica cancellation
|
|
|
|
|
|
|
|
|
|
| 907 |
|
| 908 |
# Check for imatrix
|
| 909 |
imatrix_data = None
|
|
|
|
| 934 |
print(" β Engine: Python (numpy vectorized) β")
|
| 935 |
if fold_interleave:
|
| 936 |
print(" β Fold Interleave: ON (sort-pair DC/vesica cancellation) β")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 937 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 938 |
print()
|
| 939 |
|
|
|
|
| 1065 |
print(f" Tensors to keep as-is: {total_keep}")
|
| 1066 |
print()
|
| 1067 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1068 |
# ββ Compute output tensor sizes and offsets ββ
|
| 1069 |
out_tensor_infos = []
|
| 1070 |
out_data_offset = 0
|
|
|
|
| 1209 |
else:
|
| 1210 |
updated_kv.append((key, vtype, raw_value))
|
| 1211 |
|
| 1212 |
+
# ββ Pre-compute fold-interleave permutations for GGUF KV pairs ββ
|
| 1213 |
+
# Instead of per-tensor permutations (expensive to pre-compute),
|
| 1214 |
+
# just store a flag indicating fold-interleave was used.
|
| 1215 |
+
# FoldLlama will apply inverse permutation to all Q2_K tensors.
|
| 1216 |
+
if fold_interleave:
|
| 1217 |
+
updated_kv.append(("fold_interleave", 8, struct.pack('<Q', 4) + b"true"))
|
| 1218 |
+
|
| 1219 |
# ββ Write output GGUF ββ
|
| 1220 |
print(" Writing output GGUF...")
|
| 1221 |
+
n_kv = len(updated_kv)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1222 |
with open(output_path, 'wb') as fout:
|
| 1223 |
# Header
|
| 1224 |
fout.write(struct.pack('<I', GGUF_MAGIC))
|
| 1225 |
fout.write(struct.pack('<I', GGUF_VERSION))
|
| 1226 |
fout.write(struct.pack('<Q', n_tensors))
|
| 1227 |
+
fout.write(struct.pack('<Q', n_kv))
|
| 1228 |
|
| 1229 |
# KV pairs (passthrough)
|
| 1230 |
for key, vtype, raw_value in updated_kv:
|
|
|
|
| 1255 |
q2k_rmse_sum = 0.0
|
| 1256 |
q2k_tensor_count = 0
|
| 1257 |
|
|
|
|
|
|
|
|
|
|
| 1258 |
for i, ti in enumerate(tensor_infos):
|
| 1259 |
# Progress bar
|
| 1260 |
pct = (i + 1) / n_tensors * 100
|
|
|
|
| 1289 |
n_el = ti['n_elements']
|
| 1290 |
n_blocks_q8 = n_el // 32
|
| 1291 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1292 |
if use_hpc and hasattr(_HEXSTATE_LIB, 'hexstate_quantize_tensor_q8_0_hpc'):
|
| 1293 |
output_buf = np.zeros(n_blocks_q8 * 34, dtype=np.uint8)
|
| 1294 |
error = ctypes.c_float(0.0)
|
|
|
|
| 1334 |
continue
|
| 1335 |
|
| 1336 |
n_el = len(f32)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1337 |
if n_el % 32:
|
| 1338 |
raise ValueError(f"Q4_0 tensor {ti['name']} is not row/block aligned")
|
| 1339 |
n_blocks_q4 = n_el // 32
|
|
|
|
| 1397 |
fout.write(b'\x00' * pad)
|
| 1398 |
continue
|
| 1399 |
|
| 1400 |
+
# ββ Fold-Interleave pre-processing ββ
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1401 |
# Sort-pair weights within each QK_K block so that the
|
| 1402 |
# k-th smallest and k-th largest are at fold-complementary
|
| 1403 |
# positions (k, k+128). The HPC pipeline then sees blocks
|
| 1404 |
# whose vesica/DC structure is pre-aligned for cancellation.
|
| 1405 |
+
# No f32_orig copy β squared error is invariant under
|
| 1406 |
+
# permutation, so RMSE on permuted order is identical.
|
| 1407 |
+
# apply_fold_interleave permutes f32 in-place chunked.
|
| 1408 |
if fold_interleave:
|
|
|
|
| 1409 |
f32, fold_perm_flat = apply_fold_interleave(f32)
|
| 1410 |
|
| 1411 |
# Quantize to Q2_K β always use HPC with chunked processing
|
|
|
|
| 1417 |
imat_full = _expand_imatrix(ti, imatrix_data)
|
| 1418 |
|
| 1419 |
# If fold-interleave is active, permute the imatrix the same way
|
| 1420 |
+
if fold_interleave and imat_full is not None:
|
| 1421 |
imat_full = apply_fold_interleave_importance(imat_full, fold_perm_flat)
|
| 1422 |
|
| 1423 |
n_el = ti['n_elements']
|
|
|
|
| 1463 |
n_blocks = n_el // QK_K
|
| 1464 |
fout.write(q2k_data)
|
| 1465 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1466 |
# ββ Compute and report exact per-tensor RMSE ββ
|
| 1467 |
+
# RMSE is invariant under permutation: with fold-interleave
|
| 1468 |
+
# both f32 and deq_chunk are in permuted order, so direct
|
| 1469 |
+
# comparison is identical and avoids the inverse perm + copy.
|
| 1470 |
try:
|
| 1471 |
CHUNK_BLK = 100_000 # blocks per chunk to bound memory
|
| 1472 |
total_se = 0.0
|
|
|
|
| 1475 |
ce = min(ci + CHUNK_BLK, n_blocks)
|
| 1476 |
chunk_q = q2k_data[ci*84:ce*84]
|
| 1477 |
deq_chunk = dequant_q2k_fast(chunk_q, ce - ci)
|
| 1478 |
+
orig_chunk = f32[ci*QK_K:ce*QK_K]
|
|
|
|
|
|
|
|
|
|
| 1479 |
n_valid = min(len(orig_chunk), len(deq_chunk))
|
| 1480 |
diff = orig_chunk[:n_valid] - deq_chunk[:n_valid]
|
| 1481 |
total_se += np.sum(diff ** 2)
|
|
|
|
| 1483 |
tensor_rmse = np.sqrt(total_se / max(total_n, 1))
|
| 1484 |
q2k_rmse_sum += tensor_rmse
|
| 1485 |
q2k_tensor_count += 1
|
| 1486 |
+
fi_tag = "Β·FI" if fold_interleave else ""
|
| 1487 |
print(f"\n [Q2_K{fi_tag}] {ti['name'][:52]} RMSE={tensor_rmse:.6e}")
|
| 1488 |
except Exception as e:
|
| 1489 |
print(f"\n [Q2_K] {ti['name'][:55]} RMSE=err({e})")
|
|
|
|
| 1491 |
quant_count += 1
|
| 1492 |
total_quant_bytes += len(q2k_data)
|
| 1493 |
else:
|
| 1494 |
+
# Keep as-is (passthrough)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1495 |
fout.write(raw_data)
|
| 1496 |
total_keep_bytes += len(raw_data)
|
| 1497 |
|
|
|
|
| 1530 |
print()
|
| 1531 |
print(f" Output: {output_path}")
|
| 1532 |
|
| 1533 |
+
if fold_interleave:
|
| 1534 |
+
print(f" Fold-interleave: ON β permutations applied in-quad, no sidecar")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1535 |
print()
|
| 1536 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 1537 |
+
print(" β Fold-permuted weights are stored directly in the GGUF β")
|
| 1538 |
+
print(" β β no separate .npz sidecar needed. The fold permutation β")
|
| 1539 |
+
print(" β must be inverted after dequant (per QK_K block) by the β")
|
| 1540 |
+
print(" β inference engine, or equivalently permute the activation β")
|
| 1541 |
+
print(" β vector before the matmul. β")
|
| 1542 |
print(" ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ")
|
| 1543 |
print()
|
| 1544 |
|