CompressedGemma commited on
Commit
2ec3df7
Β·
verified Β·
1 Parent(s): 145817b

Fold prototype

Browse files
Files changed (1) hide show
  1. hexstate_requantize.py +111 -564
hexstate_requantize.py CHANGED
@@ -659,436 +659,98 @@ def dequant_q2k_fast(q2k_bytes, n_blocks):
659
  # across all fold pairs, so the spectral objective's vesica/DC terms
660
  # can be driven to zero by the unchanged HPC pipeline.
661
  #
662
- # Memory-efficient: permutes f32_data IN-PLACE in small chunks (no
663
- # full-tensor copy). The argsort runs on ~1 M elements at a time
664
- # instead of the whole tensor. RMSE is computed in permuted order
665
- # (SE is permutation-invariant, so the result is mathematically
666
- # identical). Sidecar perm indices are uint8 (indices 0..255).
667
-
668
- _FI_CHUNK_BLOCKS = 4096 # blocks per argsort chunk (~1 M elements)
669
-
670
- def apply_fold_interleave(f32_data, block_size=QK_K):
671
- """Sort-interleave every block IN-PLACE. Returns the uint8 perm
672
- index array (flat, same length as f32_data) for the sidecar.
673
- f32_data is modified in-place β€” no copy is made."""
674
- n = len(f32_data)
675
- n_blocks = n // block_size
676
- half = block_size // 2
677
- all_perms = np.empty(n, dtype=np.uint8)
678
-
679
- for cb in range(0, n_blocks, _FI_CHUNK_BLOCKS):
680
- ce = min(cb + _FI_CHUNK_BLOCKS, n_blocks)
681
- s = cb * block_size
682
- e = ce * block_size
683
- chunk = f32_data[s:e].reshape(ce - cb, block_size)
684
-
685
- sorted_idx = np.argsort(chunk, axis=1)
686
- perm = np.empty_like(sorted_idx)
687
- perm[:, :half] = sorted_idx[:, :half]
688
- perm[:, half:] = sorted_idx[:, -1:half-1:-1]
689
- del sorted_idx
690
-
691
- f32_data[s:e] = np.take_along_axis(chunk, perm, axis=1).reshape(-1)
692
- all_perms[s:e] = perm.astype(np.uint8).reshape(-1)
693
- del perm, chunk
694
-
695
- return all_perms
696
-
697
-
698
- def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K):
699
- """Permute an importance vector with the same fold permutation.
700
- Returns a new array (importance is typically much smaller than
701
- weights for tiled imatrix, so the copy is fine)."""
702
- n = len(importance)
703
- n_blocks = n // block_size
704
- blocks = importance.reshape(n_blocks, block_size)
705
- perms = perm_flat[:n].reshape(n_blocks, block_size).astype(np.intp)
706
- permuted = np.take_along_axis(blocks, perms, axis=1)
707
- return permuted.reshape(-1)
708
-
709
-
710
- def is_attention_tensor(name):
711
- """Detect attention Q/K/V/O projection tensors.
712
- These are the most sensitive to quantization and get promoted to Q4_0."""
713
-
714
- # ─── Fold-Basis Change ────────────────────────────────────────────────────
715
- # Whole-dimension permutations that propagate losslessly through the graph.
716
- # P_hidden: one global permutation of the residual-stream dimension.
717
- # Permutes columns of everything that reads from it (Q/K/V,
718
- # gate, up, embeddings) and rows of everything that writes
719
- # to it (O, down). Layer norms are 1D β€” just permute entries.
720
- # P_inter_N: one permutation per layer of the FFN intermediate dimension.
721
- # Permutes columns of W_down (the dimension Q2_K blocks span)
722
- # and rows of W_gate/W_up (output dim, no block effect).
723
- #
724
- # The permutations are chosen by stride-interleaving the columns sorted
725
- # by L2 energy: the k-th lowest-energy column is paired with the k-th
726
- # highest at fold-complementary positions (k, k+128) within each QK_K
727
- # block. This gives every block balanced fold pairs for vesica/DC
728
- # cancellation β€” the same spectral benefit as the per-block sort, but
729
- # baked into the weight matrix so the output GGUF is self-contained.
730
-
731
- import re as _re
732
-
733
- def _fb_tensor_perms(name):
734
- """Return list of (perm_key, axis) for this tensor.
735
- axis: 0 β†’ permute dims[0] (columns/input, what blocks span)
736
- 1 β†’ permute dims[1] (rows/output)
737
- '1d' β†’ 1D parameter (norms)
738
  """
739
- perms = []
740
- # Hidden dim as input (cols = dims[0])
741
- if _re.match(r'blk\.\d+\.attn_(q|k|v|qkv)\.weight', name) or \
742
- _re.match(r'blk\.\d+\.ffn_(gate|up)\.weight', name) or \
743
- name in ('token_embd.weight', 'output.weight'):
744
- perms.append(('hidden', 0))
745
- # Hidden dim as output (rows = dims[1])
746
- if _re.match(r'blk\.\d+\.attn_output\.weight', name) or \
747
- _re.match(r'blk\.\d+\.ffn_down\.weight', name):
748
- perms.append(('hidden', 1))
749
- # Hidden dim as 1D
750
- if _re.match(r'blk\.\d+\.(attn|ffn)_norm\.weight', name) or \
751
- name == 'output_norm.weight':
752
- perms.append(('hidden', '1d'))
753
- # Intermediate dim as input (cols = dims[0]) β†’ THIS is the big win
754
- m = _re.match(r'blk\.(\d+)\.ffn_down\.weight', name)
755
- if m:
756
- perms.append((f'inter_{m.group(1)}', 0))
757
- # Intermediate dim as output (rows = dims[1])
758
- m = _re.match(r'blk\.(\d+)\.ffn_(gate|up)\.weight', name)
759
- if m:
760
- perms.append((f'inter_{m.group(1)}', 1))
761
- return perms
762
-
763
-
764
- def _fb_compute_perm(col_stats, block_size=QK_K):
765
- """Per-block-group fold-optimal permutation.
766
-
767
- For each group of 256 consecutive columns, sorts by the column
768
- statistic and interleaves: k-th smallest at position k, k-th
769
- largest at position k+128. This IS the fold principle β€” each
770
- block group gets its own sort-interleave β€” while remaining a
771
- valid column permutation that propagates through the graph.
772
- Stock llama.cpp compatible: no modified dequant, no sidecar."""
773
- dim = len(col_stats)
774
- n_blocks = dim // block_size
775
- half = block_size // 2
776
- remainder = dim % block_size
777
-
778
- if n_blocks == 0:
779
- return np.arange(dim, dtype=np.intp)
780
-
781
- perm = np.arange(dim, dtype=np.intp) # identity for remainder
782
-
783
- for b in range(n_blocks):
784
- s = b * block_size
785
- e = s + block_size
786
- group_sorted = np.argsort(col_stats[s:e])
787
- perm[s:s + half] = s + group_sorted[:half]
788
- perm[s + half:e] = s + group_sorted[-1:half-1:-1]
789
-
790
  return perm
791
 
792
 
793
- def _fb_apply(f32, dims, perm, axis):
794
- """Apply a permutation to one axis of a tensor (flat f32 array).
795
- axis: 0 β†’ dims[0] columns, 1 β†’ dims[1] rows, '1d' β†’ 1D vector.
796
- Returns original f32 unchanged if perm length doesn't match the dim."""
797
- if axis == '1d':
798
- if len(perm) != len(f32):
799
- return f32
800
- return f32[perm].copy()
801
- d0 = int(dims[0])
802
- d1 = int(dims[1]) if len(dims) > 1 else 1
803
- # Dimension safety: skip if perm doesn't match the target axis
804
- if axis == 0 and len(perm) != d0:
805
- return f32
806
- if axis == 1 and len(perm) != d1:
807
- return f32
808
- M = f32.reshape(d1, d0)
809
- if axis == 0:
810
- M = M[:, perm]
811
- else:
812
- M = M[perm, :]
813
- return M.reshape(-1).copy()
814
 
 
 
 
 
815
 
816
- # ─── Fold-Hadamard ────────────────────────────────────────────────────────
817
- # Per-block Walsh-Hadamard transform before quantization.
818
- #
819
- # The WHT decorrelates weights within each QK_K block, making
820
- # quantization errors statistically independent. Independent errors
821
- # have zero expected DC and zero expected vesica β€” the spectral
822
- # objective is satisfied structurally, not by chasing it.
823
- #
824
- # Key properties:
825
- # H is its own inverse: HΒ·H = nΒ·I β†’ H⁻¹ = H/n
826
- # H preserves L2 norm: β€–Hxβ€–β‚‚ = √nΒ·β€–xβ€–β‚‚ (with unnormalised H)
827
- # H is deterministic: defined entirely by block size
828
- # Zero storage overhead: no permutation indices, no sidecar
829
- #
830
- # Baked into the GGUF: a metadata key signals "Hadamard Q2_K".
831
- # The inference dequant kernel applies the same transform after
832
- # unpacking: deq_orig = (1/n) Β· H Β· deq_packed
833
- #
834
- # This is QuIP#'s insight applied per-block instead of per-row,
835
- # composed with the full HPC sieve pipeline.
836
-
837
- def _wht_block(block):
838
- """In-place unnormalised Walsh-Hadamard transform of a power-of-2 array."""
839
- n = len(block)
840
- h = 1
841
- while h < n:
842
- for i in range(0, n, h * 2):
843
- for j in range(i, i + h):
844
- x = block[j]
845
- y = block[j + h]
846
- block[j] = x + y
847
- block[j + h] = x - y
848
- h *= 2
849
-
850
-
851
- def apply_fold_hadamard(f32_data, block_size=QK_K):
852
- """Apply normalised per-block WHT to a flat tensor IN-PLACE.
853
- Each block is transformed independently. O(n log n) per block."""
854
  n = len(f32_data)
855
  n_blocks = n // block_size
856
- scale = 1.0 / np.sqrt(block_size)
857
-
858
- # Vectorised butterfly: process all blocks in parallel
859
- blocks = f32_data.reshape(n_blocks, block_size).astype(np.float64)
860
- h = 1
861
- while h < block_size:
862
- # Reshape to expose butterfly pairs
863
- view = blocks.reshape(n_blocks, -1, 2 * h)
864
- left = view[:, :, :h].copy()
865
- right = view[:, :, h:].copy()
866
- view[:, :, :h] = left + right
867
- view[:, :, h:] = left - right
868
- h *= 2
869
-
870
- blocks *= scale
871
- f32_data[:] = blocks.astype(np.float32).reshape(-1)
872
-
873
-
874
- # ─── Sort-Quantize-Unsort-Refit ───────────────────────────────────────────
875
- # The full fold principle baked into standard Q2_K:
876
- #
877
- # 1. Sort each block by value (the fold-interleave)
878
- # 2. Quantize sorted blocks with full HPC pipeline
879
- # (sieve, Viterbi, greedy descent, Floyd-Steinberg β€” everything)
880
- # 3. Unsort the 2-bit codes back to original positions
881
- # 4. Refit sub-block scales (Ls, Lm) via least-squares
882
- # 5. Repack as standard Q2_K β€” stock llama.cpp compatible
883
- #
884
- # The HPC pipeline sees sorted blocks with narrow sub-block ranges
885
- # and balanced fold pairs. The code assignments it produces are
886
- # globally optimal in sorted space. After unsorting, the codes
887
- # retain their relative quality β€” element k still gets the code
888
- # that the HPC pipeline chose for it. Only the sub-block scales
889
- # need refitting because elements moved across sub-block boundaries.
890
- #
891
- # Combined with fold-basis (column permutation), the sort-refit
892
- # captures both coarse structure (consistent across rows) and fine
893
- # per-block structure (row-specific), all in standard Q2_K format.
894
-
895
- # Sub-block index mapping for Q2_K:
896
- # Element i maps to sub-block j as follows:
897
- # half = i // 128
898
- # sub = (i % 128) // 32
899
- # k = (i % 128) % 32
900
- # j = half * 8 + sub * 2 + (k >= 16)
901
- # Sub-block j covers 16 elements starting at:
902
- # base = (j//8)*128 + ((j%8)//2)*32 + 16*(j%2)
903
-
904
- _Q2K_SUB_BASES = np.array([
905
- (j // 8) * 128 + ((j % 8) // 2) * 32 + 16 * (j % 2)
906
- for j in range(16)
907
- ], dtype=np.int32)
908
-
909
-
910
- def _sort_refit_q2k(f32_orig, q2k_sorted_bytes, sort_perms, n_blocks):
911
- """Unsort HPC-quantized codes and refit sub-block scales.
912
-
913
- f32_orig: [n_blocks, 256] original (unsorted) weight values
914
- q2k_sorted_bytes: packed Q2_K data from HPC quantization of sorted blocks
915
- sort_perms: [n_blocks, 256] argsort permutations used to sort
916
- n_blocks: number of Q2_K blocks
917
-
918
- Returns repacked Q2_K bytes with unsorted codes and refit scales.
919
- Fully vectorized over blocks β€” only loops over 16 sub-blocks."""
920
-
921
- BLOCK_BYTES = 84
922
-
923
- q2k_arr = np.frombuffer(q2k_sorted_bytes, dtype=np.uint8).reshape(n_blocks, BLOCK_BYTES).copy()
924
-
925
- # ── Extract d, dmin ──
926
- d_arr = np.frombuffer(q2k_arr[:, 0:2].tobytes(),
927
- dtype=np.float16).reshape(n_blocks).astype(np.float32)
928
- dmin_arr = np.frombuffer(q2k_arr[:, 2:4].tobytes(),
929
- dtype=np.float16).reshape(n_blocks).astype(np.float32)
930
-
931
- # ── Unpack 2-bit codes: [n_blocks, 256] ──
932
- qs_raw = q2k_arr[:, 20:84] # [n_blocks, 64]
933
- q_sorted = np.zeros((n_blocks, 256), dtype=np.int32)
934
- for h in range(2):
935
- qs_half = qs_raw[:, h * 32:(h + 1) * 32]
936
- for s in range(4):
937
- q_sorted[:, h * 128 + s * 32: h * 128 + (s + 1) * 32] = (qs_half >> (s * 2)) & 3
938
-
939
- # ── Unsort codes ──
940
- inv_perms = np.argsort(sort_perms, axis=1)
941
- q_orig = np.take_along_axis(q_sorted, inv_perms, axis=1) # [n_blocks, 256]
942
-
943
- # ── Refit sub-block scales via OLS ──
944
- Ls_new = np.zeros((n_blocks, 16), dtype=np.int32)
945
- Lm_new = np.zeros((n_blocks, 16), dtype=np.int32)
946
-
947
- for j in range(16):
948
- base = int(_Q2K_SUB_BASES[j])
949
- x_sub = f32_orig[:, base:base + 16].astype(np.float64)
950
- q_sub = q_orig[:, base:base + 16].astype(np.float64)
951
-
952
- Sq = np.sum(q_sub, axis=1)
953
- Sqq = np.sum(q_sub ** 2, axis=1)
954
- Sx = np.sum(x_sub, axis=1)
955
- Sqx = np.sum(q_sub * x_sub, axis=1)
956
-
957
- n_el = 16.0
958
- denom = n_el * Sqq - Sq * Sq
959
- safe_denom = np.where(np.abs(denom) > 1e-10, denom, 1.0)
960
-
961
- alpha = (n_el * Sqx - Sq * Sx) / safe_denom
962
- beta = (Sx - alpha * Sq) / n_el
963
-
964
- safe_d = np.where(d_arr > 1e-20, d_arr, 1.0)
965
- safe_dmin = np.where(dmin_arr > 1e-20, dmin_arr, 1.0)
966
-
967
- Ls_f = alpha / safe_d
968
- Lm_f = -beta / safe_dmin
969
-
970
- valid = (np.abs(denom) > 1e-10) & (d_arr > 1e-20) & (dmin_arr > 1e-20)
971
- Ls_new[:, j] = np.where(valid, np.clip(np.round(Ls_f), 0, 15), 0).astype(np.int32)
972
- Lm_new[:, j] = np.where(valid, np.clip(np.round(Lm_f), 0, 15), 0).astype(np.int32)
973
-
974
- # ── Repack scales ──
975
- for j in range(16):
976
- q2k_arr[:, 4 + j] = (Ls_new[:, j].astype(np.uint8) & 0x0F) | \
977
- ((Lm_new[:, j].astype(np.uint8) & 0x0F) << 4)
978
-
979
- # ── Repack qs with unsorted codes ──
980
- new_qs = np.zeros((n_blocks, 64), dtype=np.uint8)
981
- for h in range(2):
982
- for s in range(4):
983
- codes = q_orig[:, h * 128 + s * 32: h * 128 + (s + 1) * 32].astype(np.uint8)
984
- new_qs[:, h * 32:(h + 1) * 32] |= codes << (s * 2)
985
- q2k_arr[:, 20:84] = new_qs
986
-
987
- return q2k_arr.tobytes()
988
-
989
-
990
- def _fb_prepass(fin, tensor_infos, data_section_start):
991
- """Pre-pass: compute fold-basis permutations from weight statistics.
992
-
993
- Memory-lean: reads only a small sample of tensors.
994
- - P_hidden: from the first 3 layers' W_q only (3 tensor reads)
995
- - P_inter_N: from W_down[N] only (1 read per layer)
996
-
997
- Each tensor is loaded, its column means computed, and freed before
998
- the next is loaded. Peak memory: one tensor in F32 at a time.
999
-
1000
- Returns dict: perm_key β†’ permutation array."""
1001
- import gc
1002
- print(" β”Œβ”€β”€ Fold-Basis pre-pass: computing dimension permutations ──┐")
1003
-
1004
- perms = {}
1005
- hidden_stat = None
1006
- hidden_dim = None
1007
- hidden_count = 0
1008
-
1009
- # ── Pass 1: P_hidden from first 3 layers' W_q ──
1010
- _HIDDEN_SAMPLE = 3
1011
- for ti in tensor_infos:
1012
- if hidden_count >= _HIDDEN_SAMPLE:
1013
- break
1014
- if not _re.match(r'blk\.\d+\.attn_q\.weight', ti['name']):
1015
- continue
1016
- if ti['n_dims'] < 2:
1017
- continue
1018
- d0 = int(ti['dims'][0])
1019
- if d0 % QK_K != 0:
1020
- continue
1021
-
1022
- abs_offset = data_section_start + ti['offset']
1023
- fin.seek(abs_offset)
1024
- raw = fin.read(ti['data_size'])
1025
- if ti['type'] == GGML_TYPE_BF16:
1026
- f32 = bf16_to_f32(raw, ti['n_elements'])
1027
- elif ti['type'] == GGML_TYPE_F16:
1028
- f32 = f16_to_f32(raw, ti['n_elements'])
1029
- elif ti['type'] == GGML_TYPE_F32:
1030
- f32 = np.frombuffer(raw, dtype=np.float32)
1031
- else:
1032
- continue
1033
-
1034
- d1 = int(ti['dims'][1]) if ti['n_dims'] > 1 else 1
1035
- M = f32.reshape(d1, d0)
1036
- col_mean = np.mean(M.astype(np.float64), axis=0)
1037
- del f32, M, raw
1038
- gc.collect()
1039
-
1040
- if hidden_stat is None:
1041
- hidden_dim = d0
1042
- hidden_stat = col_mean
1043
- else:
1044
- hidden_stat += col_mean
1045
- hidden_count += 1
1046
- print(f" β”‚ P_hidden sample: {ti['name']}")
1047
-
1048
- if hidden_stat is not None:
1049
- perms['hidden'] = _fb_compute_perm(hidden_stat)
1050
- print(f" β”‚ P_hidden: dim={hidden_dim}, from {hidden_count} W_q tensors")
1051
-
1052
- # ── Pass 2: P_inter from each layer's W_down ──
1053
- n_inter = 0
1054
- for ti in tensor_infos:
1055
- m = _re.match(r'blk\.(\d+)\.ffn_down\.weight', ti['name'])
1056
- if not m:
1057
- continue
1058
- if ti['n_dims'] < 2:
1059
- continue
1060
- d0 = int(ti['dims'][0])
1061
- if d0 % QK_K != 0:
1062
- continue
1063
-
1064
- layer = m.group(1)
1065
- abs_offset = data_section_start + ti['offset']
1066
- fin.seek(abs_offset)
1067
- raw = fin.read(ti['data_size'])
1068
- if ti['type'] == GGML_TYPE_BF16:
1069
- f32 = bf16_to_f32(raw, ti['n_elements'])
1070
- elif ti['type'] == GGML_TYPE_F16:
1071
- f32 = f16_to_f32(raw, ti['n_elements'])
1072
- elif ti['type'] == GGML_TYPE_F32:
1073
- f32 = np.frombuffer(raw, dtype=np.float32)
1074
- else:
1075
- continue
1076
-
1077
- d1 = int(ti['dims'][1]) if ti['n_dims'] > 1 else 1
1078
- M = f32.reshape(d1, d0)
1079
- col_mean = np.mean(M.astype(np.float64), axis=0)
1080
- del f32, M, raw
1081
- gc.collect()
1082
-
1083
- perms[f'inter_{layer}'] = _fb_compute_perm(col_mean)
1084
- n_inter += 1
1085
-
1086
- if n_inter > 0:
1087
- print(f" β”‚ P_inter: {n_inter} layers (from W_down column means)")
1088
 
1089
- print(f" └── {len(perms)} permutations computed β”€β”€β”˜")
1090
- print()
1091
- return perms
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1092
 
1093
 
1094
  def is_attention_tensor(name):
@@ -1232,7 +894,7 @@ def main():
1232
  if len(sys.argv) < 3:
1233
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
1234
  " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
1235
- " [--fold-interleave] [--fold-basis] [--fold-hadamard]")
1236
  sys.exit(1)
1237
 
1238
  input_path = sys.argv[1]
@@ -1242,9 +904,6 @@ def main():
1242
  q2all = '--q2all' in sys.argv
1243
  keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
1244
  fold_interleave = '--fold-interleave' in sys.argv # fold-optimal weight interleave for DC/vesica cancellation
1245
- fold_basis = '--fold-basis' in sys.argv # whole-dimension fold-optimal basis change (baked into weights)
1246
- fold_hadamard = '--fold-hadamard' in sys.argv # per-block Walsh-Hadamard transform (baked, zero storage)
1247
- fold_sort_refit = '--fold-sort-refit' in sys.argv # sort→HPC→unsort→refit (baked, stock llama compatible)
1248
 
1249
  # Check for imatrix
1250
  imatrix_data = None
@@ -1275,12 +934,6 @@ def main():
1275
  print(" β•‘ Engine: Python (numpy vectorized) β•‘")
1276
  if fold_interleave:
1277
  print(" β•‘ Fold Interleave: ON (sort-pair DC/vesica cancellation) β•‘")
1278
- if fold_basis:
1279
- print(" β•‘ Fold Basis: ON (whole-dim permutation, baked in) β•‘")
1280
- if fold_hadamard:
1281
- print(" β•‘ Fold Hadamard: ON (per-block WHT, zero storage) β•‘")
1282
- if fold_sort_refit:
1283
- print(" ║ Fold Sort-Refit: ON (sort→HPC→unsort→refit, stock llama) ║")
1284
  print(" β•šβ•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•")
1285
  print()
1286
 
@@ -1412,16 +1065,6 @@ def main():
1412
  print(f" Tensors to keep as-is: {total_keep}")
1413
  print()
1414
 
1415
- # ── Fold-Basis pre-pass ──
1416
- # Compute whole-dimension permutations from weight statistics.
1417
- # This reads relevant tensors from the input to gather column
1418
- # energies, then computes the fold-optimal permutation for each
1419
- # shared dimension. The permutations are applied during the
1420
- # main quantization loop below.
1421
- fb_perms = {}
1422
- if fold_basis:
1423
- fb_perms = _fb_prepass(fin, tensor_infos, data_section_start)
1424
-
1425
  # ── Compute output tensor sizes and offsets ──
1426
  out_tensor_infos = []
1427
  out_data_offset = 0
@@ -1566,24 +1209,22 @@ def main():
1566
  else:
1567
  updated_kv.append((key, vtype, raw_value))
1568
 
 
 
 
 
 
 
 
1569
  # ── Write output GGUF ──
1570
  print(" Writing output GGUF...")
1571
-
1572
- # ── Inject fold-hadamard metadata ──
1573
- n_kv_out = n_kv
1574
- if fold_hadamard:
1575
- # GGUF KV type 8 = BOOL (stored as int8: 1=true)
1576
- updated_kv.append(('hexstate.fold_hadamard', 8,
1577
- struct.pack('<b', 1)))
1578
- n_kv_out += 1
1579
- print(" Added hexstate.fold_hadamard=true to GGUF metadata")
1580
-
1581
  with open(output_path, 'wb') as fout:
1582
  # Header
1583
  fout.write(struct.pack('<I', GGUF_MAGIC))
1584
  fout.write(struct.pack('<I', GGUF_VERSION))
1585
  fout.write(struct.pack('<Q', n_tensors))
1586
- fout.write(struct.pack('<Q', n_kv_out))
1587
 
1588
  # KV pairs (passthrough)
1589
  for key, vtype, raw_value in updated_kv:
@@ -1614,9 +1255,6 @@ def main():
1614
  q2k_rmse_sum = 0.0
1615
  q2k_tensor_count = 0
1616
 
1617
- # Fold-interleave sidecar: tensor_name β†’ int16 permutation indices
1618
- fold_perm_sidecar = {} if fold_interleave else None
1619
-
1620
  for i, ti in enumerate(tensor_infos):
1621
  # Progress bar
1622
  pct = (i + 1) / n_tensors * 100
@@ -1651,12 +1289,6 @@ def main():
1651
  n_el = ti['n_elements']
1652
  n_blocks_q8 = n_el // 32
1653
 
1654
- # ── Fold-Basis ──
1655
- if fold_basis and fb_perms:
1656
- for perm_key, axis in _fb_tensor_perms(ti['name']):
1657
- if perm_key in fb_perms:
1658
- f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
1659
-
1660
  if use_hpc and hasattr(_HEXSTATE_LIB, 'hexstate_quantize_tensor_q8_0_hpc'):
1661
  output_buf = np.zeros(n_blocks_q8 * 34, dtype=np.uint8)
1662
  error = ctypes.c_float(0.0)
@@ -1702,13 +1334,6 @@ def main():
1702
  continue
1703
 
1704
  n_el = len(f32)
1705
-
1706
- # ── Fold-Basis ──
1707
- if fold_basis and fb_perms:
1708
- for perm_key, axis in _fb_tensor_perms(ti['name']):
1709
- if perm_key in fb_perms:
1710
- f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
1711
-
1712
  if n_el % 32:
1713
  raise ValueError(f"Q4_0 tensor {ti['name']} is not row/block aligned")
1714
  n_blocks_q4 = n_el // 32
@@ -1772,46 +1397,15 @@ def main():
1772
  fout.write(b'\x00' * pad)
1773
  continue
1774
 
1775
- # ── Fold-Basis pre-processing ───────────────────────
1776
- # Apply whole-dimension permutations computed in the
1777
- # pre-pass. These are baked into the weight matrix β€”
1778
- # the output GGUF is self-contained, no sidecar.
1779
- if fold_basis and fb_perms:
1780
- for perm_key, axis in _fb_tensor_perms(ti['name']):
1781
- if perm_key in fb_perms:
1782
- f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
1783
-
1784
- # ── Fold-Hadamard pre-processing ────────────────────
1785
- # Per-block Walsh-Hadamard transform. Decorrelates
1786
- # weights so quantization errors are independent β†’
1787
- # DC and vesica go to zero structurally. No storage
1788
- # overhead β€” H is defined by block size alone.
1789
- # The dequant kernel applies H⁻¹ = H/n after unpacking.
1790
- if fold_hadamard:
1791
- apply_fold_hadamard(f32)
1792
-
1793
- # ── Fold-Sort-Refit pre-processing ────────────────
1794
- # Sort each block β†’ quantize sorted with HPC β†’ unsort
1795
- # codes β†’ refit sub-block scales. Produces standard
1796
- # Q2_K blocks compatible with stock llama.cpp.
1797
- sort_perms_for_refit = None
1798
- f32_presort = None
1799
- if fold_sort_refit:
1800
- n_blk_sr = len(f32) // QK_K
1801
- f32_presort = f32.reshape(n_blk_sr, QK_K).copy()
1802
- sort_perms_for_refit = np.argsort(f32_presort, axis=1)
1803
- f32 = np.take_along_axis(f32_presort, sort_perms_for_refit, axis=1).reshape(-1).astype(np.float32)
1804
-
1805
- # ── Fold-Interleave pre-processing ──────────────────
1806
  # Sort-pair weights within each QK_K block so that the
1807
  # k-th smallest and k-th largest are at fold-complementary
1808
  # positions (k, k+128). The HPC pipeline then sees blocks
1809
  # whose vesica/DC structure is pre-aligned for cancellation.
1810
- # f32_orig is kept for RMSE reporting against true weights.
1811
- fold_perm_flat = None
1812
- f32_orig = f32 # alias β€” not copied unless fold is active
1813
  if fold_interleave:
1814
- f32_orig = f32.copy()
1815
  f32, fold_perm_flat = apply_fold_interleave(f32)
1816
 
1817
  # Quantize to Q2_K β€” always use HPC with chunked processing
@@ -1823,7 +1417,7 @@ def main():
1823
  imat_full = _expand_imatrix(ti, imatrix_data)
1824
 
1825
  # If fold-interleave is active, permute the imatrix the same way
1826
- if fold_interleave and imat_full is not None and fold_perm_flat is not None:
1827
  imat_full = apply_fold_interleave_importance(imat_full, fold_perm_flat)
1828
 
1829
  n_el = ti['n_elements']
@@ -1869,25 +1463,10 @@ def main():
1869
  n_blocks = n_el // QK_K
1870
  fout.write(q2k_data)
1871
 
1872
- # ── Sort-Refit post-processing ────────────────────
1873
- # Unsort the HPC-optimized codes back to original
1874
- # weight positions and refit sub-block scales.
1875
- if fold_sort_refit and sort_perms_for_refit is not None:
1876
- # Seek back β€” we need to overwrite what we just wrote
1877
- fout.seek(fout.tell() - len(q2k_data))
1878
- q2k_data = _sort_refit_q2k(
1879
- f32_presort, q2k_data,
1880
- sort_perms_for_refit, n_blocks)
1881
- fout.write(q2k_data)
1882
-
1883
- # ── Store fold-interleave permutation in sidecar ──
1884
- if fold_interleave and fold_perm_flat is not None:
1885
- fold_perm_sidecar[ti['name']] = fold_perm_flat
1886
-
1887
  # ── Compute and report exact per-tensor RMSE ──
1888
- # When fold-interleave is active, dequantized values are
1889
- # in permuted order β€” inverse-permute before comparing
1890
- # against the original (un-permuted) weights.
1891
  try:
1892
  CHUNK_BLK = 100_000 # blocks per chunk to bound memory
1893
  total_se = 0.0
@@ -1896,10 +1475,7 @@ def main():
1896
  ce = min(ci + CHUNK_BLK, n_blocks)
1897
  chunk_q = q2k_data[ci*84:ce*84]
1898
  deq_chunk = dequant_q2k_fast(chunk_q, ce - ci)
1899
- if fold_interleave and fold_perm_flat is not None:
1900
- perm_slice = fold_perm_flat[ci*QK_K:ce*QK_K]
1901
- deq_chunk = invert_fold_interleave(deq_chunk, perm_slice)
1902
- orig_chunk = f32_orig[ci*QK_K:ce*QK_K]
1903
  n_valid = min(len(orig_chunk), len(deq_chunk))
1904
  diff = orig_chunk[:n_valid] - deq_chunk[:n_valid]
1905
  total_se += np.sum(diff ** 2)
@@ -1907,7 +1483,7 @@ def main():
1907
  tensor_rmse = np.sqrt(total_se / max(total_n, 1))
1908
  q2k_rmse_sum += tensor_rmse
1909
  q2k_tensor_count += 1
1910
- fi_tag = "Β·SR" if fold_sort_refit else ("Β·FH" if fold_hadamard else ("Β·FB" if fold_basis else ("Β·FI" if fold_interleave else "")))
1911
  print(f"\n [Q2_K{fi_tag}] {ti['name'][:52]} RMSE={tensor_rmse:.6e}")
1912
  except Exception as e:
1913
  print(f"\n [Q2_K] {ti['name'][:55]} RMSE=err({e})")
@@ -1915,31 +1491,7 @@ def main():
1915
  quant_count += 1
1916
  total_quant_bytes += len(q2k_data)
1917
  else:
1918
- # Keep as-is (passthrough) β€” but apply fold-basis to
1919
- # 1D norm weights so the residual stream is consistent
1920
- if fold_basis and fb_perms:
1921
- tp = _fb_tensor_perms(ti['name'])
1922
- if tp:
1923
- if ti['type'] == GGML_TYPE_BF16:
1924
- f32 = bf16_to_f32(raw_data, ti['n_elements'])
1925
- elif ti['type'] == GGML_TYPE_F16:
1926
- f32 = f16_to_f32(raw_data, ti['n_elements'])
1927
- elif ti['type'] == GGML_TYPE_F32:
1928
- f32 = np.frombuffer(raw_data, dtype=np.float32).copy()
1929
- else:
1930
- f32 = None
1931
- if f32 is not None:
1932
- for perm_key, axis in tp:
1933
- if perm_key in fb_perms:
1934
- f32 = _fb_apply(f32, ti['dims'], fb_perms[perm_key], axis)
1935
- # Write permuted data in original type
1936
- if ti['type'] == GGML_TYPE_BF16:
1937
- raw_data = f32.astype(np.float32).view(np.uint32)
1938
- raw_data = (((raw_data >> 16) & 0xFFFF).astype(np.uint16)).tobytes()
1939
- elif ti['type'] == GGML_TYPE_F16:
1940
- raw_data = f32.astype(np.float16).tobytes()
1941
- elif ti['type'] == GGML_TYPE_F32:
1942
- raw_data = f32.astype(np.float32).tobytes()
1943
  fout.write(raw_data)
1944
  total_keep_bytes += len(raw_data)
1945
 
@@ -1978,20 +1530,15 @@ def main():
1978
  print()
1979
  print(f" Output: {output_path}")
1980
 
1981
- # ── Save fold-interleave sidecar ──
1982
- if fold_interleave and fold_perm_sidecar:
1983
- sidecar_path = output_path + '.fold_perm.npz'
1984
- np.savez_compressed(sidecar_path, **fold_perm_sidecar)
1985
- sidecar_size = os.path.getsize(sidecar_path)
1986
- n_fi_tensors = len(fold_perm_sidecar)
1987
- print(f" Fold-interleave sidecar: {sidecar_path}")
1988
- print(f" {n_fi_tensors} tensors, {sidecar_size/1024:.0f} KB compressed")
1989
  print()
1990
  print(" β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”")
1991
- print(" β”‚ Inference: for each tensor in the sidecar, apply the β”‚")
1992
- print(" β”‚ inverse permutation after Q2_K dequant (per QK_K block), β”‚")
1993
- print(" β”‚ or equivalently permute the activation vector before the β”‚")
1994
- print(" β”‚ matmul. inv_perm = np.argsort(perm) per block. β”‚")
 
1995
  print(" β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜")
1996
  print()
1997
 
 
659
  # across all fold pairs, so the spectral objective's vesica/DC terms
660
  # can be driven to zero by the unchanged HPC pipeline.
661
  #
662
+ # The permutation is a pre-processing step β€” the quantization algorithm
663
+ # sees 256 floats and runs identically. A sidecar .npz stores the
664
+ # per-block permutation indices; the inference engine applies the
665
+ # inverse permutation after dequantization (or equivalently permutes
666
+ # the activation vector).
667
+
668
+ def compute_fold_permutation(block_weights):
669
+ """Sort-interleave permutation for one QK_K block.
670
+
671
+ Returns perm such that permuted[k] = block[perm[k]], with
672
+ perm[0..127] = argsort ascending (smallest half)
673
+ perm[128..255] = argsort descending (largest half, reversed)
674
+
675
+ Every fold pair (k, k+128) therefore contains the k-th smallest
676
+ and the k-th largest weight, making their sum approximately
677
+ constant across all 128 pairs.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
678
  """
679
+ n = len(block_weights)
680
+ half = n // 2
681
+ s = np.argsort(block_weights)
682
+ perm = np.empty(n, dtype=np.intp)
683
+ perm[:half] = s[:half]
684
+ perm[half:] = s[-1:half-1:-1]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
685
  return perm
686
 
687
 
688
+ def apply_fold_interleave(f32_data, block_size=QK_K, chunk_blocks=32768):
689
+ """Apply fold-optimal permutation to every block in a flat tensor.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
690
 
691
+ Returns (permuted_data, perm_indices) where perm_indices is a flat
692
+ int16 array of the same length β€” block b's permutation lives at
693
+ [b*block_size : (b+1)*block_size]. int16 suffices since indices
694
+ are 0..255.
695
 
696
+ Memory-optimized: chunks the vectorised argsort (int64 8 bytes/idx)
697
+ to avoid a 400 MB full-tensor allocation on 50M-element tensors
698
+ (Qwen 27B attn_qkv), and permutes in-place so peak is ~1.7x
699
+ less. Squared error is permutation-invariant, so RMSE needs no
700
+ inverse permutation.
701
+ """
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
702
  n = len(f32_data)
703
  n_blocks = n // block_size
704
+ # Ensure contiguous + writeable for in-place chunk updates
705
+ if not f32_data.flags.writeable or not f32_data.flags.c_contiguous:
706
+ f32_data = np.ascontiguousarray(f32_data)
707
+ perm_flat = np.empty(n, dtype=np.int16)
708
+ blocks = f32_data.reshape(n_blocks, block_size)
709
+ perm_blocks = perm_flat.reshape(n_blocks, block_size)
710
+ half = block_size // 2
711
+ for start in range(0, n_blocks, chunk_blocks):
712
+ end = min(start + chunk_blocks, n_blocks)
713
+ chunk = blocks[start:end] # view into f32_data
714
+ sorted_idx = np.argsort(chunk, axis=1) # chunk-sized int64 only
715
+ perm_chunk = np.empty((end - start, block_size), dtype=np.intp)
716
+ perm_chunk[:, :half] = sorted_idx[:, :half]
717
+ perm_chunk[:, half:] = sorted_idx[:, -1:half-1:-1]
718
+ tmp = np.take_along_axis(chunk, perm_chunk, axis=1)
719
+ chunk[:, :] = tmp
720
+ perm_blocks[start:end] = perm_chunk.astype(np.int16, copy=False)
721
+ return f32_data, perm_flat
722
+
723
+
724
+ def apply_fold_interleave_importance(importance, perm_flat, block_size=QK_K, chunk_blocks=32768):
725
+ """Permute an importance vector with the same fold permutation.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
726
 
727
+ Chunked + in-place: avoids a full-tensor int64 perm copy
728
+ (400 MB at 50M elements) and a duplicate float32 buffer.
729
+ """
730
+ n = len(importance)
731
+ n_blocks = n // block_size
732
+ if not importance.flags.writeable or not importance.flags.c_contiguous:
733
+ importance = np.ascontiguousarray(importance)
734
+ blocks = importance.reshape(n_blocks, block_size)
735
+ perm_blocks = perm_flat.reshape(n_blocks, block_size)
736
+ for start in range(0, n_blocks, chunk_blocks):
737
+ end = min(start + chunk_blocks, n_blocks)
738
+ chunk = blocks[start:end]
739
+ perm_chunk = perm_blocks[start:end].astype(np.intp, copy=False)
740
+ tmp = np.take_along_axis(chunk, perm_chunk, axis=1)
741
+ chunk[:, :] = tmp
742
+ return importance
743
+
744
+
745
+ def invert_fold_interleave(data, perm_flat, block_size=QK_K):
746
+ """Apply the inverse fold permutation to restore original order."""
747
+ n = len(data)
748
+ n_blocks = n // block_size
749
+ blocks = data.reshape(n_blocks, block_size)
750
+ perms = perm_flat.reshape(n_blocks, block_size).astype(np.intp)
751
+ inv_perms = np.argsort(perms, axis=1)
752
+ result = np.take_along_axis(blocks, inv_perms, axis=1)
753
+ return result.reshape(-1)
754
 
755
 
756
  def is_attention_tensor(name):
 
894
  if len(sys.argv) < 3:
895
  print("Usage: python3 hexstate_requantize.py <input.gguf> <output.gguf>"
896
  " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]"
897
+ " [--fold-interleave]")
898
  sys.exit(1)
899
 
900
  input_path = sys.argv[1]
 
904
  q2all = '--q2all' in sys.argv
905
  keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0
906
  fold_interleave = '--fold-interleave' in sys.argv # fold-optimal weight interleave for DC/vesica cancellation
 
 
 
907
 
908
  # Check for imatrix
909
  imatrix_data = None
 
934
  print(" β•‘ Engine: Python (numpy vectorized) β•‘")
935
  if fold_interleave:
936
  print(" β•‘ Fold Interleave: ON (sort-pair DC/vesica cancellation) β•‘")
 
 
 
 
 
 
937
  print(" β•šβ•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•β•")
938
  print()
939
 
 
1065
  print(f" Tensors to keep as-is: {total_keep}")
1066
  print()
1067
 
 
 
 
 
 
 
 
 
 
 
1068
  # ── Compute output tensor sizes and offsets ──
1069
  out_tensor_infos = []
1070
  out_data_offset = 0
 
1209
  else:
1210
  updated_kv.append((key, vtype, raw_value))
1211
 
1212
+ # ── Pre-compute fold-interleave permutations for GGUF KV pairs ──
1213
+ # Instead of per-tensor permutations (expensive to pre-compute),
1214
+ # just store a flag indicating fold-interleave was used.
1215
+ # FoldLlama will apply inverse permutation to all Q2_K tensors.
1216
+ if fold_interleave:
1217
+ updated_kv.append(("fold_interleave", 8, struct.pack('<Q', 4) + b"true"))
1218
+
1219
  # ── Write output GGUF ──
1220
  print(" Writing output GGUF...")
1221
+ n_kv = len(updated_kv)
 
 
 
 
 
 
 
 
 
1222
  with open(output_path, 'wb') as fout:
1223
  # Header
1224
  fout.write(struct.pack('<I', GGUF_MAGIC))
1225
  fout.write(struct.pack('<I', GGUF_VERSION))
1226
  fout.write(struct.pack('<Q', n_tensors))
1227
+ fout.write(struct.pack('<Q', n_kv))
1228
 
1229
  # KV pairs (passthrough)
1230
  for key, vtype, raw_value in updated_kv:
 
1255
  q2k_rmse_sum = 0.0
1256
  q2k_tensor_count = 0
1257
 
 
 
 
1258
  for i, ti in enumerate(tensor_infos):
1259
  # Progress bar
1260
  pct = (i + 1) / n_tensors * 100
 
1289
  n_el = ti['n_elements']
1290
  n_blocks_q8 = n_el // 32
1291
 
 
 
 
 
 
 
1292
  if use_hpc and hasattr(_HEXSTATE_LIB, 'hexstate_quantize_tensor_q8_0_hpc'):
1293
  output_buf = np.zeros(n_blocks_q8 * 34, dtype=np.uint8)
1294
  error = ctypes.c_float(0.0)
 
1334
  continue
1335
 
1336
  n_el = len(f32)
 
 
 
 
 
 
 
1337
  if n_el % 32:
1338
  raise ValueError(f"Q4_0 tensor {ti['name']} is not row/block aligned")
1339
  n_blocks_q4 = n_el // 32
 
1397
  fout.write(b'\x00' * pad)
1398
  continue
1399
 
1400
+ # ── Fold-Interleave pre-processing ──
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1401
  # Sort-pair weights within each QK_K block so that the
1402
  # k-th smallest and k-th largest are at fold-complementary
1403
  # positions (k, k+128). The HPC pipeline then sees blocks
1404
  # whose vesica/DC structure is pre-aligned for cancellation.
1405
+ # No f32_orig copy β€” squared error is invariant under
1406
+ # permutation, so RMSE on permuted order is identical.
1407
+ # apply_fold_interleave permutes f32 in-place chunked.
1408
  if fold_interleave:
 
1409
  f32, fold_perm_flat = apply_fold_interleave(f32)
1410
 
1411
  # Quantize to Q2_K β€” always use HPC with chunked processing
 
1417
  imat_full = _expand_imatrix(ti, imatrix_data)
1418
 
1419
  # If fold-interleave is active, permute the imatrix the same way
1420
+ if fold_interleave and imat_full is not None:
1421
  imat_full = apply_fold_interleave_importance(imat_full, fold_perm_flat)
1422
 
1423
  n_el = ti['n_elements']
 
1463
  n_blocks = n_el // QK_K
1464
  fout.write(q2k_data)
1465
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1466
  # ── Compute and report exact per-tensor RMSE ──
1467
+ # RMSE is invariant under permutation: with fold-interleave
1468
+ # both f32 and deq_chunk are in permuted order, so direct
1469
+ # comparison is identical and avoids the inverse perm + copy.
1470
  try:
1471
  CHUNK_BLK = 100_000 # blocks per chunk to bound memory
1472
  total_se = 0.0
 
1475
  ce = min(ci + CHUNK_BLK, n_blocks)
1476
  chunk_q = q2k_data[ci*84:ce*84]
1477
  deq_chunk = dequant_q2k_fast(chunk_q, ce - ci)
1478
+ orig_chunk = f32[ci*QK_K:ce*QK_K]
 
 
 
1479
  n_valid = min(len(orig_chunk), len(deq_chunk))
1480
  diff = orig_chunk[:n_valid] - deq_chunk[:n_valid]
1481
  total_se += np.sum(diff ** 2)
 
1483
  tensor_rmse = np.sqrt(total_se / max(total_n, 1))
1484
  q2k_rmse_sum += tensor_rmse
1485
  q2k_tensor_count += 1
1486
+ fi_tag = "Β·FI" if fold_interleave else ""
1487
  print(f"\n [Q2_K{fi_tag}] {ti['name'][:52]} RMSE={tensor_rmse:.6e}")
1488
  except Exception as e:
1489
  print(f"\n [Q2_K] {ti['name'][:55]} RMSE=err({e})")
 
1491
  quant_count += 1
1492
  total_quant_bytes += len(q2k_data)
1493
  else:
1494
+ # Keep as-is (passthrough)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1495
  fout.write(raw_data)
1496
  total_keep_bytes += len(raw_data)
1497
 
 
1530
  print()
1531
  print(f" Output: {output_path}")
1532
 
1533
+ if fold_interleave:
1534
+ print(f" Fold-interleave: ON β€” permutations applied in-quad, no sidecar")
 
 
 
 
 
 
1535
  print()
1536
  print(" β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”")
1537
+ print(" β”‚ Fold-permuted weights are stored directly in the GGUF β”‚")
1538
+ print(" β”‚ β€” no separate .npz sidecar needed. The fold permutation β”‚")
1539
+ print(" β”‚ must be inverted after dequant (per QK_K block) by the β”‚")
1540
+ print(" β”‚ inference engine, or equivalently permute the activation β”‚")
1541
+ print(" β”‚ vector before the matmul. β”‚")
1542
  print(" β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜")
1543
  print()
1544