#!/usr/bin/env python3 """ HexState GGUF Re-Quantizer — GGUF-to-GGUF Q2_K quantization. Reads a source GGUF (F16/BF16/F32), copies all metadata verbatim, and re-quantizes eligible weight tensors to Q2_K. When libhexstate_q2k.so is available, the C HExState optimizer is used for the Q2_K/Q4_0/Q8_0 paths; otherwise the numpy fallback is used. This bypasses the tokenizer parsing problem entirely — the source GGUF (from llama.cpp's convert_hf_to_gguf.py) has correct metadata. Usage: python3 hexstate_requantize.py input.gguf output.gguf """ import struct import sys import time import os import io import ctypes import numpy as np # ─── HExState C Library (HPC-optimized Q2_K quantization) ────────────────── _HEXSTATE_LIB = None def _load_hexstate_lib(): """Try to load the HExState C shared library for HPC-optimized quantization.""" global _HEXSTATE_LIB if _HEXSTATE_LIB is not None: return _HEXSTATE_LIB lib_dir = os.path.dirname(os.path.abspath(__file__)) lib_path = os.path.join(lib_dir, "libhexstate_q2k.so") if not os.path.exists(lib_path): return None try: lib = ctypes.CDLL(lib_path) # void hexstate_init(void) lib.hexstate_init.restype = None lib.hexstate_init.argtypes = [] # void hexstate_quantize_tensor_q2k(const float*, int64_t, void*, float*, int, int) lib.hexstate_quantize_tensor_q2k.restype = None lib.hexstate_quantize_tensor_q2k.argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.c_int, # opt_mode (0=HPC, 1=MSE, 2=Hybrid) ctypes.c_int, # verbose ] lib.hexstate_q2k_block_bytes.restype = ctypes.c_int lib.hexstate_q2k_block_bytes.argtypes = [] lib.hexstate_q2k_block_elements.restype = ctypes.c_int lib.hexstate_q2k_block_elements.argtypes = [] # imatrix-aware version lib.hexstate_quantize_tensor_q2k_imat.restype = None lib.hexstate_quantize_tensor_q2k_imat.argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.c_int, # opt_mode ctypes.POINTER(ctypes.c_float), # imat_importance (can be NULL) ctypes.c_int, # verbose ] # Row-aware imatrix version (DC cancellation respects row boundaries) if hasattr(lib, 'hexstate_quantize_tensor_q2k_imat_rowaware'): lib.hexstate_quantize_tensor_q2k_imat_rowaware.restype = None lib.hexstate_quantize_tensor_q2k_imat_rowaware.argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.c_int, # opt_mode ctypes.POINTER(ctypes.c_float), # imat_importance (can be NULL) ctypes.c_int, # verbose ctypes.c_int64, # row_width ] # Q8_0 HPC quantizer (Sieve pipeline; tied embeddings / LM head) if hasattr(lib, 'hexstate_quantize_tensor_q8_0_hpc'): lib.hexstate_quantize_tensor_q8_0_hpc.restype = None lib.hexstate_quantize_tensor_q8_0_hpc.argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.POINTER(ctypes.c_float), # imat_importance (can be NULL) ctypes.c_int, # verbose ] # Q4_0 HPC quantizer (for attention tensors) if hasattr(lib, 'hexstate_quantize_tensor_q4_0_hpc'): lib.hexstate_quantize_tensor_q4_0_hpc.restype = None lib.hexstate_quantize_tensor_q4_0_hpc.argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.POINTER(ctypes.c_float), # imat_importance (can be NULL) ctypes.c_int, # verbose ] if hasattr(lib, 'hexstate_set_spectral_params'): lib.hexstate_set_spectral_params.restype = None lib.hexstate_set_spectral_params.argtypes = [ ctypes.c_float, ctypes.c_float, ctypes.c_float] if hasattr(lib, 'hexstate_set_sse_budget'): lib.hexstate_set_sse_budget.restype = None lib.hexstate_set_sse_budget.argtypes = [ctypes.c_float] # IQ2_XS / IQ2_S (E8 codebook) quantizers + dequant for suffix in ('iq2_xs', 'iq2_s'): qfn = f'hexstate_quantize_tensor_{suffix}_hpc' if not hasattr(lib, qfn): continue getattr(lib, qfn).restype = None getattr(lib, qfn).argtypes = [ ctypes.POINTER(ctypes.c_float), # weights ctypes.c_int64, # n_elements ctypes.c_void_p, # output ctypes.POINTER(ctypes.c_float), # out_error ctypes.POINTER(ctypes.c_float), # imat_importance (can be NULL) ctypes.c_int, # verbose ctypes.c_int64, # row_width ] dfn = getattr(lib, f'hexstate_dequant_{suffix}') dfn.restype = None dfn.argtypes = [ctypes.c_void_p, ctypes.c_int64, ctypes.POINTER(ctypes.c_float)] lib.hexstate_init() dc_l = os.environ.get('HEX_DC_LAMBDA') vw_l = os.environ.get('HEX_VW_LAMBDA') dc_d = os.environ.get('HEX_DC_DECAY') if hasattr(lib, 'hexstate_set_spectral_params') and (dc_l or vw_l or dc_d): lib.hexstate_set_spectral_params( ctypes.c_float(float(dc_l) if dc_l else 1.0), ctypes.c_float(float(vw_l) if vw_l else 1.0), ctypes.c_float(float(dc_d) if dc_d else 0.85), ) # Relative SSE the DC/vesica shaper may spend. Q2_K default 5e-4; # IQ2_XS has no per-sub-block offset, so cancelling DC means swapping # codewords — it needs ~2e-2 (≈ +0.2% RMSE) to be effective. budget = os.environ.get('HEX_SSE_BUDGET') if budget is None and LOWBIT_FORMAT in ('iq2xs', 'iq2s', 'auto'): budget = '0.02' if budget is not None and hasattr(lib, 'hexstate_set_sse_budget'): lib.hexstate_set_sse_budget(ctypes.c_float(float(budget))) # Fold pyramid: HEX_FOLD_DEPTH levels below DC get vesica weight # (1 = classic single-fold vesica [default], 7 = full tree), geometric # per-level weight HEX_FOLD_GAMMA. HEX_CARRY_CUM=0 restores the legacy # previous-block-only carry (default carries the cumulative residual). fd = os.environ.get('HEX_FOLD_DEPTH') fg = os.environ.get('HEX_FOLD_GAMMA') fc = os.environ.get('HEX_CARRY_CUM') if hasattr(lib, 'hexstate_set_fold_params') and (fd or fg or fc): lib.hexstate_set_fold_params.restype = None lib.hexstate_set_fold_params.argtypes = [ctypes.c_int, ctypes.c_float, ctypes.c_int] lib.hexstate_set_fold_params( ctypes.c_int(int(fd) if fd else -1), ctypes.c_float(float(fg) if fg else 0.0), ctypes.c_int(int(fc) if fc else -1)) _HEXSTATE_LIB = lib return lib except Exception as e: print(f" WARNING: Failed to load HexState library: {e}") return None def _skip_gguf_kv_value(f, vtype): """Skip a GGUF KV value of the given type.""" import struct as st size_map = {0:1, 1:1, 2:2, 3:2, 4:4, 5:4, 6:4, 7:1, 10:8, 11:8, 12:8} if vtype == 8: # string slen = st.unpack(' normalized importance array (float32) """ import struct as st imat = {} with open(path, 'rb') as f: magic = st.unpack(' 0: importance = in_sum2 / count if use_sqrt: importance = np.sqrt(importance) else: importance = np.ones_like(in_sum2) mean = importance.mean() if mean > 1e-30: imat[base_name] = importance / mean else: imat[base_name] = np.ones_like(importance) else: # Legacy format: first 4 bytes were n_entries f.seek(0) n_entries = st.unpack(' 1e-30: imat[name] = values / mean else: imat[name] = np.ones_like(values) return imat def quantize_tensor_q2k_hpc(f32_data, opt_mode=2, importance=None, row_width=0): """Quantize tensor using HexState HPC-optimized C implementation. opt_mode: 0=HPC (BP only), 1=MSE (grid search), 2=Hybrid (recommended) importance: optional per-element importance weights (from imatrix) row_width: tensor row width in elements (for row-aware DC cancellation) Returns: (bytes, n_blocks) same as quantize_tensor_q2k() """ lib = _load_hexstate_lib() if lib is None: raise RuntimeError("HexState library not available") n_elements = len(f32_data) if n_elements % QK_K != 0: pad_len = QK_K - (n_elements % QK_K) f32_data = np.concatenate([f32_data, np.zeros(pad_len, dtype=np.float32)]) if importance is not None: importance = np.concatenate([importance, np.ones(pad_len, dtype=np.float32)]) n_elements = len(f32_data) n_blocks = n_elements // QK_K block_bytes = lib.hexstate_q2k_block_bytes() # 84 # Allocate output buffer output = np.zeros(n_blocks * block_bytes, dtype=np.uint8) error = ctypes.c_float(0.0) # Call C quantizer with or without importance weights f32_contiguous = np.ascontiguousarray(f32_data, dtype=np.float32) if importance is not None: imat_contiguous = np.ascontiguousarray(importance, dtype=np.float32) imat_ptr = imat_contiguous.ctypes.data_as(ctypes.POINTER(ctypes.c_float)) else: imat_ptr = None # Prefer row-aware path when available (DC cancellation respects row boundaries) if row_width > 0 and hasattr(lib, 'hexstate_quantize_tensor_q2k_imat_rowaware'): lib.hexstate_quantize_tensor_q2k_imat_rowaware( f32_contiguous.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ctypes.c_int64(n_elements), output.ctypes.data_as(ctypes.c_void_p), ctypes.byref(error), ctypes.c_int(opt_mode), imat_ptr, ctypes.c_int(0), # verbose ctypes.c_int64(row_width), ) else: lib.hexstate_quantize_tensor_q2k_imat( f32_contiguous.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ctypes.c_int64(n_elements), output.ctypes.data_as(ctypes.c_void_p), ctypes.byref(error), ctypes.c_int(opt_mode), imat_ptr, ctypes.c_int(0), # verbose ) return output.tobytes(), n_blocks # ─── Constants ────────────────────────────────────────────────────────────── GGUF_MAGIC = 0x46554747 GGUF_VERSION = 3 ALIGNMENT = 32 QK_K = 256 GGML_TYPE_F32 = 0 GGML_TYPE_F16 = 1 GGML_TYPE_Q4_0 = 2 GGML_TYPE_Q8_0 = 8 GGML_TYPE_Q2_K = 10 GGML_TYPE_IQ2_XS = 17 GGML_TYPE_IQ2_S = 22 GGML_TYPE_BF16 = 30 IQ2_XS_BLOCK_BYTES = 74 # d(fp16) + 32×u16 codes + 8 scale bytes IQ2_S_BLOCK_BYTES = 82 # d(fp16) + 32 idx + 32 sign bytes + 8 qh + 8 scales # Low-bit target for the "Q2_K plan" tensors: # 'q2k' (default) Q2_K, 2.625 bpw # 'iq2xs' (--iq2xs) IQ2_XS E8 codebook, 2.3125 bpw # 'iq2s' (--iq2s) IQ2_S E8 codebook, 2.5625 bpw # 'auto' (--iq2auto) per tensor: IQ2_S where its imatrix-weighted RMSE on a # row sample beats Q2_K (Gaussian-ish tensors), Q2_K where the # weights are outlier-heavy and Q2_K's min/scale offset wins. LOWBIT_FORMAT = 'q2k' # kind -> (ggml type, block bytes, display name, LLAMA_FTYPE_MOSTLY_*) LOWBIT_KINDS = { 'q2k': (GGML_TYPE_Q2_K, 84, 'Q2_K', 10), 'iq2xs': (GGML_TYPE_IQ2_XS, IQ2_XS_BLOCK_BYTES, 'IQ2_XS', 20), 'iq2s': (GGML_TYPE_IQ2_S, IQ2_S_BLOCK_BYTES, 'IQ2_S', 28), } TYPE_NAME = { 0: "F32", 1: "F16", 2: "Q4_0", 3: "Q4_1", 6: "Q5_0", 7: "Q5_1", 8: "Q8_0", 9: "Q8_1", 10: "Q2_K", 11: "Q3_K", 12: "Q4_K", 13: "Q5_K", 14: "Q6_K", 15: "Q8_K", 17: "IQ2_XS", 22: "IQ2_S", 30: "BF16", } # Block sizes and byte sizes for each type TYPE_BLOCK_SIZE = { 0: 1, 1: 1, 2: 32, 3: 32, 6: 32, 7: 32, 8: 32, 9: 32, 10: 256, 11: 256, 12: 256, 13: 256, 14: 256, 15: 256, 17: 256, 22: 256, 30: 1, } TYPE_BLOCK_BYTES = { 0: 4, 1: 2, 2: 18, 3: 20, 6: 20, 7: 22, 8: 34, 9: 36, 10: 84, 11: 110, 12: 144, 13: 176, 14: 210, 15: 292, 17: IQ2_XS_BLOCK_BYTES, 22: IQ2_S_BLOCK_BYTES, 30: 2, } def align_offset(offset, alignment=ALIGNMENT): return (offset + alignment - 1) & ~(alignment - 1) def read_string(f): slen = struct.unpack('> 16).astype(np.uint16) return bf16.tobytes() # ─── Q2_K quantization — faithful port of ggml quantize_row_q2_K_ref ─────── # Vectorized with numpy for performance. Uses make_qkx2_quants algorithm: # - Weighted MAD error with weights[i] = |x[i]| # - Joint scale+min least-squares solve # - 16-step grid search for initial iscale def quantize_tensor_q8_0(f32_data): """Vectorized ggml-faithful Q8_0 (fallback when the HPC lib is absent). Block: 32 weights -> fp16 d + 32 x int8 = 34 bytes; y = q * d. d = amax/127 (float), q = round(x/d), d stored as fp16 -- matches ggml quantize_row_q8_0_ref. Returns (bytes, n_blocks, sse).""" n = len(f32_data) if n % 32 != 0: f32_data = np.concatenate( [f32_data, np.zeros(32 - n % 32, dtype=np.float32)]) n = len(f32_data) blocks = f32_data.reshape(-1, 32).astype(np.float32) nb = blocks.shape[0] amax = np.max(np.abs(blocks), axis=1) d = amax / 127.0 id_ = np.where(d > 0, 1.0 / np.where(d > 0, d, 1.0), 0.0) qs = np.clip(np.rint(blocks * id_[:, None]), -127, 127).astype(np.int8) d16 = d.astype(' 0 this_scale = np.where(valid_D, (sum_w * sum_xl - sum_x * sum_l) / np.maximum(D, 1e-30), 0.0) this_min = np.where(valid_D, (sum_l2 * sum_x - sum_l * sum_xl) / np.maximum(D, 1e-30), 0.0) # If this_min > 0, clamp to 0 and recompute scale pos_min = this_min > 0 this_min = np.where(pos_min, 0.0, this_min) this_scale = np.where(pos_min & (sum_l2 > 0), sum_xl / np.maximum(sum_l2, 1e-30), this_scale) # Compute error for this trial recon = this_scale[:, :, None] * Laux + this_min[:, :, None] cur_error = (weights * np.abs(recon - data)).sum(axis=2) # Update where this trial is better better = valid_D & (cur_error < best_error) & ~degenerate if better.any(): # Expand mask to weight dimension for L update better3d = better[:, :, None] best_error = np.where(better, cur_error, best_error) best_scale = np.where(better, this_scale, best_scale) best_min = np.where(better, this_min, best_min) # the_min = -best_min (make positive) sb_scale = np.maximum(best_scale, 0.0).astype(np.float32) # [n_blocks, 16] sb_the_min = np.maximum(-best_min, 0.0).astype(np.float32) # [n_blocks, 16] # Handle degenerate sub-blocks sb_scale[degenerate] = 0.0 sb_the_min[degenerate] = np.maximum(-sb_min[degenerate], 0.0).astype(np.float32) # ── Phase 2: quantize scales/mins to 4-bit ── max_scale = sb_scale.max(axis=1) # [n_blocks] max_min = sb_the_min.max(axis=1) # [n_blocks] # Quantize sub-block scales to 4-bit has_scale = max_scale > 0 iscale_s = np.where(has_scale, q4scale / np.maximum(max_scale, 1e-30), 0.0) scales_q = np.where(has_scale[:, None], np.clip(np.round(iscale_s[:, None] * sb_scale), 0, 15), 0.0).astype(np.uint8) # Quantize sub-block mins to 4-bit has_min = max_min > 0 iscale_m = np.where(has_min, q4scale / np.maximum(max_min, 1e-30), 0.0) mins_q = np.where(has_min[:, None], np.clip(np.round(iscale_m[:, None] * sb_the_min), 0, 15), 0.0).astype(np.uint8) d_fp16 = np.where(has_scale, max_scale / q4scale, 0.0).astype(np.float16) dmin_fp16 = np.where(has_min, max_min / q4scale, 0.0).astype(np.float16) # ── Phase 3: requantize using fp16-truncated d/dmin ── scales_packed = scales_q | (mins_q << 4) # [n_blocks, 16] d_f32 = d_fp16.astype(np.float32) dmin_f32 = dmin_fp16.astype(np.float32) d_sub = d_f32[:, None] * (scales_packed & 0xF).astype(np.float32) dm_sub = dmin_f32[:, None] * (scales_packed >> 4).astype(np.float32) # l = nearest_int((x + dm) / d), clamp [0,3] valid_d = d_sub > 0 inv_d = np.where(valid_d, 1.0 / np.maximum(d_sub, 1e-30), 0.0) q_vals = np.where(valid_d[:, :, None], np.clip(np.round( (f32_data.reshape(n_blocks, 16, 16) + dm_sub[:, :, None]) * inv_d[:, :, None] ), 0, 3), 0).astype(np.uint8) # ── Phase 4: pack ── q_flat = q_vals.reshape(n_blocks, QK_K) q_groups = q_flat.reshape(n_blocks, 2, 4, 32) qs_packed = (q_groups[:, :, 0, :] | (q_groups[:, :, 1, :] << 2) | (q_groups[:, :, 2, :] << 4) | (q_groups[:, :, 3, :] << 6)).astype(np.uint8) qs_packed = qs_packed.reshape(n_blocks, 64) # Build output: [n_blocks, 84] bytes # Layout matches ggml block_q2_K: scales[16] | qs[64] | d(fp16) | dmin(fp16) result = np.zeros((n_blocks, 84), dtype=np.uint8) result[:, 0:16] = scales_packed result[:, 16:80] = qs_packed result[:, 80:82] = d_fp16.view(np.uint8).reshape(n_blocks, 2) result[:, 82:84] = dmin_fp16.view(np.uint8).reshape(n_blocks, 2) return result.tobytes(), n_blocks def dequant_q2k_fast(q2k_bytes, n_blocks): """Vectorized Q2_K dequantization for RMSE computation. Block layout (84 bytes) — same for both C struct and Python writer: scales[16] (bytes 0-15) | qs[64] (bytes 16-79) | d(fp16, bytes 80-81) | dmin(fp16, bytes 82-83) The C struct BlockQ2K in gguf_format.h is: { uint8_t scales[16]; uint8_t qs[64]; uint16_t d; uint16_t dmin; } Dequantization follows gguf_dequantize_q2_k_block() exactly: For each half (0..1), qs_half = qs[half*32 : half*32+32] For each shift j (0..3): scale_idx = half*8 + j*2 elements [0..15] use scales[scale_idx], from qs_half[0..15] >> (j*2) elements [16..31] use scales[scale_idx+1], from qs_half[16..31] >> (j*2) """ data = np.frombuffer(q2k_bytes, dtype=np.uint8).reshape(n_blocks, 84) # Extract fields scales_packed = data[:, 0:16] # [n_blocks, 16] qs = data[:, 16:80] # [n_blocks, 64] d_fp16 = data[:, 80:82].copy().view(np.float16).astype(np.float32).reshape(n_blocks) dmin_fp16 = data[:, 82:84].copy().view(np.float16).astype(np.float32).reshape(n_blocks) # Extract scale (low 4 bits) and min (high 4 bits) per sub-block sc = (scales_packed & 0xF).astype(np.float32) # [n_blocks, 16] mn = (scales_packed >> 4).astype(np.float32) # [n_blocks, 16] # Compute per-sub-block d_sub and m_sub d_sub = d_fp16[:, np.newaxis] * sc # [n_blocks, 16] m_sub = dmin_fp16[:, np.newaxis] * mn # [n_blocks, 16] # Unpack 2-bit quants from qs[64] into 256 values per block. # Matches C reference: two scales per 32-byte extraction (16 elements each). # half=0: qs[0..31], half=1: qs[32..63] # shift j=0..3: scale_idx = half*8 + j*2 (first 16), +1 (second 16) result = np.zeros((n_blocks, QK_K), dtype=np.float32) for half in range(2): qs_half = qs[:, half * 32:(half + 1) * 32] # [n_blocks, 32] for sub in range(4): # Extract 2-bit quants at this shift position q_vals = ((qs_half >> (sub * 2)) & 3).astype(np.float32) # [n_blocks, 32] base_idx = half * 128 + sub * 32 # First 16 elements: qs_half[0..15], scale index = half*8 + sub*2 si_0 = half * 8 + sub * 2 result[:, base_idx:base_idx + 16] = ( d_sub[:, si_0:si_0+1] * q_vals[:, :16] - m_sub[:, si_0:si_0+1] ) # Second 16 elements: qs_half[16..31], scale index = si_0 + 1 si_1 = si_0 + 1 result[:, base_idx + 16:base_idx + 32] = ( d_sub[:, si_1:si_1+1] * q_vals[:, 16:] - m_sub[:, si_1:si_1+1] ) return result.reshape(-1) def is_attention_tensor(name): """Detect attention Q/K/V/O projection tensors. These are the most sensitive to quantization and get promoted to Q4_0.""" attn_patterns = [ 'attn_q.weight', 'attn_k.weight', 'attn_v.weight', 'attn_output.weight', 'attn_qkv.weight', 'attn_gate.weight', 'self_attn.q_proj.weight', 'self_attn.k_proj.weight', 'self_attn.v_proj.weight', 'self_attn.o_proj.weight', # Qwen 3.6 DeltaNet SSM projections — treat as attention-class 'ssm_in_qkv.weight', 'ssm_in_z.weight', 'ssm_out.weight', 'linear_attn.in_proj_qkv.weight', 'linear_attn.in_proj_z.weight', 'linear_attn.out_proj.weight', ] for pat in attn_patterns: if pat in name: return True return False def is_q4_tensor(name): """Tensors demoted to Q4_0 (HPC sieve) while Q2_K tensors stay Q2_K. The full-precision leftovers: LM head / embedding tables plus SSM-side weights (ssm_alpha mirrors ssm_beta, which already ships Q2_K; ssm_conv1d kernels are small dense 2D projections). Attention/FFN are NOT matched here — they keep their Q2_K plan. NOTE: 'output.weight' is exact-matched so 'attn_output.weight' never matches. """ if name == 'output.weight' or name.endswith('.output.weight'): return True if name == 'token_embd.weight' or '.token_embd.weight' in name: return True for pat in ('ssm_alpha.weight', 'ssm_conv1d.weight', 'conv1d.weight'): if pat in name: return True return False def q2k_row_compatible(n_dims, dims): return n_dims >= 2 and bool(dims) and int(dims[0]) % QK_K == 0 def q4_row_compatible(n_dims, dims): return n_dims >= 2 and bool(dims) and int(dims[0]) % 32 == 0 def _expand_imatrix(ti, imatrix_data): """Expand an imatrix vector only when its length matches a tensor row.""" if not imatrix_data: return None iw = imatrix_data.get(ti['name']) if iw is None: return None iw = np.asarray(iw, dtype=np.float32).reshape(-1) n_el = int(ti['n_elements']) if iw.size == n_el: return np.ascontiguousarray(iw) if ti['n_dims'] < 2 or not ti['dims']: return None row_width = int(ti['dims'][0]) if row_width <= 0 or iw.size != row_width or n_el % row_width: print(f" WARNING: ignoring imatrix for {ti['name']}: importance length {iw.size} does not match row width {row_width}") return None return np.ascontiguousarray(np.tile(iw, n_el // row_width)) def _chunk_max_elems(): try: return max(QK_K, int(os.environ.get('HEX_CHUNK_ELEMS', '2000000'))) except ValueError: return 2_000_000 def _src_elem_nbytes(ttype): if ttype == GGML_TYPE_F32: return 4 if ttype in (GGML_TYPE_F16, GGML_TYPE_BF16): return 2 return None def _load_rows_f32(fin, abs_offset, ttype, d0, r0, r1): n = (r1 - r0) * d0 es = _src_elem_nbytes(ttype) if es is None: raise ValueError(f'cannot decode type {ttype} to f32') fin.seek(abs_offset + r0 * d0 * es) raw = fin.read(n * es) if ttype == GGML_TYPE_F32: return np.frombuffer(raw, dtype=np.float32).copy() if ttype == GGML_TYPE_F16: return np.frombuffer(raw, dtype=np.float16).astype(np.float32, copy=False) u = np.frombuffer(raw, dtype=np.uint16) return (u.astype(np.uint32) << 16).view(np.float32).copy() def _imat_chunk(ti, imatrix_data, r0, r1, d0): if not imatrix_data: return None iw = imatrix_data.get(ti['name']) if iw is None: return None iw = np.asarray(iw, dtype=np.float32).reshape(-1) n_rows = r1 - r0 if iw.size == int(ti['n_elements']): return np.ascontiguousarray(iw[r0 * d0:r1 * d0]) if iw.size == d0: return np.ascontiguousarray(np.tile(iw, n_rows)) return None def _copy_bytes(fin, fout, abs_offset, n_bytes): fin.seek(abs_offset) left = n_bytes buf = 16 * 1024 * 1024 written = 0 while left > 0: chunk = fin.read(min(buf, left)) if not chunk: break fout.write(chunk) written += len(chunk) left -= len(chunk) return written def quantize_tensor_iq2_hpc(f32_data, importance=None, row_width=0, kind='iq2xs', lanes=None): """IQ2_XS (74 B) or IQ2_S (82 B) per 256 weights via the HPC C library. Returns (bytes, n_blocks, dequant_f32). lanes: optional (U, ev[, lambda]) — U is (r, row_width) activation directions over the input columns (top eigenvectors of E[a aᵀ]) and ev their eigenvalues in E[a²] units; the sequential pass then also cancels the row error along each u_k with a cumulative carry.""" suffix = {'iq2xs': 'iq2_xs', 'iq2s': 'iq2_s'}[kind] block_bytes = LOWBIT_KINDS[kind][1] lib = _load_hexstate_lib() if lib is None or not hasattr(lib, f'hexstate_quantize_tensor_{suffix}_hpc'): raise RuntimeError(f'libhexstate_q2k.so lacks {LOWBIT_KINDS[kind][2]} support — rebuild') f32 = np.ascontiguousarray(f32_data, dtype=np.float32).reshape(-1) n = int(f32.size) if n % QK_K != 0: raise ValueError(f'{LOWBIT_KINDS[kind][2]} needs a multiple of {QK_K} elements, got {n}') n_blocks = n // QK_K out = np.zeros(n_blocks * block_bytes, dtype=np.uint8) err = ctypes.c_float(0.0) imat_ptr = None if importance is not None: imat_c = np.ascontiguousarray(importance, dtype=np.float32).reshape(-1) if imat_c.size == n: imat_ptr = imat_c.ctypes.data_as(ctypes.POINTER(ctypes.c_float)) lane_keep = None if lanes is not None and hasattr(lib, 'hexstate_set_activation_lanes') and row_width > 0: U = np.ascontiguousarray(lanes[0], dtype=np.float32).reshape(-1, int(row_width)) ev = np.ascontiguousarray(lanes[1], dtype=np.float32).reshape(-1) lam = float(lanes[2]) if len(lanes) > 2 else 1.0 lib.hexstate_set_activation_lanes.restype = None lib.hexstate_set_activation_lanes.argtypes = [ ctypes.POINTER(ctypes.c_float), ctypes.POINTER(ctypes.c_float), ctypes.c_int, ctypes.c_int64, ctypes.c_float] lib.hexstate_set_activation_lanes( U.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ev.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ctypes.c_int(int(min(U.shape[0], ev.size))), ctypes.c_int64(int(row_width)), ctypes.c_float(lam)) lane_keep = (U, ev) getattr(lib, f'hexstate_quantize_tensor_{suffix}_hpc')( f32.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ctypes.c_int64(n), out.ctypes.data_as(ctypes.c_void_p), ctypes.byref(err), imat_ptr, ctypes.c_int(0), ctypes.c_int64(int(row_width)), ) if lane_keep is not None: lib.hexstate_set_activation_lanes(None, None, ctypes.c_int(0), ctypes.c_int64(0), ctypes.c_float(0.0)) del lane_keep deq = np.zeros(n, dtype=np.float32) getattr(lib, f'hexstate_dequant_{suffix}')( out.ctypes.data_as(ctypes.c_void_p), ctypes.c_int64(n_blocks), deq.ctypes.data_as(ctypes.POINTER(ctypes.c_float))) return out.tobytes(), n_blocks, deq def quantize_tensor_iq2xs_hpc(f32_data, importance=None, row_width=0): return quantize_tensor_iq2_hpc(f32_data, importance, row_width, 'iq2xs') def _lowbit_dequant(kind, qbytes, n_blocks, n): """Dequantize a low-bit payload of any supported kind to f32[n].""" if kind == 'q2k': return dequant_q2k_fast(qbytes, n_blocks)[:n] lib = _load_hexstate_lib() suffix = {'iq2xs': 'iq2_xs', 'iq2s': 'iq2_s'}[kind] buf = np.frombuffer(qbytes, dtype=np.uint8) deq = np.zeros(n_blocks * QK_K, dtype=np.float32) getattr(lib, f'hexstate_dequant_{suffix}')( buf.ctypes.data_as(ctypes.c_void_p), ctypes.c_int64(n_blocks), deq.ctypes.data_as(ctypes.POINTER(ctypes.c_float))) return deq[:n] def auto_select_lowbit(fin, ti, abs_offset, imatrix_data, n_sample_rows=32, margin=None): """Pick 'iq2s' or 'q2k' for one tensor from a row sample. Encodes evenly spaced rows both ways and compares imatrix-weighted RMSE (the quantity that tracked PPL in the splice A/Bs). IQ2_S is 2.4% smaller, so it wins ties; HEX_AUTO_MARGIN (default 0) lets it win while up to that relative fraction *worse* if you want to bias toward size. Returns (kind, wrmse_q2k, wrmse_iq2s). """ if margin is None: margin = float(os.environ.get('HEX_AUTO_MARGIN', '0')) d0 = int(ti['dims'][0]) n_rows = int(ti['n_elements']) // d0 k = min(n_sample_rows, n_rows) rows = np.unique(np.linspace(0, n_rows - 1, k).astype(np.int64)) parts, imps = [], [] for r in rows: parts.append(_load_rows_f32(fin, abs_offset, ti['type'], d0, int(r), int(r) + 1)) imps.append(_imat_chunk(ti, imatrix_data, int(r), int(r) + 1, d0)) X = np.concatenate(parts).astype(np.float32) W = np.concatenate(imps).astype(np.float32) if imps[0] is not None else np.ones_like(X) if W.size != X.size: W = np.ones_like(X) n = X.size qb, nb = quantize_tensor_q2k_hpc(X, opt_mode=2, importance=W if imps[0] is not None else None, row_width=d0) e_q = _lowbit_dequant('q2k', qb, nb, n) - X sb, nb2, deq_s = quantize_tensor_iq2_hpc(X, importance=W if imps[0] is not None else None, row_width=d0, kind='iq2s') e_s = deq_s[:n] - X den = float(np.sum(W * X * X)) or 1.0 wr_q = float(np.sqrt(np.sum(W * e_q * e_q) / den)) wr_s = float(np.sqrt(np.sum(W * e_s * e_s) / den)) kind = 'iq2s' if wr_s <= wr_q * (1.0 + margin) else 'q2k' return kind, wr_q, wr_s def _stream_quantize(fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc): """Quantize one tensor in row chunks. kind: 'q2k' | 'iq2xs' | 'q4' | 'q8'. Returns (n_out_bytes, rmse_or_None, sigma_or_None). """ ttype = ti['type'] if _src_elem_nbytes(ttype) is None: n = _copy_bytes(fin, fout, abs_offset, ti['data_size']) return n, None, None if ti['n_dims'] < 2 or not ti['dims']: n = _copy_bytes(fin, fout, abs_offset, ti['data_size']) return n, None, None d0 = int(ti['dims'][0]) n_rows = int(ti['n_elements']) // d0 align = QK_K if kind in ('q2k', 'iq2xs', 'iq2s') else 32 if d0 % align != 0: raise ValueError(f'{ti["name"]} dim0={d0} not aligned to {align}') rows_per = max(1, _chunk_max_elems() // max(d0, 1)) total_se = 0.0 total_ss = 0.0 total_n = 0 written = 0 for r0 in range(0, n_rows, rows_per): r1 = min(n_rows, r0 + rows_per) f32 = _load_rows_f32(fin, abs_offset, ttype, d0, r0, r1) imp = _imat_chunk(ti, imatrix_data, r0, r1, d0) n_valid = int(f32.size) total_ss += float(np.vdot(f32, f32)) total_n += n_valid if kind in ('iq2xs', 'iq2s'): qbytes, n_blocks, deq = quantize_tensor_iq2_hpc( f32, importance=imp, row_width=d0, kind=kind) fout.write(qbytes) written += len(qbytes) diff = f32.reshape(-1)[:n_valid] - deq[:n_valid] total_se += float(np.sum(diff ** 2)) del qbytes, deq elif kind == 'q2k': if use_hpc: qbytes, n_blocks = quantize_tensor_q2k_hpc( f32, opt_mode=2, importance=imp, row_width=d0) else: qbytes, n_blocks = quantize_tensor_q2k(f32) fout.write(qbytes) written += len(qbytes) try: deq = dequant_q2k_fast(qbytes, n_blocks) n_cmp = min(n_valid, len(deq)) diff = f32[:n_cmp] - deq[:n_cmp] total_se += float(np.sum(diff ** 2)) except Exception: pass del qbytes elif kind == 'q4': n_el = n_valid n_blocks_q4 = n_el // 32 lib = _load_hexstate_lib() if use_hpc else None q4_hpc = lib is not None and hasattr(lib, 'hexstate_quantize_tensor_q4_0_hpc') if q4_hpc: output_buf = np.zeros(n_blocks_q4 * 18, dtype=np.uint8) error = ctypes.c_float(0.0) f32_c = np.ascontiguousarray(f32, dtype=np.float32) imat_ptr = None if imp is not None: imat_c = np.ascontiguousarray(imp, dtype=np.float32) imat_ptr = imat_c.ctypes.data_as(ctypes.POINTER(ctypes.c_float)) lib.hexstate_quantize_tensor_q4_0_hpc( f32_c.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), ctypes.c_int64(n_el), output_buf.ctypes.data_as(ctypes.c_void_p), ctypes.byref(error), imat_ptr, ctypes.c_int(0), ) fout.write(output_buf.tobytes()) written += output_buf.size total_se += float(error.value) del output_buf, f32_c else: blocks = f32.reshape(-1, 32) amax = np.max(np.abs(blocks), axis=1) d = amax / 7.0 d_safe = np.where(d == 0, 1.0, d) qs = np.clip(np.round(blocks / d_safe[:, None]) + 8, 0, 15).astype(np.uint8) d_fp16 = d.astype(np.float16) out_buf = bytearray(n_blocks_q4 * 18) for b in range(n_blocks_q4): off = b * 18 struct.pack_into(' " " [--keep-metadata] [--imatrix FILE] [--keep-embd] [--q2all]" " [--iq2xs | --iq2s | --iq2auto]") print(" --iq2xs low-bit tensors → IQ2_XS (E8 codebook, 2.3125 bpw) instead of Q2_K") print(" --iq2s low-bit tensors → IQ2_S (E8 codebook, 2.5625 bpw) instead of Q2_K") print(" --iq2auto per tensor: IQ2_S or Q2_K, whichever has lower imatrix-weighted") print(" RMSE on a row sample (IQ2_S wins Gaussian-ish tensors, Q2_K wins") print(" outlier-heavy ones). HEX_AUTO_MARGIN biases toward IQ2_S.") print(" HEX_CHUNK_ELEMS max f32 elements per tensor chunk (default 2000000)") print(" HEX_DC_LAMBDA DC residual weight (default 1)") print(" HEX_VW_LAMBDA vesica weight (default 1)") print(" HEX_DC_DECAY rolling residual carry 0..1 (default 0.85)") print(" HEX_SSE_BUDGET relative SSE the shaper may spend (Q2_K 5e-4, IQ2 2e-2)") print(" HEX_FOLD_DEPTH fold-pyramid levels below DC to shape, 1=vesica only (default 1, 7=full tree)") print(" HEX_FOLD_GAMMA per-level geometric weight of the pyramid (default 2)") print(" HEX_CARRY_CUM 1: carry cumulative row residual (default), 0: previous block only") sys.exit(1) global LOWBIT_FORMAT input_path = sys.argv[1] output_path = sys.argv[2] keep_metadata = '--keep-metadata' in sys.argv quantize_none = '--quantize-none' in sys.argv q2all = '--q2all' in sys.argv keep_embd = '--keep-embd' in sys.argv # keep tied embedding at source precision instead of Q8_0 if '--iq2xs' in sys.argv: LOWBIT_FORMAT = 'iq2xs' elif '--iq2s' in sys.argv: LOWBIT_FORMAT = 'iq2s' elif '--iq2auto' in sys.argv: LOWBIT_FORMAT = 'auto' lowbit_kind = LOWBIT_FORMAT # 'q2k' | 'iq2xs' | 'iq2s' | 'auto' lowbit_auto = lowbit_kind == 'auto' # For 'auto' the per-tensor kind is decided at plan time (lowbit_choice); # file-level labels use IQ2_S since that is the intended majority type. _label_kind = 'iq2s' if lowbit_auto else lowbit_kind lowbit_name = 'IQ2_S/Q2_K' if lowbit_auto else LOWBIT_KINDS[_label_kind][2] lowbit_file_type = LOWBIT_KINDS[_label_kind][3] # LLAMA_FTYPE_MOSTLY_* # Check for imatrix imatrix_data = None for i, arg in enumerate(sys.argv): if arg == '--imatrix' and i + 1 < len(sys.argv): imat_path = sys.argv[i + 1] if os.path.exists(imat_path): imatrix_data = read_imatrix(imat_path) print(f" Loaded imatrix: {len(imatrix_data)} tensors from {imat_path}") else: print(f" WARNING: imatrix file not found: {imat_path}") break # Check for HPC C library use_hpc = _load_hexstate_lib() is not None if lowbit_kind != 'q2k': lib = _load_hexstate_lib() need = 'hexstate_quantize_tensor_iq2_xs_hpc' if lowbit_kind == 'iq2xs' \ else 'hexstate_quantize_tensor_iq2_s_hpc' if lib is None or not hasattr(lib, need): print(f" ERROR: --{'iq2auto' if lowbit_auto else lowbit_kind} needs libhexstate_q2k.so " "with IQ2 support (make -f makefile.quantize.c)") sys.exit(1) print() print(" ╔════════════════════════════════════════════════════════════════╗") print(" ║ HExState GGUF Re-Quantizer ║") print(f" ║ GGUF → {lowbit_name:10s} GGUF with metadata passthrough ║") if lowbit_kind == 'iq2xs': print(" ║ Low-bit: IQ2_XS E8 codebook · fold/DC shaping · 2.3125 bpw ║") elif lowbit_kind == 'iq2s': print(" ║ Low-bit: IQ2_S E8 codebook · fold/DC shaping · 2.5625 bpw ║") elif lowbit_auto: print(" ║ Low-bit: AUTO IQ2_S vs Q2_K per tensor by weighted RMSE ║") if q2all: print(" ║ Mode: --q2all ALL eligible tensors → Q2_K (test mode) ║") if use_hpc and imatrix_data: print(" ║ Engine: HPC + iMatrix (calibrated sensitivity propagation) ║") elif use_hpc: print(" ║ Engine: HPC (BP + MSE Grid + Sensitivity Propagation) ║") else: print(" ║ Engine: Python (numpy vectorized) ║") print(f" ║ Chunk: {_chunk_max_elems():<7d} f32 elems/tensor (HEX_CHUNK_ELEMS) ║") print(" ╚════════════════════════════════════════════════════════════════╝") print() start_time = time.time() file_size = os.path.getsize(input_path) print(f" Input: {input_path}") print(f" Size: {file_size / 1024**3:.2f} GB") print(f" Output: {output_path}") print() with open(input_path, 'rb') as fin: # ── Read Header ── magic = struct.unpack(' 'q2k' | 'iq2xs' | 'iq2s' auto_counts = {} if lowbit_auto: print(" Auto-selecting IQ2_S vs Q2_K per tensor (32-row samples)...") for i, ti in enumerate(tensor_infos): if quant_plan[i]: out_dims = list(ti['dims']) dim0 = out_dims[0] if ti['n_dims'] >= 2 else ti['n_elements'] if quant_plan[i] == 'EMBD_Q8': # Tied embedding / LM head → Q8_0 (8.5 bpw, 34 B / 32 w) out_type = GGML_TYPE_Q8_0 n_blocks = ti['n_elements'] // 32 out_size = n_blocks * 34 print(f" [EMBD→Q8_0·Sieve] {ti['name']} ({ti['n_elements']:,} elements)") elif quant_plan[i] == 'Q4_HPC': # Attention tensor → Q4_0 HPC (4.5 bpw) out_type = GGML_TYPE_Q4_0 n_blocks = (ti['n_elements'] + 31) // 32 out_size = n_blocks * 18 print(f" [Q4_0·HPC] {ti['name']} ({ti['n_elements']} elements)") elif quant_plan[i] == 'Q4_0': out_type = GGML_TYPE_Q4_0 n_blocks = ti['n_elements'] // 32 out_size = n_blocks * 18 print(f" Q4_0: {ti['name']} (dims[0]={dim0})") elif quant_plan[i] is True and q2k_row_compatible(ti['n_dims'], ti['dims']): if lowbit_auto: kind, wr_q, wr_s = auto_select_lowbit( fin, ti, data_section_start + ti['offset'], imatrix_data) auto_counts[kind] = auto_counts.get(kind, 0) + 1 print(f" [AUTO→{LOWBIT_KINDS[kind][2]:6s}] {ti['name'][:48]:48s} " f"wRMSE Q2_K={wr_q:.4f} IQ2_S={wr_s:.4f}") else: kind = lowbit_kind lowbit_choice[i] = kind out_type = LOWBIT_KINDS[kind][0] n_blocks = ti['n_elements'] // QK_K out_size = n_blocks * LOWBIT_KINDS[kind][1] else: out_type = ti['type'] out_size = ti['data_size'] quant_plan[i] = False print(f" Keep: {ti['name']} (dims[0]={dim0})") else: out_type = ti['type'] out_size = ti['data_size'] out_dims = list(ti['dims']) out_tensor_infos.append({ 'name': ti['name'], 'n_dims': ti['n_dims'], 'dims': out_dims, 'type': out_type, 'offset': out_data_offset, 'data_size': out_size, }) out_data_offset += out_size out_data_offset = align_offset(out_data_offset) # ── Update KV pairs ── updated_kv = [] if keep_metadata: print(" --keep-metadata: passing through ALL KV pairs unchanged") updated_kv = list(kv_pairs) else: for key, vtype, raw_value in kv_pairs: if key == 'general.file_type' and vtype == 4: # UINT32 # LLAMA_FTYPE_MOSTLY_Q2_K = 10, MOSTLY_IQ2_XS = 20 updated_kv.append((key, vtype, struct.pack(' # spam). Fix: read the tokens array to find control-looking # tokens, then patch their types to CONTROL (3). # See: https://github.com/ggml-org/llama.cpp/issues/21321 tokens_kv = next((v for k, vt, v in kv_pairs if k == 'tokenizer.ggml.tokens' and vt == 9), None) token_names = [] if tokens_kv: bio = io.BytesIO(tokens_kv) arr_type = struct.unpack(', , , , — sentence markers # - <|turn>, , <|tool_*|> etc — delimiters # NOTE: do NOT mark as CONTROL — Gemma 4 uses # these tokens internally for thinking/channel markers # (e.g. = <|channel>). The llama.cpp parser # handles them via the peg-gemma4 format instead. is_control = False if tname in ('', '', '', '', '', '', ''): is_control = True elif re.match(r'^<\|.*\|?>$', tname) or re.match(r'^<.*\|>$', tname): is_control = True if is_control and ttypes[i] != CONTROL_TYPE: ttypes[i] = CONTROL_TYPE n_fixed += 1 print(f" Fixed {n_fixed} token types to CONTROL (Gemma 4 fix)") # Rebuild the raw value new_raw = struct.pack(' tokens. The fixed template from # llama.cpp PR #21418 pre-fills an empty thought block when # thinking is disabled: <|channel>thought\n # See: https://github.com/ggml-org/llama.cpp/pull/21418 script_dir = os.path.dirname(os.path.abspath(__file__)) workspace_dir = os.path.dirname(script_dir) template_path = os.path.join(workspace_dir, 'llama-cpp-latest', 'models', 'templates', 'google-gemma-4-31B-it.jinja') if os.path.exists(template_path): with open(template_path, 'r') as tf: new_template = tf.read() new_raw = struct.pack(' pos: fout.write(b'\x00' * (aligned - pos)) # ── Write tensor data ── quant_count = 0 total_quant_bytes = 0 total_keep_bytes = 0 total_rmse = 0.0 q2k_rmse_sum = 0.0 q2k_tensor_count = 0 for i, ti in enumerate(tensor_infos): # Progress bar pct = (i + 1) / n_tensors * 100 bar_width = 40 filled = int(bar_width * (i + 1) / n_tensors) bar = '█' * filled + '░' * (bar_width - filled) elapsed = time.time() - start_time eta = elapsed / max(i + 1, 1) * (n_tensors - i - 1) sys.stdout.write(f"\r [{bar}] {pct:5.1f}% ({i+1}/{n_tensors}) {elapsed:.0f}s ETA:{eta:.0f}s {ti['name'][:50]}") sys.stdout.flush() abs_offset = data_section_start + ti['offset'] plan = quant_plan[i] if plan == 'EMBD_Q8': nbytes, rmse, sigma = _stream_quantize( fin, fout, ti, abs_offset, 'q8', imatrix_data, use_hpc) if rmse is not None: print(f"\n [Q8_0·Sieve] {ti['name']} RMSE={rmse:.6e}" f" σ={sigma:.4f} rel={rmse / max(sigma, 1e-30):.4f}") quant_count += 1 total_quant_bytes += nbytes elif plan in ('Q4_0', 'Q4_HPC'): q4_hpc = (plan == 'Q4_HPC' and use_hpc) nbytes, rmse, sigma = _stream_quantize( fin, fout, ti, abs_offset, 'q4', imatrix_data, q4_hpc) tag = 'Q4_0·HPC' if q4_hpc else 'Q4_0' if rmse is not None: print(f"\n [{tag}] {ti['name']} RMSE={rmse:.6e}" f" σ={sigma:.4f} rel={rmse / max(sigma, 1e-30):.4f}") quant_count += 1 total_quant_bytes += nbytes elif plan: kind = lowbit_choice.get(i, 'q2k') kname = LOWBIT_KINDS[kind][2] nbytes, rmse, sigma = _stream_quantize( fin, fout, ti, abs_offset, kind, imatrix_data, use_hpc) if rmse is not None: q2k_rmse_sum += rmse q2k_tensor_count += 1 print(f"\n [{kname}] {ti['name'][:50]} RMSE={rmse:.6e}" f" σ={sigma:.4f} rel={rmse / max(sigma, 1e-30):.4f}") else: print(f"\n [{kname}] {ti['name'][:55]} RMSE=n/a") quant_count += 1 total_quant_bytes += nbytes else: nbytes = _copy_bytes(fin, fout, abs_offset, ti['data_size']) total_keep_bytes += nbytes # Alignment padding pad = align_offset(fout.tell()) - fout.tell() if pad > 0: fout.write(b'\x00' * pad) final_size = fout.tell() elapsed = time.time() - start_time print(f"\r {'█' * 40} 100.0% ({n_tensors}/{n_tensors}) {elapsed:.0f}s" + " " * 60) print() # ── Summary ── original_bytes = sum(ti['data_size'] for ti in tensor_infos) compression = original_bytes / max(final_size, 1) print(" ╔════════════════════════════════════════════════════════════════╗") print(" ║ RE-QUANTIZATION SUMMARY ║") print(" ╠════════════════════════════════════════════════════════════════╣") print(f" ║ Tensors quantized ({lowbit_name[:10]:10s}): {quant_count:<27d} ║") if lowbit_auto: print(f" ║ Auto split: IQ2_S {auto_counts.get('iq2s', 0):<5d} Q2_K {auto_counts.get('q2k', 0):<27d} ║") print(f" ║ Tensors kept as-is: {total_keep:<33d} ║") print(f" ║ {lowbit_name[:10]:10s} data: {total_quant_bytes:>12,} bytes ({total_quant_bytes/1024**2:>7.1f} MB) ║") print(f" ║ Kept data: {total_keep_bytes:>12,} bytes ({total_keep_bytes/1024**2:>7.1f} MB) ║") print(f" ║ Original size: {file_size:>12,} bytes ({file_size/1024**3:>7.2f} GB) ║") print(f" ║ Output size: {final_size:>12,} bytes ({final_size/1024**3:>7.2f} GB) ║") print(f" ║ Compression: {compression:>42.1f}x ║") if q2k_tensor_count > 0: mean_rmse = q2k_rmse_sum / q2k_tensor_count print(f" ║ Mean {lowbit_name[:10]:10s} RMSE: {mean_rmse:>12.6e} ║") print(f" ║ Total time: {elapsed:>39.1f} sec ║") print(" ╚════════════════════════════════════════════════════════════════╝") print() print(f" Output: {output_path}") print() if __name__ == '__main__': main()