import heapq import warnings import numpy as np from functools import lru_cache from typing import Dict, List, Set, Tuple, Any from dataclasses import dataclass, field from constants import ( TIER_ORDER, TIER_BPW, GGUF_OVERHEAD_FACTOR, CLASS_MAX_TIER, CAN_Q3, ALLOW_LOWER_FLOOR, MTP_DEPLOY_TIER, EMBD_DEPLOY_TIER, EMBD_PIN_TYPES, ROUTER_PIN_TYPES, TOXICITY_SUB4, F32_BPW, FORCE_F32_TYPES, get_tensor_class, get_tensor_type, is_mtp_tensor, ) # Выносим делитель в константу (8 бит * 1024 байт * 1024 кбайт) BITS_IN_MIB = 8 * 1024 * 1024.0 # Filled in by the last classify() call so main.py can report the tail of the # budget without threading a third return value through everything. _LAST_RUN_INFO: Dict[str, float] = {"slack_mib": 0.0, "polish_steps": 0} # Предвычисляем множители размеров для каждого тира (ускорение математики) TIER_SIZE_MULTIPLIER = { tier: (bpw / BITS_IN_MIB) * GGUF_OVERHEAD_FACTOR for tier, bpw in TIER_BPW.items() } # K-quants используют блоки по 256 элементов и требуют выравнивания (padding) 3D-тензоров MoE K_QUANTS = {"Q3_K", "Q4_K", "Q5_K", "Q6_K"} MOE_PAD_TYPES = {"ffn_gate_exps", "ffn_up_exps", "ffn_down_exps", "ffn_down"} MSE_BPW = { "IQ1_S": 1.5625, "IQ2_XXS": 2.0625, "IQ2_XS": 2.3125, "IQ2_S": 2.5, "IQ3_XXS": 3.0625, "Q3_K": 3.4375, "IQ3_S": 3.44, # IQ4_NL gets +0.15 effective bpw over its real 4.5bpw twin IQ4_XS: # K-quant bonus. Must stay strictly between IQ4_XS and Q4_K, otherwise # the IQ4_XS<->IQ4_NL steps become zero-gain and strand both greedy # chains (upgrades refuse paid zero-gain, downgrades stall). "IQ4_XS": 4.25, "IQ4_NL": 4.40, "Q4_K": 4.50, "Q5_K": 5.50, "Q6_K": 6.5625, "Q8_0": 8.50, "F16": 16.0, } @dataclass(order=True, slots=True) class UpgradeItem: """Элемент очереди апгрейдов. Сравнивается только по neg_utility.""" neg_utility: float group_id: int = field(compare=False) next_tier: str = field(compare=False) cost_delta: float = field(compare=False) def _tier_index(tier: str, order: list | None = None) -> int: order = order or TIER_ORDER if tier not in order: raise ValueError(f"Unknown tier: {tier}") return order.index(tier) def _tier_at(idx: int, order: list | None = None) -> str: order = order or TIER_ORDER if not (0 <= idx < len(order)): raise IndexError(f"Tier index {idx} out of range") return order[idx] # SQUEEZE mode: binary allocation — every tensor is either dungeon or # palace, nothing in between. Separate path, other modes untouched. def _size_mib(tier: str, n_elements: int, force_f32: bool = False) -> float: """Размер тензора в MiB с учётом оверхеда GGUF. 1D tensors (norms, biases) are F32 in the file whatever the tier says — llama.cpp does not quantize 1D. Accounting them at the assigned tier both under-counts the budget and hides the fact that the assignment is a no-op, so they are priced at F32 bits instead. """ if n_elements <= 0: return 0.0 if force_f32: return n_elements * (F32_BPW / BITS_IN_MIB) * GGUF_OVERHEAD_FACTOR return n_elements * TIER_SIZE_MULTIPLIER.get(tier, 0.0) def _f32_map(model_tensors: Dict[str, Any]) -> Dict[str, bool]: """Tensors the binary will write as F32 whatever tier we assign. Two proven sources: 1D tensors (norms, biases — llama.cpp does not quantize rank-1), and the types in FORCE_F32_TYPES (ssm_conv1d, caught by preflight on RINIQ-M2). Anything assigned here is pinned outside the budget: keeping it in the queue only produces decisions that never land. """ out = {} for k, v in model_tensors.items(): shape = v.get("shape") or [] out[k] = bool(v.get("rank1", len(shape) <= 1)) or \ get_tensor_type(k) in FORCE_F32_TYPES return out @lru_cache(maxsize=64) def _mse_eff(tier: str) -> float: """Effective MSE with sub-4-bit toxicity penalty (see TOXICITY_SUB4).""" base = 2 ** (-2 * MSE_BPW[tier]) if TIER_BPW[tier] < 4.0: base *= TOXICITY_SUB4 return base @lru_cache(maxsize=128) def _mse_delta(cur_tier: str, next_tier: str) -> float: return _mse_eff(cur_tier) - _mse_eff(next_tier) # ---------------------------------------------------------------- battlefield # Experimental hybrid utility: MSE is not the only lens. Modes: # mse — baseline (current behavior, uopt ignored) # rmse — sqrt scale: compresses dynamic range, cheap small steps rank higher # hybrid — ΔMSE, discounted for concentrated groups (bits wasted on junk # members) and boosted for large worst-case-error drops. # Concentration rationale: group timp [100,0,0] vs [34,33,33], same sum and # cost — the first wastes 2/3 of the upgrade on zero-importance members, so # its gain is divided by (1 + cv_w * cv^2). def _gain(cur_tier: str, next_tier: str, g_names: List[str], tensor_importance: Dict[str, float], uopt: Dict[str, Any]) -> float: """Quality gain of cur->next step under the selected utility metric.""" import math mode = (uopt or {}).get("mode", "mse") if mode == "mix": # Per-group utility: kings (top importance rank) evaluated by MSE # (spread/protect), everyone else by mix_base (default: sacrifice). # mix_top is a set of group rep names, built once by the caller. # SCALE WARNING (scar 2026-09-18): raw formulas live on different # scales (mse Δ ~1e-3, smape Δ ~1) — compared raw, smape junk # outbids mse kings ~800:1 and mix collapses to pure smape. # _scale() below normalizes every formula to O(1) first. top = (uopt or {}).get("mix_top") or frozenset() rep = g_names[0] if g_names else "" mode = "mse" if rep in top else (uopt or {}).get("mix_base", "smape") return _scale(mode, cur_tier, next_tier, uopt or {}) * _gain_mode( mode, cur_tier, next_tier, g_names, tensor_importance, uopt or {}) # non-mix: raw formula, behavior unchanged (zoo baselines intact) return _gain_mode(mode, cur_tier, next_tier, g_names, tensor_importance, uopt or {}) _scale_cache: Dict[str, float] = {} def _scale(mode: str, cur_tier: str, next_tier: str, uopt: Dict[str, Any]) -> float: """Normalize a utility formula to O(1). Reference = |raw Q4_K→Q5_K delta| for that mode (cached). Without this, bounded formulas (smape ≤ 2) outbid absolute ones (mse Δ ~1e-3) ~800:1 and any mix collapses to the bounded side. cur/next args kept for future per-step references; currently unused. """ if mode not in _scale_cache: ref = abs(_gain_mode(mode, "Q4_K", "Q5_K", [], {}, uopt)) _scale_cache[mode] = 1.0 / ref if ref > 0 else 1.0 return _scale_cache[mode] def _gain_mode(mode: str, cur_tier: str, next_tier: str, g_names: List[str], tensor_importance: Dict[str, float], uopt: Dict[str, Any]) -> float: """Single-formula gain (mode already resolved, never 'mix').""" import math assert mode != "mix", "mix must be resolved before _gain_mode" from experimental import experimental_gain, experimental_gain_modes if mode == "rmse": return math.sqrt(_mse_eff(cur_tier)) - math.sqrt(_mse_eff(next_tier)) if mode == "logcosh": # Gain reshaper on the MSE delta: linear-ish for small steps, # compressed for large ones (outlier-heavy steps stop dominating). d = _mse_delta(cur_tier, next_tier) s = (uopt or {}).get("huber_delta", 3e-4) # splits the typical Δ range if d < 0: return d return math.log(math.cosh(d / s)) * s if mode in experimental_gain_modes(): return experimental_gain(cur_tier, next_tier, g_names, tensor_importance, uopt) if mode == "smape": # Relative lens: symmetric relative error drop, bounded in [0, 2]. # Favors steps that slash the *remaining* error proportionally, # regardless of absolute scale. ec, en = _mse_eff(cur_tier), _mse_eff(next_tier) if ec + en <= 0: return 0.0 return 2.0 * (ec - en) / (ec + en) if mode in ("smse", "balance"): # SMSE (ex-BALANCE, renamed 2026-09-23: SMAPE+MSE blend) — # 'balance' kept as a deprecated alias. # geometric blend of the relative lens (SMAPE barbell shape) and # the absolute lens (MSE weight + live sub-4 toxicity). Both halves # O(1)-normalized first (MIX lesson), so neither outbids 800:1. # Toxicity survives here because the MSE half is divided by a # CONSTANT reference — unlike SMAPE's (ec+en) denominator, which # cancels it. TOX multipliers reverted (dead code, wrong direction). ec, en = _mse_eff(cur_tier), _mse_eff(next_tier) if ec + en <= 0: return 0.0 rel = 2.0 * (ec - en) / (ec + en) aba = ec - en if rel <= 0.0 or aba <= 0.0: return 0.0 s_ref = _scale("smape", cur_tier, next_tier, uopt or {}) m_ref = _scale("mse", cur_tier, next_tier, uopt or {}) if s_ref <= 0.0 or m_ref <= 0.0: return 0.0 return math.sqrt((rel / s_ref) * (aba / m_ref)) if mode == "srmse": # SMAPE x RMSE blend (2026-09-23, proposed as "what if"): relative # lens x COMPRESSED absolute (sqrt). sqrt compresses the toxicity # multiplier (x1.41 instead of x2.0), so this carries LESS toxicity # than SMSE — expect a gentler barbell: more penthouses, weaker # lava rescue. Verdict duel queued (MiMo-5100-SRMSE vs SMSE). ec, en = _mse_eff(cur_tier), _mse_eff(next_tier) if ec + en <= 0: return 0.0 rel = 2.0 * (ec - en) / (ec + en) rba = math.sqrt(ec) - math.sqrt(en) if rel <= 0.0 or rba <= 0.0: return 0.0 s_ref = _scale("smape", cur_tier, next_tier, uopt or {}) r_ref = _scale("rmse", cur_tier, next_tier, uopt or {}) if s_ref <= 0.0 or r_ref <= 0.0: return 0.0 return math.sqrt((rel / s_ref) * (rba / r_ref)) if mode == "ssim": # Measured structural gain, per group MEMBER (not rep): members of a # tied group share importance but not weights, so ΔSSIM differs per # member — a genuine per-group signal. No toxicity fiction: SSIM # measures the sub-4 collapse directly (that's the hypothesis). return _ssim_gain(cur_tier, next_tier, g_names, tensor_importance, (uopt or {}).get("ssim_table"), relative=False) if mode == "smape_ssim": # Relative lens on the measured signal: symmetric relative structural # recovery per member, importance-weighted. The zoo final boss. return _ssim_gain(cur_tier, next_tier, g_names, tensor_importance, (uopt or {}).get("ssim_table"), relative=True) if mode == "smape_frag": # Don't replace the champion — modulate it. SMAPE-on-MSE decides the # shape; per-group structural fragility (measured Q4 damage, mean over # members, normalized) reweights the vote. Fragile groups get boosted # upgrades, oak groups get discounted ones. ec, en = _mse_eff(cur_tier), _mse_eff(next_tier) base = 0.0 if ec + en <= 0 else 2.0 * (ec - en) / (ec + en) frag = _group_fragility(g_names, (uopt or {}).get("ssim_table")) return base * (1.0 + (uopt or {}).get("frag_w", 0.5) * (frag - 1.0)) gain = _mse_delta(cur_tier, next_tier) return gain def _ssim_gain(cur_tier: str, next_tier: str, g_names: List[str], tensor_importance: Dict[str, float], table_path: str | None, relative: bool = False) -> float: """Importance-weighted mean ΔSSIM over group members. Returns an average (not a sum): the caller multiplies by Σtimp, so the result is Σt·ΔS — members with more importance contribute more gain. Tiers outside the measured table are bpw-interpolated on the measured curve; tensors missing from the table contribute 0 (pinned anyway). """ from constants import TIER_BPW table = _ssim_table(table_path) if not table: return _mse_delta(cur_tier, next_tier) # table not ready: mse fallback def s_of(name: str, tier: str) -> float | None: row = table.get(name) if row is None: return None if tier in row: return row[tier] if tier == "F16": return 1.0 # bpw-interpolate on the measured curve b = TIER_BPW.get(tier) pts = sorted((TIER_BPW[t], s) for t, s in row.items() if t in TIER_BPW) if b is None or not pts: return None if b <= pts[0][0]: b0, s0 = pts[0] b1, s1 = pts[1] if len(pts) > 1 else (b0 + 1.0, 1.0) s = s0 + (s1 - s0) * (b - b0) / (b1 - b0) return max(0.0, min(1.0, s)) for (b0, s0), (b1, s1) in zip(pts, pts[1:]): if b0 <= b <= b1: f = (b - b0) / (b1 - b0) if b1 > b0 else 0.0 return s0 + (s1 - s0) * f return pts[-1][1] # above measured range: clamp to best num, den = 0.0, 0.0 for n in g_names: t = tensor_importance.get(n, 0.0) sc, sn = s_of(n, cur_tier), s_of(n, next_tier) if sc is None or sn is None: continue if relative: # Relative lens on DAMAGE (1-S), not similarity: damage spans an # order of magnitude (vs similarity stuck in [0.987, 1]), so the # relative rescaling actually bites — same trick that made SMAPE # win on MSE. Positive when damage drops. dc, dn = 1.0 - sc, 1.0 - sn g = 0.0 if dc + dn <= 0 else 2.0 * (dc - dn) / (dc + dn) else: g = sn - sc num += t * g den += t return num / den if den > 0 else 0.0 _SSIM_CACHE: Dict[str, Dict[str, Dict[str, float]]] = {} _FRAG_CACHE: Dict[str, float] = {} def _group_fragility(g_names: List[str], table_path: str | None) -> float: """Mean measured Q4 damage over group members, normalized by global mean. Returns 1.0 when the table is missing (modulation becomes a no-op). """ table = _ssim_table(table_path) if not table: return 1.0 key = table_path + "|mean" if key not in _FRAG_CACHE: vals = [1.0 - row["Q4_K"] for row in table.values() if "Q4_K" in row] _FRAG_CACHE[key] = sum(vals) / len(vals) if vals else 1.0 gmean = _FRAG_CACHE[key] ds = [1.0 - table[n]["Q4_K"] for n in g_names if n in table and "Q4_K" in table[n]] if not ds or gmean <= 0: return 1.0 return (sum(ds) / len(ds)) / gmean def _ssim_table(path: str | None) -> Dict[str, Dict[str, float]]: if not path or path in _SSIM_CACHE: return _SSIM_CACHE.get(path, {}) try: z = np.load(path, allow_pickle=False) except Exception: _SSIM_CACHE[path] = {} return {} tiers = ["Q4_K", "Q5_K", "Q6_K", "Q8_0"] table = {} for name in z.files: arr = z[name] table[name] = {t: float(arr[i][0]) for i, t in enumerate(tiers)} _SSIM_CACHE[path] = table return table def _step_cost(cur_tier: str, next_tier: str, g_elements: int, g_elements_padded: int) -> float: """MiB delta of one ladder step, honouring K-quant padding.""" cur_n = g_elements_padded if cur_tier in K_QUANTS else g_elements next_n = g_elements_padded if next_tier in K_QUANTS else g_elements return _size_mib(next_tier, next_n) - _size_mib(cur_tier, cur_n) def _best_step_within(group_id, group_registry, assignments, tensor_importance, importance_table, slack, ceiling=None, uopt=None, order=None): """Best ladder step this group can take with `slack` MiB left. The greedy drain only ever offers cur->cur+1, and a step that does not fit is dropped from the queue for good (the budget only shrinks from there). So the tail of the budget goes unspent: a group whose next rung is 400 MiB stays at its floor with 300 MiB idle, even though a higher rung might cost less than the near one. This walks the whole reachable window and returns the best value-per-MiB step inside it. """ order = order or TIER_ORDER g_names, g_elements, g_elements_padded, g_f32 = group_registry[group_id] if g_f32 or slack <= 0: return None rep_name = g_names[0] cur_tier = assignments[rep_name] cur_idx = _tier_index(cur_tier, order) rep_info = importance_table.get(rep_name, {}) ttype = rep_info["type"] if "type" in rep_info else get_tensor_type(rep_name) max_tier = ceiling or CLASS_MAX_TIER.get(get_tensor_class(ttype), "Q8_0") top_idx = min(_tier_index(max_tier, order), len(order) - 1) if cur_idx >= top_idx: return None total_g_imp = sum(tensor_importance.get(n, 0) for n in g_names) best = None idx = cur_idx + 1 while idx <= top_idx: cand = _tier_at(idx, order) cost = _step_cost(cur_tier, cand, g_elements, g_elements_padded) if cost > slack: idx += 1 continue gain = _gain(cur_tier, cand, g_names, tensor_importance, uopt or {}) if gain < 0: idx += 1 continue if cost == 0: util = float("inf") elif gain == 0: idx += 1 continue else: util = (total_g_imp * gain) / cost if best is None or util > best[0]: best = (util, cand, cost) idx += 1 return best def _polish_slack(assignments, group_registry, current_size, effective_target, tensor_importance, importance_table, uopt=None, order=None, ceiling_of=None): """Spend the tail of the budget that the single-rung greedy left idle. Returns (current_size, steps_taken). Terminates when no group can use the remaining slack. Bounded: each iteration raises some group's tier, and tiers only move up, so it cannot cycle. """ order = order or TIER_ORDER taken = 0 while True: slack = effective_target - current_size if slack <= 0: break best = None for g_id in group_registry: cand = _best_step_within( g_id, group_registry, assignments, tensor_importance, importance_table, slack, ceiling=(ceiling_of(g_id) if ceiling_of else None), uopt=uopt, order=order) if cand is None: continue util, tier, cost = cand if best is None or util > best[0]: best = (util, g_id, tier, cost) if best is None: break _, g_id, tier, cost = best for n in group_registry[g_id][0]: assignments[n] = tier current_size += cost taken += 1 # main.py reports this: leftover slack that no rung can absorb is a # property of the geometry (the next step is too big), not a silent bug. _LAST_RUN_INFO["slack_mib"] = max(0.0, effective_target - current_size) _LAST_RUN_INFO["polish_steps"] = taken return current_size, taken def _push_upgrade(group_id: int, group_registry: Dict[int, Tuple[List[str], int, int]], assignments: Dict[str, str], tensor_importance: Dict[str, float], upgrade_queue: List[UpgradeItem], importance_table: Dict[str, Any], ceiling: str | None = None, uopt: Dict[str, Any] | None = None, order: list | None = None): order = order or TIER_ORDER # Храним 4 элемента: имена, реальный размер, размер с padding, признак 1D g_names, g_elements, g_elements_padded, g_f32 = group_registry[group_id] if g_f32: return # F32 in the file whatever we assign — no decision to make rep_name = g_names[0] cur_tier = assignments[rep_name] cur_idx = _tier_index(cur_tier, order) rep_info = importance_table.get(rep_name, {}) ttype = rep_info["type"] if "type" in rep_info else get_tensor_type(rep_name) cls = get_tensor_class(ttype) max_tier = ceiling or CLASS_MAX_TIER.get(cls, "Q8_0") if cur_idx >= _tier_index(max_tier, order) or cur_idx >= len(order) - 1: return next_tier = _tier_at(cur_idx + 1, order) # Выбираем размер в зависимости от того, относится ли тир к K-quants cur_size_g = g_elements_padded if cur_tier in K_QUANTS else g_elements next_size_g = g_elements_padded if next_tier in K_QUANTS else g_elements cost_delta = _size_mib(next_tier, next_size_g) - _size_mib(cur_tier, cur_size_g) quality_delta = _gain(cur_tier, next_tier, g_names, tensor_importance, uopt or {}) if quality_delta < 0: return if cost_delta < 0: return if cost_delta == 0: # Free step: genuine free upgrade (IQ4_NL→Q4_K, same real bpw) or # zero-cost transit across the IQ4_NL/IQ4_XS fiction boundary. utility_per_mb = float('inf') else: if quality_delta == 0: return # paying real MB for zero modeled gain total_g_imp = sum(tensor_importance.get(n, 0) for n in g_names) utility_per_mb = (total_g_imp * quality_delta) / cost_delta heapq.heappush(upgrade_queue, UpgradeItem(-utility_per_mb, group_id, next_tier, cost_delta)) def _base_floor(ttype: str, cls: str, allow_q3: bool, has_imatrix: bool, is_qat: bool = False) -> str: """Minimum tier for a tensor: bottom-up starts here, top-down stops here.""" if allow_q3 and (ttype in CAN_Q3 or cls in CAN_Q3) and has_imatrix: return ALLOW_LOWER_FLOOR if is_qat: return "Q4_K" if cls == "attn_proj" else "IQ4_XS" return "Q4_K" def compute_initial_assignments(non_mtp_names: Set[str], mtp_names: Set[str], importance_table: Dict, allow_q3: bool, is_qat: bool = False, floor: str | None = None, f32_map: Dict[str, bool] | None = None) -> Dict[str, str]: f32_map = f32_map or {} assignments = {name: MTP_DEPLOY_TIER for name in mtp_names} for name in non_mtp_names: rep_info = importance_table.get(name, {}) has_imatrix = "importance_mean" in rep_info ttype = rep_info["type"] if "type" in rep_info else get_tensor_type(name) cls = get_tensor_class(ttype) # output/token_embd are pinned in optimal_classify (never reach here # via non_mtp_names) — kept out of the upgrade budget entirely. if f32_map.get(name, False): # norms/biases: llama.cpp writes 1D as F32 regardless of the # rules. Assigning anything else is a decision that never lands # (preflight caught 75 of them doing exactly that on MiniCPM5-2B). assignments[name] = "F32" elif cls in ("norms", "ssm_params"): assignments[name] = "F16" else: assignments[name] = floor or _base_floor(ttype, cls, allow_q3, has_imatrix, is_qat) return assignments def build_groups(tied_groups: List[List[str]], non_mtp_names: Set[str], ne_map: Dict[str, int], padded_ne_map: Dict[str, int], f32_map: Dict[str, bool] | None = None) -> Dict[int, Tuple[List[str], int, int, bool]]: f32_map = f32_map or {} group_registry = {} assigned_tensors = set() for group_idx, group in enumerate(tied_groups): clean_group = [n for n in group if n in non_mtp_names] if clean_group: g_elements = sum(ne_map.get(n, 0) for n in clean_group) g_elements_padded = sum(padded_ne_map.get(n, 0) for n in clean_group) g_f32 = all(f32_map.get(n, False) for n in clean_group) group_registry[group_idx] = (clean_group, g_elements, g_elements_padded, g_f32) assigned_tensors.update(clean_group) unassigned_tensors = non_mtp_names - assigned_tensors next_group_idx = len(group_registry) for name in unassigned_tensors: group_registry[next_group_idx] = ([name], ne_map.get(name, 0), padded_ne_map.get(name, 0), f32_map.get(name, False)) next_group_idx += 1 return group_registry def optimal_classify(importance_table: dict, tied_groups: list, model: dict, target_size_mib: float, allow_q3: bool = False, free_pins: bool = False, uopt: Dict[str, Any] | None = None, relief: Dict[str, float] | None = None, relief_thr: float = -0.5, allowed_tiers: list | None = None, legacy_1d: bool = False, ceiling: str | None = None) -> Tuple[dict, dict]: """relief: {sweep_unit_tag: Q5->Q4 damage}. Groups with damage below relief_thr get ceiling Q4 (their measured sweet spot): the queue never upgrades them above Q4, and the saved budget flows to other groups. ceiling: global per-group ceiling override (e.g. F16) used when no relief/squeeze ceiling applies — runtime override for CLASS_MAX_TIER, constants.py untouched. allowed_tiers: restrict the ladder (SQUEEZE mode, e.g. ["IQ1_S", "F16"] = dungeon or palace, nothing between). Floor/ceiling follow the list; relief ceilings are ignored in squeeze mode.""" if target_size_mib <= 0: raise ValueError("target_size_mib must be positive") uopt = uopt or {"mode": "mse"} order = allowed_tiers or TIER_ORDER squeeze = allowed_tiers is not None features = model.get("features", {}) has_mtp = features.get("has_mtp", False) and not free_pins n_layers = features.get("n_layers", 31) is_qat = features.get("is_qat", False) model_tensors = model.get("tensors", {}) ne_map = {k: v["n_elements"] for k, v in model_tensors.items()} for tname, info in importance_table.items(): if tname not in ne_map: ne_map[tname] = info["n_elements"] # 1D tensors (norms, biases) are F32 in the file no matter what tier the # rules ask for. They are pinned and priced at F32 so the budget matches # the artifact and the queue stops spending decisions on them. f32_map = {} if legacy_1d else _f32_map(model_tensors) # --- Вычисление MoE Padding (Целочисленное выравнивание) --- moe_d_ff = features.get("moe_intermediate_size", 0) padded_ne_map = dict(ne_map) if moe_d_ff > 0 and moe_d_ff % 256 != 0: aligned_d_ff = ((moe_d_ff + 255) // 256) * 256 for name, n_el in ne_map.items(): ttype = importance_table.get(name, {}).get("type", get_tensor_type(name)) if ttype in MOE_PAD_TYPES: padded_ne_map[name] = (n_el // moe_d_ff) * aligned_d_ff # ----------------------------------------------------------- all_names = set(ne_map.keys()) mtp_names = {n for n in all_names if is_mtp_tensor(n, n_layers)} if has_mtp else set() non_mtp_names = all_names - mtp_names # Pinned output/token_embd: fixed tier, excluded from budget and upgrades # (llama.cpp writes them at the output/token types unconditionally). # --free-pins: experiment — they join the normal pool instead. embd_names = { n for n in non_mtp_names if importance_table.get(n, {}).get("type", get_tensor_type(n)) in EMBD_PIN_TYPES } if not free_pins else set() non_mtp_names -= embd_names # Pinned MoE routers: always F16, outside the budget. router_names = { n for n in non_mtp_names if importance_table.get(n, {}).get("type", get_tensor_type(n)) in ROUTER_PIN_TYPES } if not free_pins else set() non_mtp_names -= router_names tensor_importance = {} for name in non_mtp_names: raw_imp = importance_table.get(name, {}).get("importance_mean", 0.0) tensor_importance[name] = raw_imp assignments = compute_initial_assignments(non_mtp_names, mtp_names, importance_table, allow_q3, is_qat, floor=(order[0] if squeeze else None), f32_map=f32_map) for n in embd_names: assignments[n] = EMBD_DEPLOY_TIER for n in router_names: assignments[n] = "F16" group_registry = build_groups(tied_groups, non_mtp_names, ne_map, padded_ne_map, f32_map=f32_map) # Relief ceilings: measured sweet spots from the damage sweep. def _unit_tag(unit): return "+".join(n.replace(".weight", "").split(".")[-1] + "@" + n.split(".")[1] for n in unit if n.startswith("blk.")) or "global" relief_ceiling: Dict[int, str] = {} if relief: for g_id, (g_names, _, _, _) in group_registry.items(): d = relief.get(_unit_tag(g_names)) if d is not None and d < relief_thr: relief_ceiling[g_id] = "Q4_K" mtp_cost = sum(_size_mib(MTP_DEPLOY_TIER, ne_map.get(n, 0), f32_map.get(n, False)) for n in mtp_names) embd_cost = sum(_size_mib(EMBD_DEPLOY_TIER, ne_map.get(n, 0), f32_map.get(n, False)) for n in embd_names) router_cost = sum(_size_mib("F16", ne_map.get(n, 0), f32_map.get(n, False)) for n in router_names) effective_target = target_size_mib - mtp_cost - embd_cost - router_cost current_size = sum( _size_mib(assignments[n], padded_ne_map.get(n, ne_map.get(n, 0)) if assignments[n] in K_QUANTS else ne_map.get(n, 0), f32_map.get(n, False)) for n in non_mtp_names ) if current_size > effective_target: warnings.warn(f"Initial size {current_size:.1f} MiB already exceeds target {effective_target:.1f} MiB", RuntimeWarning) upgrade_queue = [] for g_id in group_registry: _push_upgrade(g_id, group_registry, assignments, tensor_importance, upgrade_queue, importance_table, ceiling=(order[-1] if squeeze else (relief_ceiling.get(g_id) or ceiling)), uopt=uopt, order=order) while upgrade_queue: item = heapq.heappop(upgrade_queue) g_id, next_tier, cost_delta = item.group_id, item.next_tier, item.cost_delta if cost_delta > 0 and current_size + cost_delta > effective_target: continue for n in group_registry[item.group_id][0]: assignments[n] = next_tier current_size += cost_delta _push_upgrade(item.group_id, group_registry, assignments, tensor_importance, upgrade_queue, importance_table, ceiling=(order[-1] if squeeze else (relief_ceiling.get(item.group_id) or ceiling)), uopt=uopt, order=order) # Tail of the budget: the drain only offers cur->cur+1, so whatever is # left when the cheapest single rungs stop fitting stays unspent. if not squeeze: current_size, polished = _polish_slack( assignments, group_registry, current_size, effective_target, tensor_importance, importance_table, uopt=uopt, order=order, ceiling_of=lambda g_id: (relief_ceiling.get(g_id) or ceiling)) return assignments, padded_ne_map # ---------------------------------------------------------------- top-down # Crazy mode: everything quantizable starts at F16 (norms included) and is # greedily downgraded — cheapest quality-loss-per-MB first — until under target. @dataclass(order=True, slots=True) class DowngradeItem: """Downgrade candidate. Min-heap on loss_per_mb (cheapest loss first).""" loss_per_mb: float group_id: int = field(compare=False) next_tier: str = field(compare=False) saved: float = field(compare=False) def _push_downgrade(group_id: int, group_registry: Dict[int, Tuple[List[str], int, int]], group_floors: Dict[int, str], assignments: Dict[str, str], tensor_importance: Dict[str, float], downgrade_queue: List[DowngradeItem], uopt: Dict[str, Any] | None = None): g_names, g_elements, g_elements_padded, g_f32 = group_registry[group_id] if g_f32: return # F32 in the file whatever we assign — no decision to make rep_name = g_names[0] cur_tier = assignments[rep_name] cur_idx = _tier_index(cur_tier) floor_idx = _tier_index(group_floors[group_id]) if cur_idx <= floor_idx: return next_tier = _tier_at(cur_idx - 1) cur_size_g = g_elements_padded if cur_tier in K_QUANTS else g_elements next_size_g = g_elements_padded if next_tier in K_QUANTS else g_elements saved = _size_mib(cur_tier, cur_size_g) - _size_mib(next_tier, next_size_g) if saved < 0: return # padding quirk: downgrade would grow — stall here loss = _gain(next_tier, cur_tier, g_names, tensor_importance, uopt or {}) # quality increase reversed if loss < 0: return total_g_imp = sum(tensor_importance.get(n, 0) for n in g_names) if saved == 0: # Transit step (e.g. Q4_K→IQ4_NL: same bpw, must pass through to reach # deeper tiers). No saving, so defer to the end with inf utility — # it only fires if real savings elsewhere weren't enough. utility = float("inf") elif loss == 0: # Free savings under the effective-MSE fiction (IQ4_NL→IQ4_XS): # take first. utility = 0.0 else: utility = (total_g_imp * loss) / saved heapq.heappush(downgrade_queue, DowngradeItem(utility, group_id, next_tier, saved)) def optimal_classify_topdown(importance_table: dict, tied_groups: list, model: dict, target_size_mib: float, allow_q3: bool = False, free_pins: bool = False, uopt: Dict[str, Any] | None = None, pin_norms: bool = False, legacy_1d: bool = False) -> Tuple[dict, dict]: """Top-down classification: start at F16, downgrade to fit. pin_norms: keep norms/small tensors at F16 outside the budget (the native norms shield — post-hoc forcing disrupts the greedy path). """ if target_size_mib <= 0: raise ValueError("target_size_mib must be positive") uopt = uopt or {"mode": "mse"} if target_size_mib <= 0: raise ValueError("target_size_mib must be positive") features = model.get("features", {}) has_mtp = features.get("has_mtp", False) and not free_pins n_layers = features.get("n_layers", 31) is_qat = features.get("is_qat", False) model_tensors = model.get("tensors", {}) ne_map = {k: v["n_elements"] for k, v in model_tensors.items()} for tname, info in importance_table.items(): if tname not in ne_map: ne_map[tname] = info["n_elements"] # 1D tensors are F32 in the file whatever the tier says. They are pinned # here (outside the budget) and priced at F32 — top-down used to feed # them into the downgrade queue, "freeing" megabytes that the binary # never gave back, and generating rules that cannot take effect. f32_map = {} if legacy_1d else _f32_map(model_tensors) # --- MoE padding (same as bottom-up) --- moe_d_ff = features.get("moe_intermediate_size", 0) padded_ne_map = dict(ne_map) if moe_d_ff > 0 and moe_d_ff % 256 != 0: aligned_d_ff = ((moe_d_ff + 255) // 256) * 256 for name, n_el in ne_map.items(): ttype = importance_table.get(name, {}).get("type", get_tensor_type(name)) if ttype in MOE_PAD_TYPES: padded_ne_map[name] = (n_el // moe_d_ff) * aligned_d_ff # -------------------------------------- all_names = set(ne_map.keys()) mtp_names = {n for n in all_names if is_mtp_tensor(n, n_layers)} if has_mtp else set() rest = all_names - mtp_names embd_names = { n for n in rest if importance_table.get(n, {}).get("type", get_tensor_type(n)) in EMBD_PIN_TYPES } if not free_pins else set() # No pinning here: every other 2D tensor starts at F16 and fights for # budget — except MoE routers (F16) and 1D tensors (F32, physically: # llama.cpp does not quantize 1D, verified via --dry-run). flex_names = rest - embd_names router_names = { n for n in flex_names if importance_table.get(n, {}).get("type", get_tensor_type(n)) in ROUTER_PIN_TYPES } if not free_pins else set() flex_names -= router_names one_d_names = {n for n in flex_names if f32_map.get(n, False)} flex_names -= one_d_names # Native norms shield: norms/small tensors stay F16 outside the budget # (post-hoc forcing was proven to disrupt the greedy path — TDN-SMAPE). norm_names = { n for n in flex_names if "norm" in n or ne_map.get(n, 10 ** 9) < 100000 } if pin_norms else set() flex_names -= norm_names tensor_importance = {} for name in flex_names: tensor_importance[name] = importance_table.get(name, {}).get("importance_mean", 0.0) assignments = {n: MTP_DEPLOY_TIER for n in mtp_names} for n in embd_names: assignments[n] = EMBD_DEPLOY_TIER for n in router_names: assignments[n] = "F16" for n in one_d_names: assignments[n] = "F32" for n in norm_names: assignments[n] = "F16" for n in flex_names: assignments[n] = "F16" # Groups over flex tensors only, with per-group floors group_registry = build_groups(tied_groups, flex_names, ne_map, padded_ne_map, f32_map=f32_map) group_floors = {} for g_id, (g_names, _, _, _) in group_registry.items(): floors = [] for n in g_names: rep_info = importance_table.get(n, {}) ttype = rep_info["type"] if "type" in rep_info else get_tensor_type(n) cls = get_tensor_class(ttype) floors.append(_tier_index(_base_floor( ttype, cls, allow_q3, "importance_mean" in rep_info, is_qat))) group_floors[g_id] = _tier_at(min(floors)) fixed_cost = ( sum(_size_mib(MTP_DEPLOY_TIER, ne_map.get(n, 0), f32_map.get(n, False)) for n in mtp_names) + sum(_size_mib(EMBD_DEPLOY_TIER, ne_map.get(n, 0), f32_map.get(n, False)) for n in embd_names) + sum(_size_mib("F16", ne_map.get(n, 0), f32_map.get(n, False)) for n in router_names) + sum(_size_mib("F32", ne_map.get(n, 0), True) for n in one_d_names) + sum(_size_mib("F16", ne_map.get(n, 0), f32_map.get(n, False)) for n in norm_names) ) effective_target = target_size_mib - fixed_cost current_size = sum(_size_mib("F16", ne_map.get(n, 0)) for n in flex_names) downgrade_queue: List[DowngradeItem] = [] for g_id in group_registry: _push_downgrade(g_id, group_registry, group_floors, assignments, tensor_importance, downgrade_queue, uopt=uopt) while current_size > effective_target and downgrade_queue: item = heapq.heappop(downgrade_queue) for n in group_registry[item.group_id][0]: assignments[n] = item.next_tier current_size -= item.saved _push_downgrade(item.group_id, group_registry, group_floors, assignments, tensor_importance, downgrade_queue, uopt=uopt) if current_size > effective_target: warnings.warn(f"Top-down hit all floors at {current_size:.1f} MiB, " f"still over target {effective_target:.1f} MiB", RuntimeWarning) return assignments, padded_ne_map # ---- Phase 2: spend leftover slack on mirrored upgrades ---- # Same imp×ΔMSE/cost metric as bottom-up (not raw importance), ceiling F16: # the last downgraded (most precious) groups are the first to recover. upgrade_queue: List[UpgradeItem] = [] for g_id in group_registry: _push_upgrade(g_id, group_registry, assignments, tensor_importance, upgrade_queue, importance_table, ceiling="F16", uopt=uopt) while upgrade_queue: item = heapq.heappop(upgrade_queue) if item.cost_delta > 0 and current_size + item.cost_delta > effective_target: continue for n in group_registry[item.group_id][0]: assignments[n] = item.next_tier current_size += item.cost_delta _push_upgrade(item.group_id, group_registry, assignments, tensor_importance, upgrade_queue, importance_table, ceiling="F16", uopt=uopt) # Same tail-of-budget gap as bottom-up: single-rung steps only. current_size, polished = _polish_slack( assignments, group_registry, current_size, effective_target, tensor_importance, importance_table, uopt=uopt, order=TIER_ORDER, ceiling_of=lambda g_id: "F16") return assignments, padded_ne_map def compute_stats(assignments: dict, ne_map: dict = None, padded_ne_map: dict = None, f32_map: dict = None) -> dict: """Собирает статистику по тирам с точным учётом K_QUANTS padding.""" f32_map = f32_map or {} stats = {"by_tier_count": {}, "by_tier_mib": {}, "total_mib": 0.0, "tensor_count": 0} for name, tier in assignments.items(): if not isinstance(tier, str): continue stats["tensor_count"] += 1 stats["by_tier_count"][tier] = stats["by_tier_count"].get(tier, 0) + 1 if ne_map: if padded_ne_map and tier in K_QUANTS: elements = padded_ne_map.get(name, ne_map.get(name, 0)) else: elements = ne_map.get(name, 0) size = _size_mib(tier, elements, f32_map.get(name, False)) stats["by_tier_mib"][tier] = stats["by_tier_mib"].get(tier, 0.0) + size stats["total_mib"] += size return stats