/** * What precision a pack is actually in — the browser half. * * The JavaScript port of `tools/quantize/precision_map.py`. Both are held * together by `parity.mjs`, which runs real tensor names harvested from * nine toolchains through each and diffs. The rules took a wrong answer * to get right, and a port that drifted from them would bring it back * silently. * * The wrong answer is worth stating, because it is the whole reason this * file classifies the way it does. The first version keyed on tensor * NAMES, learned from compressed-tensors packs where the payload is * called `weight_packed`. Probing the ecosystem showed that convention is * the minority: bitsandbytes, ModelOpt FP8, ModelOpt NVFP4 and * compressed-tensors' own int-quantized and naive-quantized formats all * store the quantized payload under the plain name `weight`. A * name-driven reader reports every one of those packs as 100% full * precision — confidently, and about most of what is published. * * So dtype is the primary evidence here. Names are used only to separate * quantization metadata from model weights, a distinction that genuinely * has no dtype signature. */ /** Bytes per element, keyed by the safetensors dtype string. */ export const DTYPE_BYTES = { BF16: 2, F16: 2, F32: 4, F64: 8, F8_E4M3: 1, F8_E5M2: 1, F4: 1, I8: 1, U8: 1, I16: 2, I32: 4, I64: 8, BOOL: 1, }; /** * A tensor in one of these dtypes is carrying quantized payload. Every * packing scheme measured lands here: compressed-tensors packs into I32, * GPTQ/AWQ/AutoRound into I32, bitsandbytes and MXFP4 and NVFP4 into U8, * FP8 into F8_E4M3, int8 into I8. */ export const PAYLOAD_DTYPES = new Set( ['I32', 'I8', 'U8', 'F8_E4M3', 'F8_E5M2', 'F4', 'I16']); /** * Excluded on purpose: index buffers and masks (`position_ids`, causal * masks, `weight_shape`), never packed weights. */ export const BUFFER_DTYPES = new Set(['I64', 'BOOL']); /** * Quantization metadata: scales, zero points, group permutations, the * stored original shape. Real bytes, reported as their own category — * counting them as full precision would overstate what stayed whole, and * counting them as payload would hide the overhead. * * One union across all toolchains rather than a table per format: these * names do not collide, so a reader that misidentifies the format still * gets the metadata right. */ export const METADATA_SUFFIXES = [ // compressed-tensors 'weight_scale', 'weight_shape', 'weight_zero_point', 'weight_g_idx', 'input_scale', 'output_scale', // gptq / awq / autoround 'qzeros', 'scales', 'g_idx', // bitsandbytes — double quantization, so the scales are themselves quantized 'absmax', 'quant_map', 'nested_absmax', 'nested_quant_map', 'quant_state.bitsandbytes__nf4', 'quant_state.bitsandbytes__fp4', // ModelOpt / native fp8 + fp4 'weight_scale_2', 'k_scale', 'v_scale', 'q_scale', 'prob_scale', // mxfp4 '_scales', ]; /** Formats that give the payload its own name; the rest reuse `weight`. */ export const PAYLOAD_SUFFIXES = ['weight_packed', 'qweight', '_blocks']; /** * Every format here was verified by reading a real published pack of it. * * `producer` says where that pack records the version of the tool that * wrote it, and `null` means the tool records nothing — which is worth * reporting rather than hiding, since it tells the reader no version * evidence exists for those bytes. * * AutoAWQ is deliberately `null` even though its config has a `version` * field: that field holds `"gemm"`, the kernel variant. Printing it as a * release number would be a small invented fact. */ export const FORMATS = { 'compressed-tensors': { tool: 'llm-compressor / compressed-tensors', producer: ['config', 'version'], verifiedOn: 'aleada/Qwen3.8-27B-W4A16 (0.17.1), RedHatAI w8a8, RedHatAI FP8', note: 'pack-quantized names the payload weight_packed; int-quantized and naive-quantized reuse `weight`', }, gptq: { tool: 'GPTQModel / AutoGPTQ', producer: ['config', 'meta.quantizer'], verifiedOn: 'ModelCloud vortex-v3 (gptqmodel:1.4.4), TheBloke/Llama-2-7B-Chat-GPTQ', note: 'qweight + qzeros + scales + g_idx', }, awq: { tool: 'AutoAWQ', producer: null, verifiedOn: 'casperhansen/llama-3-8b-instruct-awq, Qwen/Qwen3-8B-AWQ', note: 'qweight + qzeros + scales, no g_idx', }, 'auto-round': { tool: 'Intel AutoRound', producer: ['config', 'autoround_version'], verifiedOn: 'OPEA/Qwen2.5-7B-Instruct-int4-sym-inc (0.4.0.dev)', note: 'exports in gptq or compressed-tensors layout', }, 'intel/auto-round': { tool: 'Intel AutoRound', producer: ['config', 'autoround_version'], verifiedOn: 'Intel/Qwen2-7B-int4-inc', note: 'same layout, older method spelling', }, bitsandbytes: { tool: 'bitsandbytes', producer: null, verifiedOn: 'unsloth/Meta-Llama-3.1-8B-Instruct-bnb-4bit', note: 'payload is U8 under the plain name `weight`; absmax/quant_map are the scales', }, mxfp4: { tool: 'MXFP4 (native export)', producer: null, verifiedOn: 'openai/gpt-oss-20b', note: '*_blocks payload + *_scales, per fused expert tensor', }, quark: { tool: 'AMD Quark', producer: null, verifiedOn: 'amd/Qwen3.8-27B-Quark-AWQ-INT4-W4A16', note: 'payload is `weight` at I32, with weight_scale and an I32 weight_zero_point', }, modelopt: { tool: 'NVIDIA TensorRT ModelOpt', producer: ['hf_quant_config', 'producer.version'], verifiedOn: 'nvidia/Llama-3.3-70B-Instruct-FP4 (0.23.0), nvidia/Llama-3.1-8B-Instruct-FP8', note: 'config.json declares nothing — the method lives in hf_quant_config.json', }, }; export const RULES_MEASURED_ON = '2026-09-07'; /** * Which module a full-precision tensor belongs to. Ordered: the first * match wins, so specific patterns precede general ones. */ export const FAMILIES = [ [/(^|\.)mtp\.|(^|\.)draft/, 'auxiliary head (MTP / draft)'], [/(^|\.)(lm_head|output_layer)\b/, 'output head (lm_head)'], [/embed|embeddings?\b|wte\b/, 'embeddings'], [/vision_tower|vision_model|visual|image_encoder|patch_embed/, 'vision tower'], [/multi_modal_projector|mm_projector|merger/, 'multimodal projector'], [/norm|layernorm|rmsnorm/, 'norms'], [/A_log|dt_bias|conv1d|\.D$/, 'state-space parameters'], [/(^|\.)(gate|router)\b|e_score_correction|sinks/, 'routers and sinks'], [/rotary|inv_freq|position/, 'position buffers'], [/\.bias$/, 'biases'], [/experts?\./, 'experts (left unquantized)'], ]; const HEAD_RE = /lm_head|output_layer/; /** * quantized | metadata | full | buffer. * * Name first, but only to pull metadata out — a scale is a scale whatever * its dtype. Then dtype, which is what actually decides whether a tensor * holds full-precision weights. */ export function classify(name, dtype) { for (const suf of METADATA_SUFFIXES) if (name.endsWith(suf)) return 'metadata'; for (const suf of PAYLOAD_SUFFIXES) if (name.endsWith(suf)) return 'quantized'; if (BUFFER_DTYPES.has(dtype)) return 'buffer'; if (PAYLOAD_DTYPES.has(dtype)) return 'quantized'; return 'full'; } export function family(name) { for (const [pattern, label] of FAMILIES) if (pattern.test(name)) return label; return 'other linear weights'; } /** * `meta.quantizer` out of a nested object, tolerating a list value: * GPTQModel writes that field as a bare string in some releases and a * one-element array in others, and both are in the wild. */ export function dig(obj, path) { let cur = obj; for (const part of path.split('.')) { if (cur === null || typeof cur !== 'object' || Array.isArray(cur)) return null; cur = cur[part]; if (cur === undefined) return null; } if (Array.isArray(cur)) return cur.length ? cur.map(String).join(', ') : null; return cur === undefined ? null : cur; } /** * The quantization method, from wherever the tool chose to record it. * * Never from the repo name: `cyankiwi/Qwen3-VL-8B-Instruct-AWQ-4bit` * declares `compressed-tensors`, and a reader that trusted the name would * apply the wrong convention to it. */ export function declaredMethod(cfg, hfQuant) { let q = cfg?.quantization_config; if (!q && cfg?.text_config && typeof cfg.text_config === 'object') { q = cfg.text_config.quantization_config; } const m = q?.quant_method; if (m) return String(m); if (hfQuant) return 'modelopt'; return 'none'; } /** * Fold a list of `{name, dtype, shape}` into the report. * * Split out from any fetching so the parity gate can run it on fixtures * and so the same arithmetic serves whichever transport reads the header. */ export function summarize(tensors, method) { const spec = FORMATS[method] || null; const buckets = { quantized: 0, metadata: 0, full: 0, buffer: 0 }; const families = {}; const dtypeBytes = {}; const unknownDtypes = new Set(); let headQuantized = false; for (const t of tensors) { if (!(t.dtype in DTYPE_BYTES)) unknownDtypes.add(t.dtype); let n = 1; for (const d of t.shape) n *= d; const size = n * (DTYPE_BYTES[t.dtype] ?? 2); const kind = classify(t.name, t.dtype); buckets[kind] += size; dtypeBytes[t.dtype] = (dtypeBytes[t.dtype] || 0) + size; if (kind === 'full') { const fam = family(t.name); families[fam] = (families[fam] || 0) + size; } else if (kind === 'quantized' && HEAD_RE.test(t.name)) { headQuantized = true; } } const total = Object.values(buckets).reduce((a, b) => a + b, 0) || 1; // Largest first, ties broken by name. The tie is not hypothetical: an // untied model's embeddings and output head are the same tensor // transposed, so their byte counts are equal exactly. Shards are read // concurrently, so without the second key those two rows would swap // between runs of the same pack — and a report that changes when // nothing changed is a report nobody can quote. // // Compared by code point, not localeCompare: Python's `sorted` orders // by code point, and a locale-aware comparison disagrees with it on // case and punctuation. The parity gate can only catch that when a // fixture happens to contain a tie, so the two are made to agree by // construction instead of by luck. const byName = (a, b) => (a < b ? -1 : a > b ? 1 : 0); const sortDesc = (o) => Object.fromEntries( Object.entries(o).sort((a, b) => b[1] - a[1] || byName(a[0], b[0]))); return { method, tool: spec ? spec.tool : null, verifiedOn: spec ? spec.verifiedOn : null, note: spec ? spec.note : null, totalBytes: total, buckets, dtypeBytes: sortDesc(dtypeBytes), fullByFamily: sortDesc(families), fullPct: (buckets.full / total) * 100, headQuantized, unknownDtypes: [...unknownDtypes].sort(), // A declared method with no payload found means this reader does not // understand that tool's layout — not that the pack is unquantized. // Saying so is the one guard that keeps a future convention change // from producing a confident wrong answer. recognised: buckets.quantized > 0 || method === 'none', knownFormat: spec !== null || method === 'none', }; } // ------------------------------------------------------------- transport const HF = 'https://huggingface.co'; async function json(url) { const r = await fetch(url); if (!r.ok) return null; try { return await r.json(); } catch { return null; } } /** * Read one safetensors file's header without downloading the file. * * The format opens with a little-endian u64 giving the header length, * then that many bytes of JSON describing every tensor. Two ranged * requests fetch it — typically a few hundred kilobytes against shards * of many gigabytes, which is what makes this possible from a browser * at all. */ export async function readHeader(repoId, file) { const url = `${HF}/${repoId}/resolve/main/${file}`; const head = await fetch(url, { headers: { Range: 'bytes=0-7' } }); if (!head.ok) throw new Error(`cannot read ${file} (HTTP ${head.status})`); const len = Number(new DataView(await head.arrayBuffer()).getBigUint64(0, true)); if (!Number.isFinite(len) || len <= 0 || len > 200_000_000) { throw new Error(`${file} does not look like a safetensors file`); } const body = await fetch(url, { headers: { Range: `bytes=8-${7 + len}` } }); if (!body.ok) throw new Error(`cannot read ${file} header (HTTP ${body.status})`); const parsed = JSON.parse(new TextDecoder().decode(await body.arrayBuffer())); const tensors = []; for (const [name, info] of Object.entries(parsed)) { if (name === '__metadata__') continue; tensors.push({ name, dtype: info.dtype, shape: info.shape || [] }); } return tensors; } /** Which safetensors files a repo publishes, sharded or not. */ export async function shardList(repoId) { const index = await json(`${HF}/${repoId}/resolve/main/model.safetensors.index.json`); if (index?.weight_map) return [...new Set(Object.values(index.weight_map))]; const head = await fetch(`${HF}/${repoId}/resolve/main/model.safetensors`, { headers: { Range: 'bytes=0-7' } }); if (head.ok) return ['model.safetensors']; return []; } /** Everything the page needs about one repo. */ export async function inspect(repoId, onProgress = () => {}) { const cfg = await json(`${HF}/${repoId}/resolve/main/config.json`); if (!cfg) throw new Error('no config.json — gated, private, or not a model repo'); const hfQuant = await json(`${HF}/${repoId}/resolve/main/hf_quant_config.json`); const method = declaredMethod(cfg, hfQuant); let producerVersion = null; const spec = FORMATS[method]; if (spec?.producer) { const [where, path] = spec.producer; const src = where === 'config' ? (cfg.quantization_config || {}) : (hfQuant || {}); const v = dig(src, path); producerVersion = v === null || v === undefined ? null : String(v); } const files = await shardList(repoId); if (!files.length) { // A pack whose weights are a .pt or .bin cannot be read this way, and // must not be reported as unquantized. throw new Error('no safetensors weights published — torchao and HQQ ' + 'packs often ship a .pt instead, and their contents cannot be read ' + 'from metadata'); } onProgress(0, files.length); // Read the shards a few at a time rather than one after another. A // 70B pack is nine shards and two ranged requests each; sequentially // that is long enough that the page reads as broken. Four at a time // is enough to hide most of the latency without opening a burst of // connections against the Hub for a page that is only reading // metadata. const tensors = []; let next = 0; let finished = 0; const worker = async () => { while (next < files.length) { const mine = files[next++]; const got = await readHeader(repoId, mine); tensors.push(...got); onProgress(++finished, files.length); } }; await Promise.all( Array.from({ length: Math.min(4, files.length) }, worker)); return { repo: repoId, producerVersion, shards: files.length, tensorCount: tensors.length, ...summarize(tensors, method), }; }