Spaces:
Running
Running
File size: 15,630 Bytes
e73d8ef | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 | /**
* What precision a pack is actually in — the browser half.
*
* The JavaScript port of `tools/quantize/precision_map.py`. Both are held
* together by `parity.mjs`, which runs real tensor names harvested from
* nine toolchains through each and diffs. The rules took a wrong answer
* to get right, and a port that drifted from them would bring it back
* silently.
*
* The wrong answer is worth stating, because it is the whole reason this
* file classifies the way it does. The first version keyed on tensor
* NAMES, learned from compressed-tensors packs where the payload is
* called `weight_packed`. Probing the ecosystem showed that convention is
* the minority: bitsandbytes, ModelOpt FP8, ModelOpt NVFP4 and
* compressed-tensors' own int-quantized and naive-quantized formats all
* store the quantized payload under the plain name `weight`. A
* name-driven reader reports every one of those packs as 100% full
* precision — confidently, and about most of what is published.
*
* So dtype is the primary evidence here. Names are used only to separate
* quantization metadata from model weights, a distinction that genuinely
* has no dtype signature.
*/
/** Bytes per element, keyed by the safetensors dtype string. */
export const DTYPE_BYTES = {
BF16: 2, F16: 2, F32: 4, F64: 8, F8_E4M3: 1, F8_E5M2: 1, F4: 1,
I8: 1, U8: 1, I16: 2, I32: 4, I64: 8, BOOL: 1,
};
/**
* A tensor in one of these dtypes is carrying quantized payload. Every
* packing scheme measured lands here: compressed-tensors packs into I32,
* GPTQ/AWQ/AutoRound into I32, bitsandbytes and MXFP4 and NVFP4 into U8,
* FP8 into F8_E4M3, int8 into I8.
*/
export const PAYLOAD_DTYPES = new Set(
['I32', 'I8', 'U8', 'F8_E4M3', 'F8_E5M2', 'F4', 'I16']);
/**
* Excluded on purpose: index buffers and masks (`position_ids`, causal
* masks, `weight_shape`), never packed weights.
*/
export const BUFFER_DTYPES = new Set(['I64', 'BOOL']);
/**
* Quantization metadata: scales, zero points, group permutations, the
* stored original shape. Real bytes, reported as their own category —
* counting them as full precision would overstate what stayed whole, and
* counting them as payload would hide the overhead.
*
* One union across all toolchains rather than a table per format: these
* names do not collide, so a reader that misidentifies the format still
* gets the metadata right.
*/
export const METADATA_SUFFIXES = [
// compressed-tensors
'weight_scale', 'weight_shape', 'weight_zero_point', 'weight_g_idx',
'input_scale', 'output_scale',
// gptq / awq / autoround
'qzeros', 'scales', 'g_idx',
// bitsandbytes — double quantization, so the scales are themselves quantized
'absmax', 'quant_map', 'nested_absmax', 'nested_quant_map',
'quant_state.bitsandbytes__nf4', 'quant_state.bitsandbytes__fp4',
// ModelOpt / native fp8 + fp4
'weight_scale_2', 'k_scale', 'v_scale', 'q_scale', 'prob_scale',
// mxfp4
'_scales',
];
/** Formats that give the payload its own name; the rest reuse `weight`. */
export const PAYLOAD_SUFFIXES = ['weight_packed', 'qweight', '_blocks'];
/**
* Every format here was verified by reading a real published pack of it.
*
* `producer` says where that pack records the version of the tool that
* wrote it, and `null` means the tool records nothing — which is worth
* reporting rather than hiding, since it tells the reader no version
* evidence exists for those bytes.
*
* AutoAWQ is deliberately `null` even though its config has a `version`
* field: that field holds `"gemm"`, the kernel variant. Printing it as a
* release number would be a small invented fact.
*/
export const FORMATS = {
'compressed-tensors': {
tool: 'llm-compressor / compressed-tensors',
producer: ['config', 'version'],
verifiedOn: 'aleada/Qwen3.8-27B-W4A16 (0.17.1), RedHatAI w8a8, RedHatAI FP8',
note: 'pack-quantized names the payload weight_packed; int-quantized and naive-quantized reuse `weight`',
},
gptq: {
tool: 'GPTQModel / AutoGPTQ',
producer: ['config', 'meta.quantizer'],
verifiedOn: 'ModelCloud vortex-v3 (gptqmodel:1.4.4), TheBloke/Llama-2-7B-Chat-GPTQ',
note: 'qweight + qzeros + scales + g_idx',
},
awq: {
tool: 'AutoAWQ',
producer: null,
verifiedOn: 'casperhansen/llama-3-8b-instruct-awq, Qwen/Qwen3-8B-AWQ',
note: 'qweight + qzeros + scales, no g_idx',
},
'auto-round': {
tool: 'Intel AutoRound',
producer: ['config', 'autoround_version'],
verifiedOn: 'OPEA/Qwen2.5-7B-Instruct-int4-sym-inc (0.4.0.dev)',
note: 'exports in gptq or compressed-tensors layout',
},
'intel/auto-round': {
tool: 'Intel AutoRound',
producer: ['config', 'autoround_version'],
verifiedOn: 'Intel/Qwen2-7B-int4-inc',
note: 'same layout, older method spelling',
},
bitsandbytes: {
tool: 'bitsandbytes',
producer: null,
verifiedOn: 'unsloth/Meta-Llama-3.1-8B-Instruct-bnb-4bit',
note: 'payload is U8 under the plain name `weight`; absmax/quant_map are the scales',
},
mxfp4: {
tool: 'MXFP4 (native export)',
producer: null,
verifiedOn: 'openai/gpt-oss-20b',
note: '*_blocks payload + *_scales, per fused expert tensor',
},
quark: {
tool: 'AMD Quark',
producer: null,
verifiedOn: 'amd/Qwen3.8-27B-Quark-AWQ-INT4-W4A16',
note: 'payload is `weight` at I32, with weight_scale and an I32 weight_zero_point',
},
modelopt: {
tool: 'NVIDIA TensorRT ModelOpt',
producer: ['hf_quant_config', 'producer.version'],
verifiedOn: 'nvidia/Llama-3.3-70B-Instruct-FP4 (0.23.0), nvidia/Llama-3.1-8B-Instruct-FP8',
note: 'config.json declares nothing — the method lives in hf_quant_config.json',
},
};
export const RULES_MEASURED_ON = '2026-09-07';
/**
* Which module a full-precision tensor belongs to. Ordered: the first
* match wins, so specific patterns precede general ones.
*/
export const FAMILIES = [
[/(^|\.)mtp\.|(^|\.)draft/, 'auxiliary head (MTP / draft)'],
[/(^|\.)(lm_head|output_layer)\b/, 'output head (lm_head)'],
[/embed|embeddings?\b|wte\b/, 'embeddings'],
[/vision_tower|vision_model|visual|image_encoder|patch_embed/, 'vision tower'],
[/multi_modal_projector|mm_projector|merger/, 'multimodal projector'],
[/norm|layernorm|rmsnorm/, 'norms'],
[/A_log|dt_bias|conv1d|\.D$/, 'state-space parameters'],
[/(^|\.)(gate|router)\b|e_score_correction|sinks/, 'routers and sinks'],
[/rotary|inv_freq|position/, 'position buffers'],
[/\.bias$/, 'biases'],
[/experts?\./, 'experts (left unquantized)'],
];
const HEAD_RE = /lm_head|output_layer/;
/**
* quantized | metadata | full | buffer.
*
* Name first, but only to pull metadata out — a scale is a scale whatever
* its dtype. Then dtype, which is what actually decides whether a tensor
* holds full-precision weights.
*/
export function classify(name, dtype) {
for (const suf of METADATA_SUFFIXES) if (name.endsWith(suf)) return 'metadata';
for (const suf of PAYLOAD_SUFFIXES) if (name.endsWith(suf)) return 'quantized';
if (BUFFER_DTYPES.has(dtype)) return 'buffer';
if (PAYLOAD_DTYPES.has(dtype)) return 'quantized';
return 'full';
}
export function family(name) {
for (const [pattern, label] of FAMILIES) if (pattern.test(name)) return label;
return 'other linear weights';
}
/**
* `meta.quantizer` out of a nested object, tolerating a list value:
* GPTQModel writes that field as a bare string in some releases and a
* one-element array in others, and both are in the wild.
*/
export function dig(obj, path) {
let cur = obj;
for (const part of path.split('.')) {
if (cur === null || typeof cur !== 'object' || Array.isArray(cur)) return null;
cur = cur[part];
if (cur === undefined) return null;
}
if (Array.isArray(cur)) return cur.length ? cur.map(String).join(', ') : null;
return cur === undefined ? null : cur;
}
/**
* The quantization method, from wherever the tool chose to record it.
*
* Never from the repo name: `cyankiwi/Qwen3-VL-8B-Instruct-AWQ-4bit`
* declares `compressed-tensors`, and a reader that trusted the name would
* apply the wrong convention to it.
*/
export function declaredMethod(cfg, hfQuant) {
let q = cfg?.quantization_config;
if (!q && cfg?.text_config && typeof cfg.text_config === 'object') {
q = cfg.text_config.quantization_config;
}
const m = q?.quant_method;
if (m) return String(m);
if (hfQuant) return 'modelopt';
return 'none';
}
/**
* Fold a list of `{name, dtype, shape}` into the report.
*
* Split out from any fetching so the parity gate can run it on fixtures
* and so the same arithmetic serves whichever transport reads the header.
*/
export function summarize(tensors, method) {
const spec = FORMATS[method] || null;
const buckets = { quantized: 0, metadata: 0, full: 0, buffer: 0 };
const families = {};
const dtypeBytes = {};
const unknownDtypes = new Set();
let headQuantized = false;
for (const t of tensors) {
if (!(t.dtype in DTYPE_BYTES)) unknownDtypes.add(t.dtype);
let n = 1;
for (const d of t.shape) n *= d;
const size = n * (DTYPE_BYTES[t.dtype] ?? 2);
const kind = classify(t.name, t.dtype);
buckets[kind] += size;
dtypeBytes[t.dtype] = (dtypeBytes[t.dtype] || 0) + size;
if (kind === 'full') {
const fam = family(t.name);
families[fam] = (families[fam] || 0) + size;
} else if (kind === 'quantized' && HEAD_RE.test(t.name)) {
headQuantized = true;
}
}
const total = Object.values(buckets).reduce((a, b) => a + b, 0) || 1;
// Largest first, ties broken by name. The tie is not hypothetical: an
// untied model's embeddings and output head are the same tensor
// transposed, so their byte counts are equal exactly. Shards are read
// concurrently, so without the second key those two rows would swap
// between runs of the same pack — and a report that changes when
// nothing changed is a report nobody can quote.
//
// Compared by code point, not localeCompare: Python's `sorted` orders
// by code point, and a locale-aware comparison disagrees with it on
// case and punctuation. The parity gate can only catch that when a
// fixture happens to contain a tie, so the two are made to agree by
// construction instead of by luck.
const byName = (a, b) => (a < b ? -1 : a > b ? 1 : 0);
const sortDesc = (o) => Object.fromEntries(
Object.entries(o).sort((a, b) => b[1] - a[1] || byName(a[0], b[0])));
return {
method,
tool: spec ? spec.tool : null,
verifiedOn: spec ? spec.verifiedOn : null,
note: spec ? spec.note : null,
totalBytes: total,
buckets,
dtypeBytes: sortDesc(dtypeBytes),
fullByFamily: sortDesc(families),
fullPct: (buckets.full / total) * 100,
headQuantized,
unknownDtypes: [...unknownDtypes].sort(),
// A declared method with no payload found means this reader does not
// understand that tool's layout — not that the pack is unquantized.
// Saying so is the one guard that keeps a future convention change
// from producing a confident wrong answer.
recognised: buckets.quantized > 0 || method === 'none',
knownFormat: spec !== null || method === 'none',
};
}
// ------------------------------------------------------------- transport
const HF = 'https://huggingface.co';
async function json(url) {
const r = await fetch(url);
if (!r.ok) return null;
try { return await r.json(); } catch { return null; }
}
/**
* Read one safetensors file's header without downloading the file.
*
* The format opens with a little-endian u64 giving the header length,
* then that many bytes of JSON describing every tensor. Two ranged
* requests fetch it — typically a few hundred kilobytes against shards
* of many gigabytes, which is what makes this possible from a browser
* at all.
*/
export async function readHeader(repoId, file) {
const url = `${HF}/${repoId}/resolve/main/${file}`;
const head = await fetch(url, { headers: { Range: 'bytes=0-7' } });
if (!head.ok) throw new Error(`cannot read ${file} (HTTP ${head.status})`);
const len = Number(new DataView(await head.arrayBuffer()).getBigUint64(0, true));
if (!Number.isFinite(len) || len <= 0 || len > 200_000_000) {
throw new Error(`${file} does not look like a safetensors file`);
}
const body = await fetch(url, { headers: { Range: `bytes=8-${7 + len}` } });
if (!body.ok) throw new Error(`cannot read ${file} header (HTTP ${body.status})`);
const parsed = JSON.parse(new TextDecoder().decode(await body.arrayBuffer()));
const tensors = [];
for (const [name, info] of Object.entries(parsed)) {
if (name === '__metadata__') continue;
tensors.push({ name, dtype: info.dtype, shape: info.shape || [] });
}
return tensors;
}
/** Which safetensors files a repo publishes, sharded or not. */
export async function shardList(repoId) {
const index = await json(`${HF}/${repoId}/resolve/main/model.safetensors.index.json`);
if (index?.weight_map) return [...new Set(Object.values(index.weight_map))];
const head = await fetch(`${HF}/${repoId}/resolve/main/model.safetensors`,
{ headers: { Range: 'bytes=0-7' } });
if (head.ok) return ['model.safetensors'];
return [];
}
/** Everything the page needs about one repo. */
export async function inspect(repoId, onProgress = () => {}) {
const cfg = await json(`${HF}/${repoId}/resolve/main/config.json`);
if (!cfg) throw new Error('no config.json — gated, private, or not a model repo');
const hfQuant = await json(`${HF}/${repoId}/resolve/main/hf_quant_config.json`);
const method = declaredMethod(cfg, hfQuant);
let producerVersion = null;
const spec = FORMATS[method];
if (spec?.producer) {
const [where, path] = spec.producer;
const src = where === 'config' ? (cfg.quantization_config || {}) : (hfQuant || {});
const v = dig(src, path);
producerVersion = v === null || v === undefined ? null : String(v);
}
const files = await shardList(repoId);
if (!files.length) {
// A pack whose weights are a .pt or .bin cannot be read this way, and
// must not be reported as unquantized.
throw new Error('no safetensors weights published — torchao and HQQ '
+ 'packs often ship a .pt instead, and their contents cannot be read '
+ 'from metadata');
}
onProgress(0, files.length);
// Read the shards a few at a time rather than one after another. A
// 70B pack is nine shards and two ranged requests each; sequentially
// that is long enough that the page reads as broken. Four at a time
// is enough to hide most of the latency without opening a burst of
// connections against the Hub for a page that is only reading
// metadata.
const tensors = [];
let next = 0;
let finished = 0;
const worker = async () => {
while (next < files.length) {
const mine = files[next++];
const got = await readHeader(repoId, mine);
tensors.push(...got);
onProgress(++finished, files.length);
}
};
await Promise.all(
Array.from({ length: Math.min(4, files.length) }, worker));
return {
repo: repoId,
producerVersion,
shards: files.length,
tensorCount: tensors.length,
...summarize(tensors, method),
};
}
|