File size: 15,630 Bytes
e73d8ef
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
/**

 * What precision a pack is actually in — the browser half.

 *

 * The JavaScript port of `tools/quantize/precision_map.py`. Both are held

 * together by `parity.mjs`, which runs real tensor names harvested from

 * nine toolchains through each and diffs. The rules took a wrong answer

 * to get right, and a port that drifted from them would bring it back

 * silently.

 *

 * The wrong answer is worth stating, because it is the whole reason this

 * file classifies the way it does. The first version keyed on tensor

 * NAMES, learned from compressed-tensors packs where the payload is

 * called `weight_packed`. Probing the ecosystem showed that convention is

 * the minority: bitsandbytes, ModelOpt FP8, ModelOpt NVFP4 and

 * compressed-tensors' own int-quantized and naive-quantized formats all

 * store the quantized payload under the plain name `weight`. A

 * name-driven reader reports every one of those packs as 100% full

 * precision — confidently, and about most of what is published.

 *

 * So dtype is the primary evidence here. Names are used only to separate

 * quantization metadata from model weights, a distinction that genuinely

 * has no dtype signature.

 */

/** Bytes per element, keyed by the safetensors dtype string. */
export const DTYPE_BYTES = {
  BF16: 2, F16: 2, F32: 4, F64: 8, F8_E4M3: 1, F8_E5M2: 1, F4: 1,
  I8: 1, U8: 1, I16: 2, I32: 4, I64: 8, BOOL: 1,
};

/**

 * A tensor in one of these dtypes is carrying quantized payload. Every

 * packing scheme measured lands here: compressed-tensors packs into I32,

 * GPTQ/AWQ/AutoRound into I32, bitsandbytes and MXFP4 and NVFP4 into U8,

 * FP8 into F8_E4M3, int8 into I8.

 */
export const PAYLOAD_DTYPES = new Set(
  ['I32', 'I8', 'U8', 'F8_E4M3', 'F8_E5M2', 'F4', 'I16']);

/**

 * Excluded on purpose: index buffers and masks (`position_ids`, causal

 * masks, `weight_shape`), never packed weights.

 */
export const BUFFER_DTYPES = new Set(['I64', 'BOOL']);

/**

 * Quantization metadata: scales, zero points, group permutations, the

 * stored original shape. Real bytes, reported as their own category —

 * counting them as full precision would overstate what stayed whole, and

 * counting them as payload would hide the overhead.

 *

 * One union across all toolchains rather than a table per format: these

 * names do not collide, so a reader that misidentifies the format still

 * gets the metadata right.

 */
export const METADATA_SUFFIXES = [
  // compressed-tensors
  'weight_scale', 'weight_shape', 'weight_zero_point', 'weight_g_idx',
  'input_scale', 'output_scale',
  // gptq / awq / autoround
  'qzeros', 'scales', 'g_idx',
  // bitsandbytes — double quantization, so the scales are themselves quantized
  'absmax', 'quant_map', 'nested_absmax', 'nested_quant_map',
  'quant_state.bitsandbytes__nf4', 'quant_state.bitsandbytes__fp4',
  // ModelOpt / native fp8 + fp4
  'weight_scale_2', 'k_scale', 'v_scale', 'q_scale', 'prob_scale',
  // mxfp4
  '_scales',
];

/** Formats that give the payload its own name; the rest reuse `weight`. */
export const PAYLOAD_SUFFIXES = ['weight_packed', 'qweight', '_blocks'];

/**

 * Every format here was verified by reading a real published pack of it.

 *

 * `producer` says where that pack records the version of the tool that

 * wrote it, and `null` means the tool records nothing — which is worth

 * reporting rather than hiding, since it tells the reader no version

 * evidence exists for those bytes.

 *

 * AutoAWQ is deliberately `null` even though its config has a `version`

 * field: that field holds `"gemm"`, the kernel variant. Printing it as a

 * release number would be a small invented fact.

 */
export const FORMATS = {
  'compressed-tensors': {
    tool: 'llm-compressor / compressed-tensors',
    producer: ['config', 'version'],
    verifiedOn: 'aleada/Qwen3.8-27B-W4A16 (0.17.1), RedHatAI w8a8, RedHatAI FP8',
    note: 'pack-quantized names the payload weight_packed; int-quantized and naive-quantized reuse `weight`',
  },
  gptq: {
    tool: 'GPTQModel / AutoGPTQ',
    producer: ['config', 'meta.quantizer'],
    verifiedOn: 'ModelCloud vortex-v3 (gptqmodel:1.4.4), TheBloke/Llama-2-7B-Chat-GPTQ',
    note: 'qweight + qzeros + scales + g_idx',
  },
  awq: {
    tool: 'AutoAWQ',
    producer: null,
    verifiedOn: 'casperhansen/llama-3-8b-instruct-awq, Qwen/Qwen3-8B-AWQ',
    note: 'qweight + qzeros + scales, no g_idx',
  },
  'auto-round': {
    tool: 'Intel AutoRound',
    producer: ['config', 'autoround_version'],
    verifiedOn: 'OPEA/Qwen2.5-7B-Instruct-int4-sym-inc (0.4.0.dev)',
    note: 'exports in gptq or compressed-tensors layout',
  },
  'intel/auto-round': {
    tool: 'Intel AutoRound',
    producer: ['config', 'autoround_version'],
    verifiedOn: 'Intel/Qwen2-7B-int4-inc',
    note: 'same layout, older method spelling',
  },
  bitsandbytes: {
    tool: 'bitsandbytes',
    producer: null,
    verifiedOn: 'unsloth/Meta-Llama-3.1-8B-Instruct-bnb-4bit',
    note: 'payload is U8 under the plain name `weight`; absmax/quant_map are the scales',
  },
  mxfp4: {
    tool: 'MXFP4 (native export)',
    producer: null,
    verifiedOn: 'openai/gpt-oss-20b',
    note: '*_blocks payload + *_scales, per fused expert tensor',
  },
  quark: {
    tool: 'AMD Quark',
    producer: null,
    verifiedOn: 'amd/Qwen3.8-27B-Quark-AWQ-INT4-W4A16',
    note: 'payload is `weight` at I32, with weight_scale and an I32 weight_zero_point',
  },
  modelopt: {
    tool: 'NVIDIA TensorRT ModelOpt',
    producer: ['hf_quant_config', 'producer.version'],
    verifiedOn: 'nvidia/Llama-3.3-70B-Instruct-FP4 (0.23.0), nvidia/Llama-3.1-8B-Instruct-FP8',
    note: 'config.json declares nothing — the method lives in hf_quant_config.json',
  },
};

export const RULES_MEASURED_ON = '2026-09-07';

/**

 * Which module a full-precision tensor belongs to. Ordered: the first

 * match wins, so specific patterns precede general ones.

 */
export const FAMILIES = [
  [/(^|\.)mtp\.|(^|\.)draft/, 'auxiliary head (MTP / draft)'],
  [/(^|\.)(lm_head|output_layer)\b/, 'output head (lm_head)'],
  [/embed|embeddings?\b|wte\b/, 'embeddings'],
  [/vision_tower|vision_model|visual|image_encoder|patch_embed/, 'vision tower'],
  [/multi_modal_projector|mm_projector|merger/, 'multimodal projector'],
  [/norm|layernorm|rmsnorm/, 'norms'],
  [/A_log|dt_bias|conv1d|\.D$/, 'state-space parameters'],
  [/(^|\.)(gate|router)\b|e_score_correction|sinks/, 'routers and sinks'],
  [/rotary|inv_freq|position/, 'position buffers'],
  [/\.bias$/, 'biases'],
  [/experts?\./, 'experts (left unquantized)'],
];

const HEAD_RE = /lm_head|output_layer/;

/**

 * quantized | metadata | full | buffer.

 *

 * Name first, but only to pull metadata out — a scale is a scale whatever

 * its dtype. Then dtype, which is what actually decides whether a tensor

 * holds full-precision weights.

 */
export function classify(name, dtype) {
  for (const suf of METADATA_SUFFIXES) if (name.endsWith(suf)) return 'metadata';
  for (const suf of PAYLOAD_SUFFIXES) if (name.endsWith(suf)) return 'quantized';
  if (BUFFER_DTYPES.has(dtype)) return 'buffer';
  if (PAYLOAD_DTYPES.has(dtype)) return 'quantized';
  return 'full';
}

export function family(name) {
  for (const [pattern, label] of FAMILIES) if (pattern.test(name)) return label;
  return 'other linear weights';
}

/**

 * `meta.quantizer` out of a nested object, tolerating a list value:

 * GPTQModel writes that field as a bare string in some releases and a

 * one-element array in others, and both are in the wild.

 */
export function dig(obj, path) {
  let cur = obj;
  for (const part of path.split('.')) {
    if (cur === null || typeof cur !== 'object' || Array.isArray(cur)) return null;
    cur = cur[part];
    if (cur === undefined) return null;
  }
  if (Array.isArray(cur)) return cur.length ? cur.map(String).join(', ') : null;
  return cur === undefined ? null : cur;
}

/**

 * The quantization method, from wherever the tool chose to record it.

 *

 * Never from the repo name: `cyankiwi/Qwen3-VL-8B-Instruct-AWQ-4bit`

 * declares `compressed-tensors`, and a reader that trusted the name would

 * apply the wrong convention to it.

 */
export function declaredMethod(cfg, hfQuant) {
  let q = cfg?.quantization_config;
  if (!q && cfg?.text_config && typeof cfg.text_config === 'object') {
    q = cfg.text_config.quantization_config;
  }
  const m = q?.quant_method;
  if (m) return String(m);
  if (hfQuant) return 'modelopt';
  return 'none';
}

/**

 * Fold a list of `{name, dtype, shape}` into the report.

 *

 * Split out from any fetching so the parity gate can run it on fixtures

 * and so the same arithmetic serves whichever transport reads the header.

 */
export function summarize(tensors, method) {
  const spec = FORMATS[method] || null;
  const buckets = { quantized: 0, metadata: 0, full: 0, buffer: 0 };
  const families = {};
  const dtypeBytes = {};
  const unknownDtypes = new Set();
  let headQuantized = false;

  for (const t of tensors) {
    if (!(t.dtype in DTYPE_BYTES)) unknownDtypes.add(t.dtype);
    let n = 1;
    for (const d of t.shape) n *= d;
    const size = n * (DTYPE_BYTES[t.dtype] ?? 2);
    const kind = classify(t.name, t.dtype);
    buckets[kind] += size;
    dtypeBytes[t.dtype] = (dtypeBytes[t.dtype] || 0) + size;
    if (kind === 'full') {
      const fam = family(t.name);
      families[fam] = (families[fam] || 0) + size;
    } else if (kind === 'quantized' && HEAD_RE.test(t.name)) {
      headQuantized = true;
    }
  }

  const total = Object.values(buckets).reduce((a, b) => a + b, 0) || 1;
  // Largest first, ties broken by name. The tie is not hypothetical: an
  // untied model's embeddings and output head are the same tensor
  // transposed, so their byte counts are equal exactly. Shards are read
  // concurrently, so without the second key those two rows would swap
  // between runs of the same pack — and a report that changes when
  // nothing changed is a report nobody can quote.
  //
  // Compared by code point, not localeCompare: Python's `sorted` orders
  // by code point, and a locale-aware comparison disagrees with it on
  // case and punctuation. The parity gate can only catch that when a
  // fixture happens to contain a tie, so the two are made to agree by
  // construction instead of by luck.
  const byName = (a, b) => (a < b ? -1 : a > b ? 1 : 0);
  const sortDesc = (o) => Object.fromEntries(
    Object.entries(o).sort((a, b) => b[1] - a[1] || byName(a[0], b[0])));

  return {
    method,
    tool: spec ? spec.tool : null,
    verifiedOn: spec ? spec.verifiedOn : null,
    note: spec ? spec.note : null,
    totalBytes: total,
    buckets,
    dtypeBytes: sortDesc(dtypeBytes),
    fullByFamily: sortDesc(families),
    fullPct: (buckets.full / total) * 100,
    headQuantized,
    unknownDtypes: [...unknownDtypes].sort(),
    // A declared method with no payload found means this reader does not
    // understand that tool's layout — not that the pack is unquantized.
    // Saying so is the one guard that keeps a future convention change
    // from producing a confident wrong answer.
    recognised: buckets.quantized > 0 || method === 'none',
    knownFormat: spec !== null || method === 'none',
  };
}

// ------------------------------------------------------------- transport

const HF = 'https://huggingface.co';

async function json(url) {
  const r = await fetch(url);
  if (!r.ok) return null;
  try { return await r.json(); } catch { return null; }
}

/**

 * Read one safetensors file's header without downloading the file.

 *

 * The format opens with a little-endian u64 giving the header length,

 * then that many bytes of JSON describing every tensor. Two ranged

 * requests fetch it — typically a few hundred kilobytes against shards

 * of many gigabytes, which is what makes this possible from a browser

 * at all.

 */
export async function readHeader(repoId, file) {
  const url = `${HF}/${repoId}/resolve/main/${file}`;
  const head = await fetch(url, { headers: { Range: 'bytes=0-7' } });
  if (!head.ok) throw new Error(`cannot read ${file} (HTTP ${head.status})`);
  const len = Number(new DataView(await head.arrayBuffer()).getBigUint64(0, true));
  if (!Number.isFinite(len) || len <= 0 || len > 200_000_000) {
    throw new Error(`${file} does not look like a safetensors file`);
  }
  const body = await fetch(url, { headers: { Range: `bytes=8-${7 + len}` } });
  if (!body.ok) throw new Error(`cannot read ${file} header (HTTP ${body.status})`);
  const parsed = JSON.parse(new TextDecoder().decode(await body.arrayBuffer()));

  const tensors = [];
  for (const [name, info] of Object.entries(parsed)) {
    if (name === '__metadata__') continue;
    tensors.push({ name, dtype: info.dtype, shape: info.shape || [] });
  }
  return tensors;
}

/** Which safetensors files a repo publishes, sharded or not. */
export async function shardList(repoId) {
  const index = await json(`${HF}/${repoId}/resolve/main/model.safetensors.index.json`);
  if (index?.weight_map) return [...new Set(Object.values(index.weight_map))];
  const head = await fetch(`${HF}/${repoId}/resolve/main/model.safetensors`,
    { headers: { Range: 'bytes=0-7' } });
  if (head.ok) return ['model.safetensors'];
  return [];
}

/** Everything the page needs about one repo. */
export async function inspect(repoId, onProgress = () => {}) {
  const cfg = await json(`${HF}/${repoId}/resolve/main/config.json`);
  if (!cfg) throw new Error('no config.json — gated, private, or not a model repo');
  const hfQuant = await json(`${HF}/${repoId}/resolve/main/hf_quant_config.json`);
  const method = declaredMethod(cfg, hfQuant);

  let producerVersion = null;
  const spec = FORMATS[method];
  if (spec?.producer) {
    const [where, path] = spec.producer;
    const src = where === 'config' ? (cfg.quantization_config || {}) : (hfQuant || {});
    const v = dig(src, path);
    producerVersion = v === null || v === undefined ? null : String(v);
  }

  const files = await shardList(repoId);
  if (!files.length) {
    // A pack whose weights are a .pt or .bin cannot be read this way, and
    // must not be reported as unquantized.
    throw new Error('no safetensors weights published — torchao and HQQ '
      + 'packs often ship a .pt instead, and their contents cannot be read '
      + 'from metadata');
  }

  onProgress(0, files.length);

  // Read the shards a few at a time rather than one after another. A
  // 70B pack is nine shards and two ranged requests each; sequentially
  // that is long enough that the page reads as broken. Four at a time
  // is enough to hide most of the latency without opening a burst of
  // connections against the Hub for a page that is only reading
  // metadata.
  const tensors = [];
  let next = 0;
  let finished = 0;
  const worker = async () => {
    while (next < files.length) {
      const mine = files[next++];
      const got = await readHeader(repoId, mine);
      tensors.push(...got);
      onProgress(++finished, files.length);
    }
  };
  await Promise.all(
    Array.from({ length: Math.min(4, files.length) }, worker));

  return {
    repo: repoId,
    producerVersion,
    shards: files.length,
    tensorCount: tensors.length,
    ...summarize(tensors, method),
  };
}