From 1b560f1f500338f891d210314f8cec6c9e7f72c6 Mon Sep 17 00:00:00 2001 From: Nirupam Bhowmick <48842933+heydryft@users.noreply.github.com> Date: Wed, 19 Aug 2026 14:18:10 +0100 Subject: [PATCH 1/4] =?UTF-8?q?feat(qtip):=20UQFF=20geometry=20discriminat?= =?UTF-8?q?or=20=E2=80=94=20K8/V4/L12=20artifacts=20declare=20themselves?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stage 2 of the K=8/V=4/L=12 rung: an artifact can now say which trellis geometry it was baked at, the loader reads it, and a build that cannot decode it refuses instead of guessing. WHY THIS NEEDS A DISCRIMINATOR AT ALL Both geometries are 2 bits per weight (`bpw = K/V`), so a K=8/V=4 row and a K=4/V=2 row of the same `in_features` occupy the SAME number of packed bytes. A mislabelled artifact therefore indexes in bounds at every symbol and returns plausible garbage rather than faulting. The table is the only tensor that differs — `[4096, 4]` BF16 vs `[65536, 2]` F32 — which is why `validate_shapes` checks the table's dtype AND its length, and why both checks are load-bearing rather than belt-and-braces. WIRE FORMAT A trailing section `[tag=3, K, L, V]`, written ONLY for a non-default geometry and written BEFORE the codebook section. Both of those are deliberate: * Default writes nothing, so every artifact Arc has already produced stays byte-identical and no checksum moves. Pinned as an exact suffix, not as a vague "unchanged": serializing the same layer with the tag flipped appends exactly [3, 8, 12, 4] and nothing else. * BEFORE the codebook, because a build that predates this field parses the trailing region by handing its first byte to `QtipCodebook::from_wire`, which refuses every tag it does not know. Tag 3 first means an old Arc fails closed. Tag 3 last would let it consume the codebook section, stop, and decode K=8/V=4 symbols as K=4/V=2 without faulting. The trailing region is now a section loop; a repeated tag is refused rather than letting the last one win. `QtipGeometry` has no per-site default: adding the field broke all 11 construction sites at compile time and each one now states its geometry, the same discipline `search` and `search_detail` already follow. `from_stacked_parts` takes it explicitly and validates it. `stack_experts` and the 3-D quantize path refuse mixed-geometry stacks — only one table survives a stack, so at most one expert could decode. The computed `sum2` codebook is refused at V=4 on both the write and the read side: it produces a PAIR of values per state and has no V=4 form. Guards were mutation-tested (F1-F9). Three passed on broken code and are now covered: `from_stacked_parts` validated nothing, the table-dtype half of the discriminator was dead (the element-count check masked it), and `stack_experts`' mixed-geometry refusal was untested. `assert_tensor_bits_eq` grew a BF16 arm — it panicked on the new table rather than comparing it. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SpVNMpb13HkUXqSqbN1o9H --- .../examples/qtip_lut_grouped_bench.rs | 6 +- mistralrs-quant/src/lib.rs | 4 +- mistralrs-quant/src/qtip/mod.rs | 783 +++++++++++++++++- 3 files changed, 776 insertions(+), 17 deletions(-) diff --git a/mistralrs-quant/examples/qtip_lut_grouped_bench.rs b/mistralrs-quant/examples/qtip_lut_grouped_bench.rs index 98f710cb7..59f430db9 100644 --- a/mistralrs-quant/examples/qtip_lut_grouped_bench.rs +++ b/mistralrs-quant/examples/qtip_lut_grouped_bench.rs @@ -32,7 +32,8 @@ use candle_core::{DType, Device, Result, Tensor}; use mistralrs_quant::{ - QtipCodebook, QtipLayer, QtipSearchDetail, QtipSearchStamp, QuantMethod, QTIP_GROUPED_TILE_K, + QtipCodebook, QtipGeometry, QtipLayer, QtipSearchDetail, QtipSearchStamp, QuantMethod, + QTIP_GROUPED_TILE_K, }; use std::time::Instant; @@ -138,6 +139,9 @@ fn build_layer( QtipSearchStamp::Unstamped, QtipSearchDetail::Unknown, QtipCodebook::Gaussian, + // This bench synthesises a `[65536, 2]` F32 table above; that is the + // K=4/V=2/L=16 geometry by construction. + QtipGeometry::K4V2L16, ) } diff --git a/mistralrs-quant/src/lib.rs b/mistralrs-quant/src/lib.rs index 2ec9ae67f..b489f148a 100644 --- a/mistralrs-quant/src/lib.rs +++ b/mistralrs-quant/src/lib.rs @@ -120,8 +120,8 @@ pub use qtip::tune::{ pub use qtip::{ bake_cache, gpu_quantize_cpu_fallback_count, grouped_launch_counts, grouped_variant, hessian_row_weights, qtip_expected_distinct_experts, qtip_expected_pairs_per_distinct_expert, - qtip_grouped_gemm_tile_fill, set_grouped_variant, viterbi_quantize_row, - BakeCacheError, BakeKey, ExpertBpwTable, Qtip2bLayer, QtipBakeConfig, QtipCodebook, QtipLayer, + qtip_grouped_gemm_tile_fill, set_grouped_variant, viterbi_quantize_row, BakeCacheError, + BakeKey, ExpertBpwTable, Qtip2bLayer, QtipBakeConfig, QtipCodebook, QtipGeometry, QtipLayer, QtipMode, QtipPackedView, QtipRotation, QtipSearchDetail, QtipSearchStamp, TrellisBpw, TrellisSearch, QTIP2B_MCG_MULT, QTIP_GATHER_GEMV_MAX_PAIRS, QTIP_GROUPED_TILE_K, QTIP_GROUPED_TILE_M, QTIP_GROUPED_TILE_N, QTIP_GROUPED_VARIANT_BASELINE, diff --git a/mistralrs-quant/src/qtip/mod.rs b/mistralrs-quant/src/qtip/mod.rs index dcf87896e..17b492953 100644 --- a/mistralrs-quant/src/qtip/mod.rs +++ b/mistralrs-quant/src/qtip/mod.rs @@ -686,6 +686,10 @@ impl QtipCodebook { /// one. Unknown tags fail closed (DOCTRINE: refuse, never launder). fn from_wire(tag: u8, read_mult: impl FnOnce() -> Result) -> Result { match tag { + GEOMETRY_WIRE_TAG => candle_core::bail!( + "QtipLayer: internal error — the geometry section reached the codebook parser. \ + The trailing-section loop must consume tag {GEOMETRY_WIRE_TAG} itself." + ), 2 => { let mult = read_mult()?; if mult == 0 || mult % 2 == 0 { @@ -705,6 +709,227 @@ impl QtipCodebook { } } +/// Trailing-section tag for the trellis geometry. +/// +/// **It is 3, and it is written BEFORE the codebook section, and both of those +/// are load-bearing.** A build that predates this field parses the trailing +/// region by reading one byte and handing it to [`QtipCodebook::from_wire`], +/// which refuses every tag it does not know. So an old Arc handed a +/// non-default-geometry artifact sees tag 3 first and fails closed with +/// "written by a newer Arc … refusing rather than decoding its symbols against +/// the wrong codebook" — which is exactly the right outcome, because it cannot +/// decode this geometry either. +/// +/// Put the section *after* the codebook and that property is lost: an old +/// reader would consume the codebook section, stop, and decode a K=8/V=4 symbol +/// stream as K=4/V=2. It would not even fault — the packed byte count is +/// identical at 2 bpw, so every index stays in bounds and the output is +/// plausible garbage. [`tests::geometry_section_precedes_the_codebook_section`] +/// pins the order. +const GEOMETRY_WIRE_TAG: u8 = 3; + +/// Which trellis geometry a [`QtipLayer`]'s symbols were baked at. +/// +/// **This is a format discriminator, not a tuning knob.** Symbols are a +/// bit-stream interpreted by a specific `(K, L, V)` and a specific reproduction +/// table; read at the wrong geometry they decode to different weights, silently. +/// +/// Both variants are **2 bits per weight** (`bpw = K/V`). They are not a +/// compression trade — they are a decode-cost trade. See +/// [`crate::k8v4l12`] for what actually differs. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum QtipGeometry { + /// K=4 / V=2 / L=16 with a `[65536, 2]` F32 table — every QTIP artifact + /// Arc has ever written. The default, and it serializes as **nothing**. + #[default] + K4V2L16, + /// K=8 / V=4 / L=12 with a `[4096, 4]` BF16 table (32,768 B — fits static + /// shared memory). See [`crate::k8v4l12`]. + K8V4L12, +} + +impl QtipGeometry { + /// Bits per symbol. + pub fn k(self) -> u32 { + match self { + QtipGeometry::K4V2L16 => K, + QtipGeometry::K8V4L12 => k8v4l12::K, + } + } + + /// Trellis state width in bits. + pub fn l(self) -> u32 { + match self { + QtipGeometry::K4V2L16 => L, + QtipGeometry::K8V4L12 => k8v4l12::L, + } + } + + /// Reproduction values per symbol. + pub fn v(self) -> u32 { + match self { + QtipGeometry::K4V2L16 => V, + QtipGeometry::K8V4L12 => k8v4l12::V, + } + } + + /// Bits per weight. **2 for both variants** — that is the point. + pub fn n_bits(self) -> usize { + (self.k() / self.v()) as usize + } + + /// Trellis symbols packed into one byte: `8 / K`. + pub fn syms_per_byte(self) -> usize { + 8 / self.k() as usize + } + + /// Total values in the reproduction table: `2^L × V`. + pub fn lut_values(self) -> usize { + (1usize << self.l()) * self.v() as usize + } + + /// Element type of the reproduction table. + /// + /// Part of the discriminator, not an implementation detail: the K=8/V=4 + /// table is BF16 precisely so it lands at 32,768 B, and an F32 table of the + /// same shape is a different artifact. + pub fn lut_dtype(self) -> DType { + match self { + QtipGeometry::K4V2L16 => DType::F32, + QtipGeometry::K8V4L12 => DType::BF16, + } + } + + /// Trellis symbols in a row of `in_features` weights. + pub fn num_symbols(self, in_features: usize) -> usize { + in_features / self.v() as usize + } + + /// Packed bytes in a row of `in_features` weights. + /// + /// Equal for both variants at the same `in_features` — both are 2 bpw — + /// which is exactly why a mislabelled artifact does not fault. + pub fn packed_len(self, in_features: usize) -> usize { + self.num_symbols(in_features) / self.syms_per_byte() + } + + /// Short label for bake headers and error messages. + pub fn tag(self) -> &'static str { + match self { + QtipGeometry::K4V2L16 => "k4v2l16", + QtipGeometry::K8V4L12 => "k8v4l12", + } + } + + /// Wire encoding. **The default geometry writes nothing**, so every + /// artifact Arc has already produced stays byte-identical and no existing + /// checksum moves. Only a geometry that cannot be inferred from absence + /// announces itself. + fn to_wire(self) -> Option<(u8, [u8; 3])> { + match self { + QtipGeometry::K4V2L16 => None, + QtipGeometry::K8V4L12 => Some(( + GEOMETRY_WIRE_TAG, + [k8v4l12::K as u8, k8v4l12::L as u8, k8v4l12::V as u8], + )), + } + } + + /// Parse a `(K, L, V)` triple into a known geometry. + /// + /// The triple is stored rather than an opaque ordinal so a payload is + /// self-describing to a human with a hex editor, and so an unsupported but + /// well-formed geometry produces a diagnosis instead of "unknown tag". + fn from_wire_body(k: u8, l: u8, v: u8) -> Result { + // Written as comparisons, not as match patterns: a bare `K` in pattern + // position is a const pattern only for as long as the const exists, and + // silently becomes a catch-all binding if it is ever renamed away. + let got = (k as u32, l as u32, v as u32); + if got == (K, L, V) { + return Ok(QtipGeometry::K4V2L16); + } + if got == (k8v4l12::K, k8v4l12::L, k8v4l12::V) { + return Ok(QtipGeometry::K8V4L12); + } + let (ok, ol, ov) = got; + candle_core::bail!( + "QtipLayer: UQFF payload declares trellis geometry K={ok}/L={ol}/V={ov}, which this \ + build has no decoder for. Refusing rather than decoding its symbols at the wrong \ + geometry — at {} bits per weight the packed byte count is identical to a geometry \ + we DO support, so a wrong guess would not even fault.", + ok.checked_div(ov).unwrap_or(0) + ) + } + + /// Check that a layer's tensors are shaped the way this geometry requires. + /// + /// The tag alone is a claim; this is what makes it true. Both supported + /// geometries are 2 bpw, so `blocks` has the same size either way and the + /// *table* is the only shape that discriminates — hence the dtype and + /// element-count checks, which are the load-bearing half. + fn validate_shapes(self, blocks: &Tensor, lut: &Tensor, in_features: usize) -> Result<()> { + if !in_features.is_multiple_of(self.v() as usize) { + candle_core::bail!( + "QtipLayer[{}]: in_features {in_features} is not a multiple of V={}", + self.tag(), + self.v() + ); + } + let want_packed = self.packed_len(in_features); + let got_packed = blocks.dims()[blocks.dims().len() - 1]; + if got_packed != want_packed { + candle_core::bail!( + "QtipLayer[{}]: blocks row is {got_packed} B but K={}/V={} with in_features \ + {in_features} needs {want_packed} B", + self.tag(), + self.k(), + self.v() + ); + } + if lut.dtype() != self.lut_dtype() { + candle_core::bail!( + "QtipLayer[{}]: reproduction table is {:?} but this geometry stores {:?}. The \ + table's element type is part of the format, not an implementation detail.", + self.tag(), + lut.dtype(), + self.lut_dtype() + ); + } + if lut.elem_count() != self.lut_values() { + candle_core::bail!( + "QtipLayer[{}]: reproduction table has {} values, expected 2^{} × {} = {}", + self.tag(), + lut.elem_count(), + self.l(), + self.v(), + self.lut_values() + ); + } + Ok(()) + } + + /// Codebooks this geometry can legally carry. + /// + /// The computed `sum2` codebook is V=2-specific — `mcg_codeword_v2` folds + /// exactly two chained MCG products, and the CUDA twin + /// (`qtip_codebook.cuh::qtip_cb_pair_from_x0`) returns a `float2`. There is + /// no V=4 form of it, so pairing it with K=8/V=4 is not a thing that can be + /// decoded, and it is refused on both the write and the read side rather + /// than left to fail somewhere less obvious. + fn check_codebook(self, codebook: QtipCodebook) -> Result<()> { + match (self, codebook) { + (QtipGeometry::K8V4L12, QtipCodebook::Mcg { .. }) => candle_core::bail!( + "QtipLayer[{}]: the computed `sum2` codebook is V=2-only (it produces a PAIR of \ + reproduction values per state) and cannot describe a V={} geometry. This rung \ + is table-only.", + self.tag(), + self.v() + ), + _ => Ok(()), + } + } +} + /// QTIP 2-bit weight layer. /// /// # Storage layout — two modes @@ -799,6 +1024,16 @@ pub struct QtipLayer { /// computes, so the table stays a correct fallback (and the CPU search /// still wants a flat one) while every GPU path skips it entirely. codebook: QtipCodebook, + /// Which trellis geometry `blocks` was baked at, and therefore how its + /// bytes decompose into symbols and how `lut` is indexed. + /// + /// [`QtipGeometry::K4V2L16`] for every artifact written before the + /// discriminator existed — which is what they are. Deliberately has no + /// per-site default: both supported geometries are 2 bits per weight, so a + /// mislabelled layer has a correctly-sized `blocks` tensor and decodes to + /// plausible garbage rather than faulting. Every construction site states + /// it, and the compiler is what enforces that. + geometry: QtipGeometry, } /// Borrowed, dequantization-free view of a [`QtipLayer`]'s packed trellis @@ -1954,6 +2189,10 @@ impl QtipLayer { search_weights.is_some(), ), codebook, + // This crate's trellis search is K=4/V=2/L=16 throughout + // (`viterbi.rs` reaches 16 successors per step and groups by + // 2^(L-K)); there is no producer of any other geometry yet. + geometry: QtipGeometry::K4V2L16, }) } @@ -2220,6 +2459,7 @@ impl QtipLayer { let mut shared_rotation_block: usize = 0; let mut shared_search_detail = QtipSearchDetail::Unknown; let mut shared_codebook = QtipCodebook::Gaussian; + let mut shared_geometry = QtipGeometry::K4V2L16; // Per-expert streaming. The full stack is kept on CPU (caller passes // device=CPU for the 3-D experts to avoid the ~4GB dense BF16 transient @@ -2341,6 +2581,7 @@ impl QtipLayer { shared_rotation_block = layer.rotation_block; shared_search_detail = layer.search_detail; shared_codebook = layer.codebook; + shared_geometry = layer.geometry; } else { debug_assert_eq!(layer.lut.dims(), shared_lut.as_ref().unwrap().dims()); debug_assert_eq!(layer.rotation_block, shared_rotation_block); @@ -2369,6 +2610,18 @@ impl QtipLayer { shared_codebook.tag() ); } + // And the geometry, for the same reason and with the same + // teeth: at 2 bpw a mixed-geometry stack has consistently + // sized `blocks`, so nothing downstream would notice. + if layer.geometry != shared_geometry { + candle_core::bail!( + "QTIP 3-D quantize: expert chunk at {expert_idx} used geometry {} but \ + the stack already carries {} — refusing to build a stack whose experts \ + decode at different geometries.", + layer.geometry.tag(), + shared_geometry.tag() + ); + } } expert_idx += this_b; } @@ -2398,6 +2651,7 @@ impl QtipLayer { search: QtipSearchStamp::for_mode(mode), search_detail: shared_search_detail, codebook: shared_codebook, + geometry: shared_geometry, }) } @@ -3084,6 +3338,7 @@ impl QtipLayer { search: QtipSearchStamp, search_detail: QtipSearchDetail, codebook: QtipCodebook, + geometry: QtipGeometry, ) -> Result { if blocks.dims().len() != 3 { candle_core::bail!( @@ -3112,6 +3367,12 @@ impl QtipLayer { ); } let e = blocks.dim(0)?; + // The caller asserts a geometry; make it true here rather than at the + // first decode. Both supported geometries are 2 bits per weight, so + // `blocks` alone cannot tell them apart — the table's dtype and length + // are what discriminate. + geometry.validate_shapes(&blocks, &lut, in_features)?; + geometry.check_codebook(codebook)?; Ok(QtipLayer { search_detail, blocks, @@ -3124,6 +3385,7 @@ impl QtipLayer { rotation_block, search, codebook, + geometry, }) } @@ -3201,6 +3463,15 @@ impl QtipLayer { head.codebook.tag() ); } + // Only `head.lut` survives the stack here too, and a table is + // meaningful only at the geometry it was built for. + if layer.geometry != head.geometry { + candle_core::bail!( + "QtipLayer::stack_experts: layer {i} geometry={} != head {}", + layer.geometry.tag(), + head.geometry.tag() + ); + } } let blocks_refs: Vec<&Tensor> = per_expert_layers.iter().map(|l| &l.blocks).collect(); @@ -3225,6 +3496,7 @@ impl QtipLayer { search: head.search, search_detail: head.search_detail, codebook: head.codebook, + geometry: head.geometry, }) } @@ -3507,6 +3779,7 @@ impl QtipLayer { search: self.search, search_detail: self.search_detail, codebook: self.codebook, + geometry: self.geometry, }) } } @@ -3549,6 +3822,11 @@ impl QuantMethod for QtipLayer { // table holds the computed values. It only forgoes the // in-register decode. codebook: QtipCodebook::Gaussian, + // Same again for the geometry: the config carries no + // discriminator, and every payload that reached this factory + // before the field existed was K=4/V=2/L=16. Reading it as + // anything else would be inventing a claim. + geometry: QtipGeometry::K4V2L16, }), _ => candle_core::bail!("QtipLayer requires QuantMethodConfig::Qtip"), } @@ -3915,6 +4193,12 @@ impl QtipLayer { self.codebook } + /// Which trellis geometry this layer's symbols were baked at, and + /// therefore which decoder can read them. See [`QtipGeometry`]. + pub fn geometry(&self) -> QtipGeometry { + self.geometry + } + /// Dequantize the i-th expert's `[N, K_in]` BF16 weight matrix (3-D mode /// only). Internal use by `gather_forward` and friends; bails when called /// on a 2-D layer or with `expert_idx >= num_experts`. @@ -4025,6 +4309,9 @@ impl QtipLayer { // Gaussian default above) — correct for any table, and never // assumes a computed codebook we cannot verify. codebook: QtipCodebook::Gaussian, + // Safetensors QTIP checkpoints predate every geometry but this + // one, and carry no field that could say otherwise. + geometry: QtipGeometry::K4V2L16, })) } @@ -4180,18 +4467,57 @@ impl QtipLayer { // meant. An unknown tag value, on the other hand, IS refused: that // means a newer Arc wrote it, and decoding its symbols against a // codebook we guessed would be silent corruption. - let codebook = match buffer.read_u8() { - Ok(tag) => QtipCodebook::from_wire(tag, || { - buffer.read_u32::().map_err(|_| { - candle_core::Error::Msg( - "QtipLayer: codebook tag claims a computed codebook but the multiplier \ - is missing (truncated payload)." - .into(), - ) - }) - })?, - Err(_) => QtipCodebook::Gaussian, - }; + // The trailing region is a sequence of tagged sections, read until EOF. + // Each tag may appear at most once: a repeat means a corrupt or + // concatenated payload, and silently letting the last one win is how a + // reader ends up decoding against something nobody wrote. + let mut codebook: Option = None; + let mut geometry: Option = None; + while let Ok(tag) = buffer.read_u8() { + if tag == GEOMETRY_WIRE_TAG { + if geometry.is_some() { + candle_core::bail!( + "QtipLayer: the UQFF payload carries two geometry sections. Refusing a \ + payload that describes itself twice." + ); + } + let mut klv = [0u8; 3]; + for (i, slot) in klv.iter_mut().enumerate() { + *slot = buffer.read_u8().map_err(|_| { + candle_core::Error::Msg(format!( + "QtipLayer: geometry section is truncated (got {i} of 3 K/L/V bytes)." + )) + })?; + } + geometry = Some(QtipGeometry::from_wire_body(klv[0], klv[1], klv[2])?); + } else { + if codebook.is_some() { + candle_core::bail!( + "QtipLayer: the UQFF payload carries two codebook sections. Refusing a \ + payload that describes itself twice." + ); + } + codebook = Some(QtipCodebook::from_wire(tag, || { + buffer.read_u32::().map_err(|_| { + candle_core::Error::Msg( + "QtipLayer: codebook tag claims a computed codebook but the \ + multiplier is missing (truncated payload)." + .into(), + ) + }) + })?); + } + } + // Absence is the answer for both, and it is the same answer it has + // always been: the historical geometry, gathering from the stored table. + let codebook = codebook.unwrap_or(QtipCodebook::Gaussian); + let geometry = geometry.unwrap_or(QtipGeometry::K4V2L16); + + // The tag is a claim; these make it true. Both supported geometries are + // 2 bits per weight, so `blocks` is the same size either way — the + // table's dtype and length are what actually discriminate. + geometry.validate_shapes(&blocks, &lut, in_features)?; + geometry.check_codebook(codebook)?; Ok(( Self { @@ -4206,6 +4532,7 @@ impl QtipLayer { search, search_detail, codebook, + geometry, }, ext_bias, )) @@ -4281,10 +4608,25 @@ impl QuantizedSerde for QtipLayer { if let Some(w) = beam_width { buffer.extend(&w.to_le_bytes()); } + // The trellis geometry, written FIRST in the trailing-section region + // and only when it is not the historical K=4/V=2/L=16 — see + // `QtipGeometry::to_wire` and `GEOMETRY_WIRE_TAG` for why the order is + // load-bearing. In short: a build that predates this field reads one + // trailing byte and hands it to the codebook parser, which refuses + // every tag it does not know. Tag 3 first means an old Arc fails + // closed on a geometry it cannot decode; tag 3 last would let it + // decode K=8/V=4 symbols as K=4/V=2 without faulting, because at 2 bpw + // the packed byte count is the same. + self.geometry.check_codebook(self.codebook)?; + if let Some((geo_tag, klv)) = self.geometry.to_wire() { + buffer.push(geo_tag); + buffer.extend(&klv); + } // wave24-AU: the codebook discriminator, appended LAST and only when // the codebook is not the historical Gaussian one — see - // `QtipCodebook::to_wire`. A Gaussian artifact is therefore byte-for- - // byte what this writer produced before the field existed. + // `QtipCodebook::to_wire`. A Gaussian artifact at the default geometry + // is therefore byte-for-byte what this writer produced before either + // field existed. // // A build that predates the field, handed a computed-codebook payload, // parses identically up to here and then ignores a trailing section it @@ -5823,6 +6165,7 @@ mod tests { let mut shared_rotation_block: usize = 0; let mut shared_search_detail = QtipSearchDetail::Unknown; let mut shared_codebook = QtipCodebook::Gaussian; + let mut shared_geometry = QtipGeometry::K4V2L16; for expert_idx in 0..e { let expert_w = w.narrow(0, expert_idx, 1)?.squeeze(0)?; let layer = QtipLayer::quantize_with_options_concrete( @@ -5840,6 +6183,7 @@ mod tests { shared_rotation_block = layer.rotation_block; shared_search_detail = layer.search_detail; shared_codebook = layer.codebook; + shared_geometry = layer.geometry; } } let blocks_3d = Tensor::stack(&blocks_slices, 0)?; @@ -5856,6 +6200,7 @@ mod tests { search: QtipSearchStamp::for_mode(mode), search_detail: shared_search_detail, codebook: shared_codebook, + geometry: shared_geometry, }) .inspect(|l| { debug_assert_eq!(l.blocks.dims(), &[e, n, k_in / 4]); @@ -6742,11 +7087,421 @@ mod tests { let bb: Vec = bv.iter().map(|x| x.to_bits()).collect(); assert_eq!(ab, bb, "{what}: f32 bit patterns differ"); } + DType::BF16 => { + // The K=8/V=4/L=12 reproduction table. Compared as raw bits, + // not as floats: a table is format, and `==` on bf16 would + // still call two NaN payloads unequal and two zeros equal. + let av: Vec = a.flatten_all()?.to_vec1()?; + let bv: Vec = b.flatten_all()?.to_vec1()?; + let ab: Vec = av.iter().map(|x| x.to_bits()).collect(); + let bb: Vec = bv.iter().map(|x| x.to_bits()).collect(); + assert_eq!(ab, bb, "{what}: bf16 bit patterns differ"); + } other => panic!("assert_tensor_bits_eq: unhandled dtype {other:?}"), } Ok(()) } + // =================================================================== + // Geometry discriminator (UQFF) + // =================================================================== + + /// A cheap 2-D layer whose fields we can then rewrite for format tests. + fn geometry_fixture(device: &Device) -> Result { + let (n, k_in) = (4usize, 64usize); + let wdata: Vec = (0..(n * k_in)) + .map(|i| ((i as f32) * 0.37).sin() * 1.1) + .collect(); + let w = Tensor::from_vec(wdata, (n, k_in), device)?; + QtipLayer::quantize_with_options_concrete(&w, None, device, QtipMode::Greedy, false) + } + + /// The default geometry must be **purely additive**: it writes nothing, so + /// every artifact Arc has already produced is byte-identical and no + /// checksum moves. Stated as an exact suffix rather than as a vague + /// "unchanged", so the tag value and its position are both pinned. + #[test] + fn default_geometry_writes_nothing_and_k8v4l12_appends_exactly_four_bytes() -> Result<()> { + let device = Device::Cpu; + let mut layer = geometry_fixture(&device)?; + assert_eq!(layer.geometry, QtipGeometry::K4V2L16); + let base = layer.serialize()?.into_owned(); + + // Flip ONLY the discriminator; every tensor stays byte-identical. + layer.geometry = QtipGeometry::K8V4L12; + let tagged = layer.serialize()?.into_owned(); + + let mut want = base.clone(); + want.extend_from_slice(&[GEOMETRY_WIRE_TAG, 8, 12, 4]); + assert_eq!( + tagged, want, + "the geometry section must be exactly [tag=3, K=8, L=12, V=4] appended to an \ + otherwise unchanged payload" + ); + Ok(()) + } + + /// The section must come BEFORE the codebook section, because that is what + /// makes an older Arc fail closed instead of mis-decoding. + #[test] + fn geometry_section_precedes_the_codebook_section() -> Result<()> { + let device = Device::Cpu; + let mut layer = geometry_fixture(&device)?; + let base = layer.serialize()?.into_owned(); + + // Codebook alone (K=4 geometry, so Mcg is legal here). + layer.codebook = QtipCodebook::COMPUTED; + let cb_only = layer.serialize()?.into_owned(); + assert_eq!( + cb_only.len(), + base.len() + 5, + "codebook section is tag + u32" + ); + assert_eq!(cb_only[base.len()], 2, "codebook tag is 2"); + + // Both sections. Geometry (4 bytes) must land first. + // + // K8V4L12 + Mcg is refused as a pair, so use a hypothetical third + // geometry is not available — instead assert the writer's ORDER + // directly by checking that the geometry bytes occupy the slot the + // codebook otherwise would. + layer.codebook = QtipCodebook::Gaussian; + layer.geometry = QtipGeometry::K8V4L12; + let geo_only = layer.serialize()?.into_owned(); + assert_eq!( + geo_only[base.len()], + GEOMETRY_WIRE_TAG, + "the geometry section must start immediately after the search-detail section — \ + any byte between them would be read as a codebook tag by an older Arc" + ); + Ok(()) + } + + /// An Arc that predates this field parses the trailing region by handing + /// its first byte to `QtipCodebook::from_wire`. That must REFUSE tag 3. + /// + /// If it ever accepted it, an old build would decode K=8/V=4 symbols as + /// K=4/V=2 — and would not fault, because at 2 bits per weight the packed + /// byte count is identical. + #[test] + fn an_older_reader_refuses_the_geometry_tag_rather_than_ignoring_it() { + let err = QtipCodebook::from_wire(GEOMETRY_WIRE_TAG, || Ok(1)) + .expect_err("tag 3 must not parse as a codebook"); + let msg = err.to_string(); + assert!( + msg.contains("geometry") || msg.contains("unknown codebook tag"), + "refusal must name the cause: {msg}" + ); + // And the general property the whole scheme rests on: every tag this + // build does not know is refused, never skipped. + for tag in [1u8, 4, 5, 200, 255] { + assert!( + QtipCodebook::from_wire(tag, || Ok(1)).is_err(), + "tag {tag} must fail closed" + ); + } + } + + /// A K=8/V=4 tag on a K=4/V=2 payload must be caught by the SHAPES, not + /// merely trusted. This is the case a hex-editor edit or a truncated + /// concatenation produces. + #[test] + fn a_geometry_tag_that_contradicts_the_tensors_is_refused_at_load() -> Result<()> { + let device = Device::Cpu; + let mut layer = geometry_fixture(&device)?; + layer.geometry = QtipGeometry::K8V4L12; + let data = layer.serialize()?.into_owned(); + + let err = QtipLayer::deserialize_concrete_unchecked( + Cow::Owned(data), + &device, + crate::QuantizeOntoGuard::new(), + ) + .expect_err("a K8V4L12 tag over a [65536, 2] F32 table must be refused"); + let msg = err.to_string(); + assert!( + msg.contains("k8v4l12") && (msg.contains("F32") || msg.contains("values")), + "refusal must name the geometry and the mismatch: {msg}" + ); + Ok(()) + } + + /// The table's ELEMENT TYPE discriminates, not only its length. + /// + /// A `[4096, 4]` F32 table has exactly the right number of values and the + /// wrong dtype — 65,536 B instead of 32,768 B, so it would not even fit + /// the shared-memory budget the geometry exists for. Without this case the + /// element-count check alone passes and the dtype check is dead + /// (measured: mutation F8 disabled it and every test stayed green). + #[test] + fn a_right_sized_table_of_the_wrong_dtype_is_refused() -> Result<()> { + let device = Device::Cpu; + let geo = QtipGeometry::K8V4L12; + let k_in = 64usize; + let blocks = Tensor::zeros((2, geo.packed_len(k_in)), DType::U8, &device)?; + let f32_table = Tensor::zeros( + (k8v4l12::LUT_STATES, k8v4l12::V as usize), + DType::F32, + &device, + )?; + assert_eq!(f32_table.elem_count(), geo.lut_values(), "same count"); + let err = geo + .validate_shapes(&blocks, &f32_table, k_in) + .expect_err("an F32 table at this geometry must be refused"); + assert!( + err.to_string().contains("F32") && err.to_string().contains("BF16"), + "refusal must name both dtypes: {err}" + ); + Ok(()) + } + + /// `from_stacked_parts` is a public constructor that takes the geometry on + /// trust from its caller; it must validate, not trust. + #[test] + fn from_stacked_parts_refuses_a_geometry_its_tensors_contradict() -> Result<()> { + let device = Device::Cpu; + let (e, n, k_in) = (2usize, 4usize, 64usize); + let blocks = Tensor::zeros((e, n, k_in / 4), DType::U8, &device)?; + let scales = Tensor::zeros((e, n), DType::F32, &device)?; + let k4_table = Tensor::zeros((1usize << L, V as usize), DType::F32, &device)?; + + // Truthful call: K=4 tensors, K=4 tag. + QtipLayer::from_stacked_parts( + blocks.clone(), + scales.clone(), + k4_table.clone(), + None, + k_in, + None, + 0, + QtipSearchStamp::Unstamped, + QtipSearchDetail::Unknown, + QtipCodebook::Gaussian, + QtipGeometry::K4V2L16, + )?; + + // Same tensors, K=8 tag. + let err = QtipLayer::from_stacked_parts( + blocks, + scales, + k4_table, + None, + k_in, + None, + 0, + QtipSearchStamp::Unstamped, + QtipSearchDetail::Unknown, + QtipCodebook::Gaussian, + QtipGeometry::K8V4L12, + ) + .expect_err("a K8V4L12 tag over a K4V2L16 table must be refused at construction"); + assert!(err.to_string().contains("k8v4l12"), "{err}"); + Ok(()) + } + + /// A stack whose experts were baked at different geometries keeps only one + /// table, so at most one expert could ever decode correctly. + #[test] + fn stack_experts_refuses_mixed_geometries() -> Result<()> { + let device = Device::Cpu; + let a = geometry_fixture(&device)?; + let mut b = geometry_fixture(&device)?; + // Identical in every respect except the discriminator, so nothing but + // the geometry check can be what fires. + b.geometry = QtipGeometry::K8V4L12; + let err = QtipLayer::stack_experts(vec![a, b]) + .expect_err("mixed-geometry stacks must be refused"); + assert!( + err.to_string().contains("geometry"), + "refusal must name the geometry: {err}" + ); + + // ...and a uniform stack still works, so the guard is not just + // rejecting everything. + let c = geometry_fixture(&device)?; + let d = geometry_fixture(&device)?; + QtipLayer::stack_experts(vec![c, d])?; + Ok(()) + } + + /// A genuine K=8/V=4/L=12 payload round-trips, tag and tensors both. + #[test] + fn k8v4l12_payload_round_trips_exactly() -> Result<()> { + let device = Device::Cpu; + let (n, k_in) = (4usize, 128usize); + let geo = QtipGeometry::K8V4L12; + let packed_per_row = geo.packed_len(k_in); + assert_eq!(packed_per_row, k_in / 4, "2 bpw either way"); + + let blocks_data: Vec = (0..(n * packed_per_row)) + .map(|i| (i * 37 % 251) as u8) + .collect(); + let blocks = Tensor::from_vec(blocks_data, (n, packed_per_row), &device)?; + let scales = Tensor::from_vec( + (0..n).map(|i| 0.01 + i as f32 * 0.003).collect::>(), + (n,), + &device, + )?; + let lut = Tensor::from_slice( + &k8v4l12::gaussian_lut_bf16(), + (k8v4l12::LUT_STATES, k8v4l12::V as usize), + &device, + )?; + + let layer = QtipLayer { + blocks, + row_scales: scales, + lut, + bias: None, + in_features: k_in, + num_experts: None, + rotation_signs: None, + rotation_block: 0, + search: QtipSearchStamp::Trellis, + search_detail: QtipSearchDetail::Known { + beam_width: None, + hessian: false, + }, + codebook: QtipCodebook::Gaussian, + geometry: geo, + }; + + let data = layer.serialize()?.into_owned(); + let (restored, _) = QtipLayer::deserialize_concrete_unchecked( + Cow::Owned(data), + &device, + crate::QuantizeOntoGuard::new(), + )?; + + assert_eq!(restored.geometry, QtipGeometry::K8V4L12); + assert_eq!(restored.in_features, k_in); + assert_eq!(restored.lut.dtype(), DType::BF16); + assert_eq!(restored.lut.elem_count(), k8v4l12::LUT_ENTRIES); + assert_tensor_bits_eq(&layer.blocks, &restored.blocks, "blocks")?; + assert_tensor_bits_eq(&layer.row_scales, &restored.row_scales, "row_scales")?; + assert_tensor_bits_eq(&layer.lut, &restored.lut, "lut")?; + Ok(()) + } + + /// The computed `sum2` codebook produces a PAIR of values per state and so + /// cannot describe V=4. Refused on write and on read, not left to fail at + /// the first decode. + #[test] + fn the_computed_codebook_is_refused_at_a_v4_geometry() -> Result<()> { + let device = Device::Cpu; + let mut layer = geometry_fixture(&device)?; + layer.geometry = QtipGeometry::K8V4L12; + layer.codebook = QtipCodebook::COMPUTED; + let err = layer + .serialize() + .expect_err("V=4 + sum2 must not serialize"); + assert!( + err.to_string().contains("V=2-only"), + "refusal must say why: {err}" + ); + // And it is legal at the geometry it was built for, so the guard is + // about the pairing and not about the codebook being blocked outright. + layer.geometry = QtipGeometry::K4V2L16; + layer.serialize()?; + Ok(()) + } + + /// Malformed trailing regions are refused rather than partially read. + #[test] + fn malformed_geometry_sections_are_refused() -> Result<()> { + let device = Device::Cpu; + let layer = geometry_fixture(&device)?; + let base = layer.serialize()?.into_owned(); + let load = |bytes: Vec| { + QtipLayer::deserialize_concrete_unchecked( + Cow::Owned(bytes), + &device, + crate::QuantizeOntoGuard::new(), + ) + }; + + // Truncated: tag present, K/L/V missing. + let mut truncated = base.clone(); + truncated.push(GEOMETRY_WIRE_TAG); + truncated.push(8); + let err = load(truncated).expect_err("truncated geometry section must be refused"); + assert!(err.to_string().contains("truncated"), "{err}"); + + // Well-formed but unsupported triple: diagnosed, not "unknown tag". + let mut unsupported = base.clone(); + unsupported.extend_from_slice(&[GEOMETRY_WIRE_TAG, 6, 18, 3]); + let err = load(unsupported).expect_err("unsupported geometry must be refused"); + let msg = err.to_string(); + assert!( + msg.contains("K=6") && msg.contains("L=18") && msg.contains("V=3"), + "refusal must quote the triple it could not decode: {msg}" + ); + + // Two geometry sections: a payload that describes itself twice. + let mut doubled = base.clone(); + doubled.extend_from_slice(&[GEOMETRY_WIRE_TAG, 4, 16, 2]); + doubled.extend_from_slice(&[GEOMETRY_WIRE_TAG, 4, 16, 2]); + let err = load(doubled).expect_err("duplicate geometry sections must be refused"); + assert!(err.to_string().contains("twice"), "{err}"); + + // An explicit default-geometry section is legal (a future writer may + // choose to be explicit) and must load as K4V2L16. + let mut explicit = base.clone(); + explicit.extend_from_slice(&[GEOMETRY_WIRE_TAG, 4, 16, 2]); + let (restored, _) = load(explicit)?; + assert_eq!(restored.geometry, QtipGeometry::K4V2L16); + Ok(()) + } + + /// Both geometries are 2 bits per weight. This is the fact that makes the + /// discriminator necessary — a mislabelled artifact has a correctly-sized + /// `blocks` tensor — so it gets an assertion rather than a comment. + #[test] + fn both_geometries_are_two_bits_per_weight_and_the_same_packed_size() { + for k_in in [64usize, 512, 4096, 7168] { + assert_eq!( + QtipGeometry::K4V2L16.packed_len(k_in), + QtipGeometry::K8V4L12.packed_len(k_in), + "k_in={k_in}: identical packed size is why the tag is load-bearing" + ); + assert_eq!(QtipGeometry::K4V2L16.n_bits(), 2); + assert_eq!(QtipGeometry::K8V4L12.n_bits(), 2); + } + // ...and the tables are what actually differ. + assert_eq!(QtipGeometry::K4V2L16.lut_values(), 65_536 * 2); + assert_eq!(QtipGeometry::K8V4L12.lut_values(), 4_096 * 4); + assert_eq!(QtipGeometry::K4V2L16.lut_dtype(), DType::F32); + assert_eq!(QtipGeometry::K8V4L12.lut_dtype(), DType::BF16); + assert_eq!(QtipGeometry::K4V2L16.syms_per_byte(), 2); + assert_eq!(QtipGeometry::K8V4L12.syms_per_byte(), 1); + } + + /// The geometry accessors must agree with the rung module they describe. + #[test] + fn geometry_accessors_match_the_rung_modules() { + assert_eq!(QtipGeometry::K4V2L16.k(), K); + assert_eq!(QtipGeometry::K4V2L16.l(), L); + assert_eq!(QtipGeometry::K4V2L16.v(), V); + assert_eq!(QtipGeometry::K8V4L12.k(), k8v4l12::K); + assert_eq!(QtipGeometry::K8V4L12.l(), k8v4l12::L); + assert_eq!(QtipGeometry::K8V4L12.v(), k8v4l12::V); + assert_eq!( + QtipGeometry::K8V4L12.lut_values(), + k8v4l12::LUT_ENTRIES, + "the discriminator and the rung must agree on the table size" + ); + for k_in in [64usize, 512, 4096] { + assert_eq!( + QtipGeometry::K8V4L12.packed_len(k_in), + k8v4l12::packed_len(k_in) + ); + assert_eq!( + QtipGeometry::K8V4L12.num_symbols(k_in), + k8v4l12::num_symbols(k_in) + ); + } + } + /// UQFF round-trip for the classic 2-D layout must stay lossless and /// byte-compatible (bias branch + no-rotation branch covered). #[test] From a8a5b9dcac4d3f0c03eb232ccacdf65b2669d0fa Mon Sep 17 00:00:00 2001 From: Nirupam Bhowmick <48842933+heydryft@users.noreply.github.com> Date: Wed, 19 Aug 2026 14:22:17 +0100 Subject: [PATCH 2/4] fix(qtip): refuse an unsupported geometry at every decode entry point MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stage 2 made a K=8/V=4/L=12 artifact loadable. On its own that is a half-applied change: every decoder behind `forward`, `gather_forward`, `dequantize_weights`, `dequantize_expert` and `qtip_packed` unpacks nibbles and indexes a [2^16, 2] table, and handed a K=8 row none of them would fault — at 2 bits per weight the packed byte count is identical and every index lands in bounds. They would serve garbage. So each entry point now states the geometries it implements. `qtip_packed` returns None rather than a `QtipPackedView`, which carries no geometry field and would therefore be read as K=4/V=2 nibbles by any consumer. The refusal names the entry point, and the test asserts that name. An entry point that merely inherits a callee's guard is not guarded: mutation G1 removed `forward`'s own check and every test stayed green, because the error still arrived from `dequantize_weights` further down. All six mutations (G1-G6) are red now. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SpVNMpb13HkUXqSqbN1o9H --- mistralrs-quant/src/qtip/mod.rs | 568 ++++++++++++++++++++++++-------- 1 file changed, 431 insertions(+), 137 deletions(-) diff --git a/mistralrs-quant/src/qtip/mod.rs b/mistralrs-quant/src/qtip/mod.rs index 17b492953..dec9a4848 100644 --- a/mistralrs-quant/src/qtip/mod.rs +++ b/mistralrs-quant/src/qtip/mod.rs @@ -734,26 +734,47 @@ const GEOMETRY_WIRE_TAG: u8 = 3; /// bit-stream interpreted by a specific `(K, L, V)` and a specific reproduction /// table; read at the wrong geometry they decode to different weights, silently. /// -/// Both variants are **2 bits per weight** (`bpw = K/V`). They are not a -/// compression trade — they are a decode-cost trade. See -/// [`crate::k8v4l12`] for what actually differs. +/// `K` is carried as data, not baked into the variant, because that is what the +/// wire already says — the section is `[tag, K, L, V]`. Within the V=4/L=12 +/// family the table is K-independent, so a new K is a new value here and not a +/// new variant. See [`crate::trellis_v4l12`]. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub enum QtipGeometry { /// K=4 / V=2 / L=16 with a `[65536, 2]` F32 table — every QTIP artifact /// Arc has ever written. The default, and it serializes as **nothing**. #[default] K4V2L16, - /// K=8 / V=4 / L=12 with a `[4096, 4]` BF16 table (32,768 B — fits static - /// shared memory). See [`crate::k8v4l12`]. - K8V4L12, + /// The V=4 / L=12 family: a `[4096, 4]` **BF16** table (32,768 B — fits + /// static shared memory), with `k` bits per symbol. + /// + /// `k = 8` is the byte-aligned control at 2.00 bpw; `k = 9` is 2.25 bpw and + /// is the quality winner. Constructed only through + /// [`QtipGeometry::trellis_v4l12`], which refuses a `k` with no decoder. + TrellisV4L12 { + /// The decode rung, which owns the K-dependent arithmetic. Stored + /// rather than a bare `k` so the packed-size formula has exactly one + /// implementation — see [`QtipGeometry::packed_len`]. + rung: trellis_v4l12::Rung, + }, } impl QtipGeometry { + /// The V=4/L=12 family at symbol width `k`, refusing a `k` this build has + /// no decoder for. + /// + /// Refuses rather than storing it: `k` arrives from an artifact, and a + /// geometry we cannot decode must not be constructible as a value that + /// later code will try to use. + pub fn trellis_v4l12(k: u32) -> Result { + let rung = trellis_v4l12::Rung::new(k).map_err(candle_core::Error::Msg)?; + Ok(QtipGeometry::TrellisV4L12 { rung }) + } + /// Bits per symbol. pub fn k(self) -> u32 { match self { QtipGeometry::K4V2L16 => K, - QtipGeometry::K8V4L12 => k8v4l12::K, + QtipGeometry::TrellisV4L12 { rung } => rung.k(), } } @@ -761,7 +782,7 @@ impl QtipGeometry { pub fn l(self) -> u32 { match self { QtipGeometry::K4V2L16 => L, - QtipGeometry::K8V4L12 => k8v4l12::L, + QtipGeometry::TrellisV4L12 { .. } => trellis_v4l12::L, } } @@ -769,18 +790,17 @@ impl QtipGeometry { pub fn v(self) -> u32 { match self { QtipGeometry::K4V2L16 => V, - QtipGeometry::K8V4L12 => k8v4l12::V, + QtipGeometry::TrellisV4L12 { .. } => trellis_v4l12::V, } } - /// Bits per weight. **2 for both variants** — that is the point. - pub fn n_bits(self) -> usize { - (self.k() / self.v()) as usize - } - - /// Trellis symbols packed into one byte: `8 / K`. - pub fn syms_per_byte(self) -> usize { - 8 / self.k() as usize + /// Bits per weight × 100 (`100·K/V`), as an integer so it compares exactly. + /// + /// 200 for the shipped rung and for K=8/V=4; **225 for K=9/V=4**; 250 for + /// K=10/V=4. Not every geometry in this enum is the same bit rate any more, + /// which is a change from when the family was K=8-only. + pub fn bpw_x100(self) -> u32 { + self.k() * 100 / self.v() } /// Total values in the reproduction table: `2^L × V`. @@ -790,34 +810,45 @@ impl QtipGeometry { /// Element type of the reproduction table. /// - /// Part of the discriminator, not an implementation detail: the K=8/V=4 - /// table is BF16 precisely so it lands at 32,768 B, and an F32 table of the - /// same shape is a different artifact. + /// Part of the discriminator, not an implementation detail: the V=4 table + /// is BF16 precisely so it lands at 32,768 B, and an F32 table of the same + /// shape is a different artifact. pub fn lut_dtype(self) -> DType { match self { QtipGeometry::K4V2L16 => DType::F32, - QtipGeometry::K8V4L12 => DType::BF16, + QtipGeometry::TrellisV4L12 { .. } => DType::BF16, } } - /// Trellis symbols in a row of `in_features` weights. + /// Trellis symbols in a row of `in_features` weights. Depends only on V. pub fn num_symbols(self, in_features: usize) -> usize { in_features / self.v() as usize } - /// Packed bytes in a row of `in_features` weights. + /// Packed bytes in a row of `in_features` weights: + /// `ceil(num_symbols · K / 8)`. /// - /// Equal for both variants at the same `in_features` — both are 2 bpw — - /// which is exactly why a mislabelled artifact does not fault. + /// **Delegated, not restated.** `num_symbols / (8 / K)` is wrong at K=9 — + /// there is no whole number of symbols per byte — and divides by zero, so + /// the formula has to be the general one. It is also invisible to test at + /// realistic shapes: every plausible `in_features` is a multiple of 32, so + /// `num_symbols · K` is a multiple of 8 and floor equals ceil. A duplicate + /// of this formula here would therefore be unguarded in practice (measured: + /// mutation W3 floored it and every format test stayed green), so the + /// family's own [`trellis_v4l12::Rung`] is the single implementation and it + /// is exercised at non-byte-aligned symbol counts by that module's tests. pub fn packed_len(self, in_features: usize) -> usize { - self.num_symbols(in_features) / self.syms_per_byte() + match self { + QtipGeometry::K4V2L16 => self.num_symbols(in_features) / 2, + QtipGeometry::TrellisV4L12 { rung } => rung.packed_len(in_features), + } } /// Short label for bake headers and error messages. - pub fn tag(self) -> &'static str { + pub fn tag(self) -> String { match self { - QtipGeometry::K4V2L16 => "k4v2l16", - QtipGeometry::K8V4L12 => "k8v4l12", + QtipGeometry::K4V2L16 => "k4v2l16".to_string(), + QtipGeometry::TrellisV4L12 { rung } => format!("k{}v4l12", rung.k()), } } @@ -828,9 +859,13 @@ impl QtipGeometry { fn to_wire(self) -> Option<(u8, [u8; 3])> { match self { QtipGeometry::K4V2L16 => None, - QtipGeometry::K8V4L12 => Some(( + QtipGeometry::TrellisV4L12 { rung } => Some(( GEOMETRY_WIRE_TAG, - [k8v4l12::K as u8, k8v4l12::L as u8, k8v4l12::V as u8], + [ + rung.k() as u8, + trellis_v4l12::L as u8, + trellis_v4l12::V as u8, + ], )), } } @@ -848,25 +883,26 @@ impl QtipGeometry { if got == (K, L, V) { return Ok(QtipGeometry::K4V2L16); } - if got == (k8v4l12::K, k8v4l12::L, k8v4l12::V) { - return Ok(QtipGeometry::K8V4L12); + if got.1 == trellis_v4l12::L && got.2 == trellis_v4l12::V { + // In-family: K is the free parameter, so let the family say + // whether it has a decoder for this one. + return QtipGeometry::trellis_v4l12(got.0); } let (ok, ol, ov) = got; candle_core::bail!( "QtipLayer: UQFF payload declares trellis geometry K={ok}/L={ol}/V={ov}, which this \ build has no decoder for. Refusing rather than decoding its symbols at the wrong \ - geometry — at {} bits per weight the packed byte count is identical to a geometry \ - we DO support, so a wrong guess would not even fault.", - ok.checked_div(ov).unwrap_or(0) + geometry — the packed byte count alone does not identify a geometry, so a wrong \ + guess need not even fault.", ) } /// Check that a layer's tensors are shaped the way this geometry requires. /// - /// The tag alone is a claim; this is what makes it true. Both supported - /// geometries are 2 bpw, so `blocks` has the same size either way and the - /// *table* is the only shape that discriminates — hence the dtype and - /// element-count checks, which are the load-bearing half. + /// The tag alone is a claim; this is what makes it true. The table's dtype + /// and length are the load-bearing half: K=4/V=2 and K=8/V=4 are both + /// 2 bits per weight, so `blocks` has the identical size either way and + /// cannot discriminate them. fn validate_shapes(self, blocks: &Tensor, lut: &Tensor, in_features: usize) -> Result<()> { if !in_features.is_multiple_of(self.v() as usize) { candle_core::bail!( @@ -913,12 +949,12 @@ impl QtipGeometry { /// The computed `sum2` codebook is V=2-specific — `mcg_codeword_v2` folds /// exactly two chained MCG products, and the CUDA twin /// (`qtip_codebook.cuh::qtip_cb_pair_from_x0`) returns a `float2`. There is - /// no V=4 form of it, so pairing it with K=8/V=4 is not a thing that can be - /// decoded, and it is refused on both the write and the read side rather - /// than left to fail somewhere less obvious. + /// no V=4 form of it, so pairing it with any V≠2 geometry is not a thing + /// that can be decoded, and it is refused on both the write and the read + /// side rather than left to fail somewhere less obvious. fn check_codebook(self, codebook: QtipCodebook) -> Result<()> { - match (self, codebook) { - (QtipGeometry::K8V4L12, QtipCodebook::Mcg { .. }) => candle_core::bail!( + match codebook { + QtipCodebook::Mcg { .. } if self.v() != 2 => candle_core::bail!( "QtipLayer[{}]: the computed `sum2` codebook is V=2-only (it produces a PAIR of \ reproduction values per state) and cannot describe a V={} geometry. This rung \ is table-only.", @@ -2727,6 +2763,7 @@ impl QtipLayer { } pub fn dequantize_weights(&self) -> Result { + self.require_k4v2l16("dequantize_weights")?; // 3-D stacked-expert path: iterate over experts and stack `[N, K_in]` // slices into `[E, N, K_in]`. We reuse the 2-D dequant per expert // rather than writing a 3-D CUDA kernel — the existing @@ -3837,10 +3874,17 @@ impl QuantMethod for QtipLayer { } fn qtip_packed(&self) -> Option> { - Some(self.packed_view()) + // `QtipPackedView` carries no geometry field, so a consumer would read + // these bytes as K=4/V=2 nibbles. Hand out nothing rather than + // something that cannot describe itself. + match self.geometry { + QtipGeometry::K4V2L16 => Some(self.packed_view()), + _ => None, + } } fn forward(&self, x: &Tensor) -> Result { + self.require_k4v2l16("forward")?; self.forward_dequantize(x) } @@ -3884,6 +3928,7 @@ impl QuantMethod for QtipLayer { /// dequantized weight in the rotated frame and matching it with a /// rotated activation — saves one rotation pass per expert. fn gather_forward(&self, a: &Tensor, indices: &Tensor) -> Result { + self.require_k4v2l16("gather_forward")?; if self.num_experts.is_none() { // The contract is "expert-sparse dispatch" — a 2-D layer is a // single-expert (i.e. non-MoE) layer and `gather_forward` @@ -4199,10 +4244,41 @@ impl QtipLayer { self.geometry } + /// Refuse any decode path that only knows K=4/V=2/L=16. + /// + /// **This is what keeps stage 2 from being a half-applied change.** The + /// geometry discriminator makes a K=8/V=4/L=12 artifact *loadable*; the + /// decoders that would then read it — `dequantize_weights_rotated_f32`, + /// `dequantize_single_expert`, `gather_forward_cpu`, and every CUDA + /// launcher in `cuda_ops` other than `fused_gemv_k8v4l12_cuda` — all + /// unpack nibbles and index a `[2^16, 2]` table. Handed a K=8 row they + /// would not fault: at 2 bits per weight the byte count is identical and + /// every index lands in bounds. They would serve garbage. + /// + /// So every entry point into decode states the geometries it can handle. + /// A layer this build cannot serve is refused loudly at the door rather + /// than quietly at the logits. + fn require_k4v2l16(&self, op: &str) -> Result<()> { + match self.geometry { + QtipGeometry::K4V2L16 => Ok(()), + other => candle_core::bail!( + "QtipLayer::{op}: this layer was baked at geometry {}, and {op} only implements \ + K=4/V=2/L=16. The {} decode path is not wired into serving yet — the fused \ + GEMV kernel exists (`kernels/qtip/qtip_gemv_k8v4l12.cu`) but nothing dispatches \ + to it. Refusing rather than decoding these symbols as K=4/V=2, which would not \ + fault (both geometries are 2 bits per weight, so the packed byte count is the \ + same) and would silently serve wrong weights.", + other.tag(), + other.tag() + ), + } + } + /// Dequantize the i-th expert's `[N, K_in]` BF16 weight matrix (3-D mode /// only). Internal use by `gather_forward` and friends; bails when called /// on a 2-D layer or with `expert_idx >= num_experts`. pub fn dequantize_expert(&self, expert_idx: usize) -> Result { + self.require_k4v2l16("dequantize_expert")?; let e = self.num_experts.ok_or_else(|| { candle_core::Error::Msg("QtipLayer::dequantize_expert called on a 2-D layer".into()) })?; @@ -7106,6 +7182,11 @@ mod tests { // Geometry discriminator (UQFF) // =================================================================== + /// The V=4/L=12 family at the byte-aligned control K, as a geometry value. + fn k8() -> QtipGeometry { + QtipGeometry::trellis_v4l12(8).unwrap() + } + /// A cheap 2-D layer whose fields we can then rewrite for format tests. fn geometry_fixture(device: &Device) -> Result { let (n, k_in) = (4usize, 64usize); @@ -7119,25 +7200,58 @@ mod tests { /// The default geometry must be **purely additive**: it writes nothing, so /// every artifact Arc has already produced is byte-identical and no /// checksum moves. Stated as an exact suffix rather than as a vague - /// "unchanged", so the tag value and its position are both pinned. + /// "unchanged", so the tag value, its position, and the K byte are all + /// pinned — **at every K in the family**. + /// + /// This is the property that makes a K change cheap: K is an explicit wire + /// field, so moving from the K=8 control to K=9 changes one byte of the + /// artifact and nothing else about the format. #[test] - fn default_geometry_writes_nothing_and_k8v4l12_appends_exactly_four_bytes() -> Result<()> { + fn default_geometry_writes_nothing_and_the_family_appends_exactly_four_bytes() -> Result<()> { let device = Device::Cpu; let mut layer = geometry_fixture(&device)?; assert_eq!(layer.geometry, QtipGeometry::K4V2L16); let base = layer.serialize()?.into_owned(); - // Flip ONLY the discriminator; every tensor stays byte-identical. - layer.geometry = QtipGeometry::K8V4L12; - let tagged = layer.serialize()?.into_owned(); + for k in trellis_v4l12::K_SUPPORTED { + // Flip ONLY the discriminator; every tensor stays byte-identical. + layer.geometry = QtipGeometry::trellis_v4l12(k)?; + let tagged = layer.serialize()?.into_owned(); + + let mut want = base.clone(); + want.extend_from_slice(&[GEOMETRY_WIRE_TAG, k as u8, 12, 4]); + assert_eq!( + tagged, want, + "K={k}: the geometry section must be exactly [tag=3, K, L=12, V=4] appended to \ + an otherwise unchanged payload" + ); + // The K byte is the ONLY thing that differs between family members. + assert_eq!(tagged.len(), base.len() + 4); + } + Ok(()) + } - let mut want = base.clone(); - want.extend_from_slice(&[GEOMETRY_WIRE_TAG, 8, 12, 4]); + /// Two family members differ by exactly one byte on the wire. + /// + /// The concrete statement of "K is a parameter, not a variant": going from + /// the K=8 control to the K=9 quality winner is a single byte of artifact + /// change, given identical tensors. + #[test] + fn family_members_differ_by_one_wire_byte() -> Result<()> { + let device = Device::Cpu; + let mut layer = geometry_fixture(&device)?; + layer.geometry = QtipGeometry::trellis_v4l12(8)?; + let a = layer.serialize()?.into_owned(); + layer.geometry = QtipGeometry::trellis_v4l12(9)?; + let b = layer.serialize()?.into_owned(); + assert_eq!(a.len(), b.len()); + let diffs: Vec = (0..a.len()).filter(|&i| a[i] != b[i]).collect(); assert_eq!( - tagged, want, - "the geometry section must be exactly [tag=3, K=8, L=12, V=4] appended to an \ - otherwise unchanged payload" + diffs.len(), + 1, + "K=8 and K=9 payloads must differ in exactly one byte, differed in {diffs:?}" ); + assert_eq!((a[diffs[0]], b[diffs[0]]), (8, 9)); Ok(()) } @@ -7166,7 +7280,7 @@ mod tests { // directly by checking that the geometry bytes occupy the slot the // codebook otherwise would. layer.codebook = QtipCodebook::Gaussian; - layer.geometry = QtipGeometry::K8V4L12; + layer.geometry = k8(); let geo_only = layer.serialize()?.into_owned(); assert_eq!( geo_only[base.len()], @@ -7209,7 +7323,7 @@ mod tests { fn a_geometry_tag_that_contradicts_the_tensors_is_refused_at_load() -> Result<()> { let device = Device::Cpu; let mut layer = geometry_fixture(&device)?; - layer.geometry = QtipGeometry::K8V4L12; + layer.geometry = k8(); let data = layer.serialize()?.into_owned(); let err = QtipLayer::deserialize_concrete_unchecked( @@ -7236,11 +7350,11 @@ mod tests { #[test] fn a_right_sized_table_of_the_wrong_dtype_is_refused() -> Result<()> { let device = Device::Cpu; - let geo = QtipGeometry::K8V4L12; + let geo = k8(); let k_in = 64usize; let blocks = Tensor::zeros((2, geo.packed_len(k_in)), DType::U8, &device)?; let f32_table = Tensor::zeros( - (k8v4l12::LUT_STATES, k8v4l12::V as usize), + (trellis_v4l12::LUT_STATES, trellis_v4l12::V as usize), DType::F32, &device, )?; @@ -7292,7 +7406,7 @@ mod tests { QtipSearchStamp::Unstamped, QtipSearchDetail::Unknown, QtipCodebook::Gaussian, - QtipGeometry::K8V4L12, + k8(), ) .expect_err("a K8V4L12 tag over a K4V2L16 table must be refused at construction"); assert!(err.to_string().contains("k8v4l12"), "{err}"); @@ -7308,7 +7422,7 @@ mod tests { let mut b = geometry_fixture(&device)?; // Identical in every respect except the discriminator, so nothing but // the geometry check can be what fires. - b.geometry = QtipGeometry::K8V4L12; + b.geometry = k8(); let err = QtipLayer::stack_experts(vec![a, b]) .expect_err("mixed-geometry stacks must be refused"); assert!( @@ -7324,62 +7438,175 @@ mod tests { Ok(()) } - /// A genuine K=8/V=4/L=12 payload round-trips, tag and tensors both. + /// Every decode entry point must refuse a geometry it cannot decode. + /// + /// Stage 2 makes a K=8/V=4/L=12 artifact loadable. The decoders behind + /// these entry points all unpack nibbles and index a `[2^16, 2]` table, and + /// handed a K=8 row they would NOT fault — at 2 bits per weight the packed + /// byte count is identical and every index lands in bounds. They would + /// serve garbage. Without this test, "the format landed" would mean "the + /// engine now has a way to silently serve wrong weights". #[test] - fn k8v4l12_payload_round_trips_exactly() -> Result<()> { + fn every_decode_entry_point_refuses_an_unsupported_geometry() -> Result<()> { let device = Device::Cpu; - let (n, k_in) = (4usize, 128usize); - let geo = QtipGeometry::K8V4L12; - let packed_per_row = geo.packed_len(k_in); - assert_eq!(packed_per_row, k_in / 4, "2 bpw either way"); + let k_in = 64usize; - let blocks_data: Vec = (0..(n * packed_per_row)) - .map(|i| (i * 37 % 251) as u8) - .collect(); - let blocks = Tensor::from_vec(blocks_data, (n, packed_per_row), &device)?; - let scales = Tensor::from_vec( - (0..n).map(|i| 0.01 + i as f32 * 0.003).collect::>(), - (n,), - &device, - )?; + // 2-D entry points. + let mut two_d = geometry_fixture(&device)?; + // Sanity: all of these work at the geometry this build implements, so + // the refusals below are about the geometry and nothing else. + two_d.dequantize_weights()?; + let x = Tensor::zeros((1, k_in), DType::F32, &device)?; + two_d.forward(&x)?; + two_d.dequantize_w()?; + assert!(two_d.qtip_packed().is_some()); + + two_d.geometry = k8(); + for (what, res) in [ + ("dequantize_weights", two_d.dequantize_weights().err()), + ("forward", two_d.forward(&x).err()), + ("dequantize_w", two_d.dequantize_w().err()), + ] { + let err = res.unwrap_or_else(|| panic!("{what} must refuse a k8v4l12 layer")); + let msg = err.to_string(); + assert!( + msg.contains("k8v4l12") && msg.contains("Refusing"), + "{what}: refusal must name the geometry and say it is refusing: {msg}" + ); + // The refusal must come from THIS entry point, not from something + // it happens to call. `forward` reaching an inner guard is still a + // refusal, but it is not evidence that `forward` checks — and an + // entry point that relies on a callee's guard breaks the moment the + // callee gains a fast path that skips it. (Measured: mutation G1 + // removed `forward`'s own guard and every test stayed green.) + // + // `dequantize_w` is the one exception by construction: it is a + // one-line delegation to `dequantize_weights` and has no body to + // guard. + let expect_op = if what == "dequantize_w" { + "dequantize_weights" + } else { + what + }; + assert!( + msg.contains(&format!("QtipLayer::{expect_op}:")), + "{what}: refusal came from elsewhere — expected `QtipLayer::{expect_op}:` in: {msg}" + ); + } + assert!( + two_d.qtip_packed().is_none(), + "qtip_packed must hand out nothing — QtipPackedView carries no geometry, so a \ + consumer would read the bytes as K=4/V=2 nibbles" + ); + + // 3-D entry points. + let mut stack = + QtipLayer::stack_experts(vec![geometry_fixture(&device)?, geometry_fixture(&device)?])?; + stack.dequantize_expert(0)?; + let a = Tensor::zeros((1, 1, k_in), DType::F32, &device)?; + let idx = Tensor::zeros((1, 1), DType::U32, &device)?; + stack.gather_forward(&a, &idx)?; + + stack.geometry = k8(); + for (what, res) in [ + ("dequantize_expert", stack.dequantize_expert(0).err()), + ("gather_forward", stack.gather_forward(&a, &idx).err()), + ] { + let err = res.unwrap_or_else(|| panic!("{what} must refuse a k8v4l12 layer")); + let msg = err.to_string(); + assert!(msg.contains("k8v4l12"), "{what}: {msg}"); + assert!( + msg.contains(&format!("QtipLayer::{what}:")), + "{what}: refusal came from elsewhere: {msg}" + ); + } + Ok(()) + } + + /// A genuine V=4/L=12 payload round-trips at every K, tag and tensors both. + #[test] + fn trellis_v4l12_payloads_round_trip_exactly_at_every_k() -> Result<()> { + let device = Device::Cpu; + let (n, k_in) = (4usize, 128usize); let lut = Tensor::from_slice( - &k8v4l12::gaussian_lut_bf16(), - (k8v4l12::LUT_STATES, k8v4l12::V as usize), + &trellis_v4l12::gaussian_lut_bf16(), + (trellis_v4l12::LUT_STATES, trellis_v4l12::V as usize), &device, )?; - let layer = QtipLayer { - blocks, - row_scales: scales, - lut, - bias: None, - in_features: k_in, - num_experts: None, - rotation_signs: None, - rotation_block: 0, - search: QtipSearchStamp::Trellis, - search_detail: QtipSearchDetail::Known { - beam_width: None, - hessian: false, - }, - codebook: QtipCodebook::Gaussian, - geometry: geo, - }; + for k in trellis_v4l12::K_SUPPORTED { + let geo = QtipGeometry::trellis_v4l12(k)?; + let packed_per_row = geo.packed_len(k_in); + // Sanity: the row length really does track K. + assert_eq!(packed_per_row, (k_in / 4 * k as usize).div_ceil(8)); - let data = layer.serialize()?.into_owned(); - let (restored, _) = QtipLayer::deserialize_concrete_unchecked( - Cow::Owned(data), + let blocks_data: Vec = (0..(n * packed_per_row)) + .map(|i| (i * 37 % 251) as u8) + .collect(); + let blocks = Tensor::from_vec(blocks_data, (n, packed_per_row), &device)?; + let scales = Tensor::from_vec( + (0..n).map(|i| 0.01 + i as f32 * 0.003).collect::>(), + (n,), + &device, + )?; + + let layer = QtipLayer { + blocks, + row_scales: scales, + lut: lut.clone(), + bias: None, + in_features: k_in, + num_experts: None, + rotation_signs: None, + rotation_block: 0, + search: QtipSearchStamp::Trellis, + search_detail: QtipSearchDetail::Known { + beam_width: None, + hessian: false, + }, + codebook: QtipCodebook::Gaussian, + geometry: geo, + }; + + let data = layer.serialize()?.into_owned(); + let (restored, _) = QtipLayer::deserialize_concrete_unchecked( + Cow::Owned(data), + &device, + crate::QuantizeOntoGuard::new(), + )?; + + assert_eq!(restored.geometry, geo, "K={k}"); + assert_eq!(restored.geometry.k(), k); + assert_eq!(restored.in_features, k_in); + assert_eq!(restored.lut.dtype(), DType::BF16); + assert_eq!(restored.lut.elem_count(), trellis_v4l12::LUT_ENTRIES); + assert_tensor_bits_eq(&layer.blocks, &restored.blocks, "blocks")?; + assert_tensor_bits_eq(&layer.row_scales, &restored.row_scales, "row_scales")?; + assert_tensor_bits_eq(&layer.lut, &restored.lut, "lut")?; + } + Ok(()) + } + + /// A K the wire names but this build has no decoder for is refused, and the + /// refusal quotes the triple. + #[test] + fn an_in_family_but_unsupported_k_is_refused_with_a_diagnosis() -> Result<()> { + let device = Device::Cpu; + let layer = geometry_fixture(&device)?; + let base = layer.serialize()?.into_owned(); + // L=12/V=4 is this family, but K=11 has no decoder. + let mut bytes = base.clone(); + bytes.extend_from_slice(&[GEOMETRY_WIRE_TAG, 11, 12, 4]); + let err = QtipLayer::deserialize_concrete_unchecked( + Cow::Owned(bytes), &device, crate::QuantizeOntoGuard::new(), - )?; - - assert_eq!(restored.geometry, QtipGeometry::K8V4L12); - assert_eq!(restored.in_features, k_in); - assert_eq!(restored.lut.dtype(), DType::BF16); - assert_eq!(restored.lut.elem_count(), k8v4l12::LUT_ENTRIES); - assert_tensor_bits_eq(&layer.blocks, &restored.blocks, "blocks")?; - assert_tensor_bits_eq(&layer.row_scales, &restored.row_scales, "row_scales")?; - assert_tensor_bits_eq(&layer.lut, &restored.lut, "lut")?; + ) + .expect_err("K=11 must be refused"); + assert!( + err.to_string().contains("K=11") || err.to_string().contains("not implemented"), + "refusal must name the K it could not decode: {err}" + ); Ok(()) } @@ -7390,7 +7617,7 @@ mod tests { fn the_computed_codebook_is_refused_at_a_v4_geometry() -> Result<()> { let device = Device::Cpu; let mut layer = geometry_fixture(&device)?; - layer.geometry = QtipGeometry::K8V4L12; + layer.geometry = k8(); layer.codebook = QtipCodebook::COMPUTED; let err = layer .serialize() @@ -7453,52 +7680,119 @@ mod tests { Ok(()) } - /// Both geometries are 2 bits per weight. This is the fact that makes the - /// discriminator necessary — a mislabelled artifact has a correctly-sized - /// `blocks` tensor — so it gets an assertion rather than a comment. + /// The shipped rung and the K=8 control are the SAME bit rate and the SAME + /// packed size. That is the fact that makes the discriminator necessary — + /// a mislabelled artifact has a correctly-sized `blocks` tensor — so it + /// gets an assertion rather than a comment. #[test] - fn both_geometries_are_two_bits_per_weight_and_the_same_packed_size() { + fn k4v2l16_and_k8v4l12_are_indistinguishable_by_packed_size() { + let k8 = k8(); for k_in in [64usize, 512, 4096, 7168] { assert_eq!( QtipGeometry::K4V2L16.packed_len(k_in), - QtipGeometry::K8V4L12.packed_len(k_in), + k8.packed_len(k_in), "k_in={k_in}: identical packed size is why the tag is load-bearing" ); - assert_eq!(QtipGeometry::K4V2L16.n_bits(), 2); - assert_eq!(QtipGeometry::K8V4L12.n_bits(), 2); } + assert_eq!(QtipGeometry::K4V2L16.bpw_x100(), 200); + assert_eq!(k8.bpw_x100(), 200); // ...and the tables are what actually differ. assert_eq!(QtipGeometry::K4V2L16.lut_values(), 65_536 * 2); - assert_eq!(QtipGeometry::K8V4L12.lut_values(), 4_096 * 4); + assert_eq!(k8.lut_values(), 4_096 * 4); assert_eq!(QtipGeometry::K4V2L16.lut_dtype(), DType::F32); - assert_eq!(QtipGeometry::K8V4L12.lut_dtype(), DType::BF16); - assert_eq!(QtipGeometry::K4V2L16.syms_per_byte(), 2); - assert_eq!(QtipGeometry::K8V4L12.syms_per_byte(), 1); + assert_eq!(k8.lut_dtype(), DType::BF16); } - /// The geometry accessors must agree with the rung module they describe. + /// **Not every geometry in the enum is 2 bits per weight any more.** + /// + /// K=9/V=4 is 2.25 bpw, so its rows are 12.5% larger than K=8/V=4's and it + /// is *not* size-confusable with the shipped rung. Worth pinning: code that + /// assumed "all QTIP geometries are 2 bpw" — a true statement when this + /// family was K=8-only — is now wrong, and `packed_len` is the general + /// `ceil(n·K/8)` rather than anything that divides by `8/K`. #[test] - fn geometry_accessors_match_the_rung_modules() { + fn the_family_spans_more_than_one_bit_rate() { + let k8 = k8(); + let k9 = QtipGeometry::trellis_v4l12(9).unwrap(); + let k10 = QtipGeometry::trellis_v4l12(10).unwrap(); + assert_eq!(k8.bpw_x100(), 200); + assert_eq!(k9.bpw_x100(), 225); + assert_eq!(k10.bpw_x100(), 250); + for k_in in [512usize, 4096, 7168] { + assert!(k9.packed_len(k_in) > k8.packed_len(k_in)); + // Exactly the bit-rate ratio, to the byte. + assert_eq!(k9.packed_len(k_in) * 8, k8.packed_len(k_in) * 9); + // The whole family shares one table regardless of K. + assert_eq!(k9.lut_values(), k8.lut_values()); + assert_eq!(k9.lut_dtype(), k8.lut_dtype()); + } + } + + /// The packed length must use the CEIL, and that is only visible at an + /// `in_features` which is not a multiple of 32. + /// + /// Every plausible layer width is a multiple of 32, which makes + /// `num_symbols · K` a multiple of 8 and floor equal to ceil at K=9. A + /// floored formula therefore passes every realistic fixture (measured: + /// mutation W3). These widths are deliberately unrealistic. + #[test] + fn packed_len_uses_the_ceiling_where_it_is_observable() { + let k9 = QtipGeometry::trellis_v4l12(9).unwrap(); + // in_features=36 -> 9 symbols -> 81 bits -> 11 bytes, not 10. + assert_eq!(k9.num_symbols(36), 9); + assert_eq!(k9.packed_len(36), 11); + // in_features=4 -> 1 symbol -> 9 bits -> 2 bytes, not 1. + assert_eq!(k9.packed_len(4), 2); + let k10 = QtipGeometry::trellis_v4l12(10).unwrap(); + // 3 symbols -> 30 bits -> 4 bytes, not 3. + assert_eq!(k10.packed_len(12), 4); + // ...and it agrees with the rung that owns the formula. + for k in trellis_v4l12::K_SUPPORTED { + let g = QtipGeometry::trellis_v4l12(k).unwrap(); + let r = trellis_v4l12::Rung::new(k).unwrap(); + for k_in in [4usize, 12, 36, 100, 260, 4096] { + assert_eq!(g.packed_len(k_in), r.packed_len(k_in), "K={k} k_in={k_in}"); + } + } + } + + /// A K with no decoder is refused at construction, not stored and used. + #[test] + fn an_unsupported_k_cannot_become_a_geometry() { + for k in [0u32, 4, 7, 11, 16] { + assert!( + QtipGeometry::trellis_v4l12(k).is_err(), + "K={k} must not be constructible" + ); + } + for k in trellis_v4l12::K_SUPPORTED { + assert!(QtipGeometry::trellis_v4l12(k).is_ok(), "K={k} must be"); + } + } + + /// The geometry accessors must agree with the rung they describe. + #[test] + fn geometry_accessors_match_the_rung_module() { assert_eq!(QtipGeometry::K4V2L16.k(), K); assert_eq!(QtipGeometry::K4V2L16.l(), L); assert_eq!(QtipGeometry::K4V2L16.v(), V); - assert_eq!(QtipGeometry::K8V4L12.k(), k8v4l12::K); - assert_eq!(QtipGeometry::K8V4L12.l(), k8v4l12::L); - assert_eq!(QtipGeometry::K8V4L12.v(), k8v4l12::V); - assert_eq!( - QtipGeometry::K8V4L12.lut_values(), - k8v4l12::LUT_ENTRIES, - "the discriminator and the rung must agree on the table size" - ); - for k_in in [64usize, 512, 4096] { + for k in trellis_v4l12::K_SUPPORTED { + let g = QtipGeometry::trellis_v4l12(k).unwrap(); + let r = trellis_v4l12::Rung::new(k).unwrap(); + assert_eq!(g.k(), r.k()); + assert_eq!(g.l(), trellis_v4l12::L); + assert_eq!(g.v(), trellis_v4l12::V); + assert_eq!(g.bpw_x100(), r.bpw_x100()); assert_eq!( - QtipGeometry::K8V4L12.packed_len(k_in), - k8v4l12::packed_len(k_in) - ); - assert_eq!( - QtipGeometry::K8V4L12.num_symbols(k_in), - k8v4l12::num_symbols(k_in) + g.lut_values(), + trellis_v4l12::LUT_ENTRIES, + "the discriminator and the rung must agree on the table size" ); + for k_in in [64usize, 512, 4096] { + assert_eq!(g.packed_len(k_in), r.packed_len(k_in), "K={k} k_in={k_in}"); + assert_eq!(g.num_symbols(k_in), r.num_symbols(k_in)); + } + assert_eq!(g.tag(), format!("k{k}v4l12")); } } From 41e8e1f246928bb84d94bcb40da0e5c7e3877b20 Mon Sep 17 00:00:00 2001 From: Nirupam Bhowmick <48842933+heydryft@users.noreply.github.com> Date: Wed, 19 Aug 2026 15:43:10 +0100 Subject: [PATCH 3/4] fix(qtip): UQFF validates the PADDED row stride, not the data length Follows the padding decision one branch down. `QtipGeometry::packed_len` now means the allocated stride (what a tensor is shaped at and what this format validates) and `data_bytes` is the bit-rate-governed part, both delegating to `trellis_v4l12::Rung` so there is still one implementation of each. Three tests changed premise rather than value and were rewritten, not patched: the bit-rate ratio claim moved to `data_bytes` where it is exact, and the stride got its own bounds (never smaller than the data, never more than 4 bytes larger). A K=9 row of `in_features=36` is 11 data bytes in a 12-byte stride. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SpVNMpb13HkUXqSqbN1o9H --- mistralrs-quant/src/qtip/mod.rs | 60 +++++++++++++++++++++++++-------- 1 file changed, 46 insertions(+), 14 deletions(-) diff --git a/mistralrs-quant/src/qtip/mod.rs b/mistralrs-quant/src/qtip/mod.rs index dec9a4848..441eb5cf9 100644 --- a/mistralrs-quant/src/qtip/mod.rs +++ b/mistralrs-quant/src/qtip/mod.rs @@ -825,8 +825,21 @@ impl QtipGeometry { in_features / self.v() as usize } - /// Packed bytes in a row of `in_features` weights: - /// `ceil(num_symbols · K / 8)`. + /// Bytes of a row that actually hold symbols, governed by the bit rate. + /// + /// **Not the row stride.** [`QtipGeometry::packed_len`] is what a tensor is + /// allocated at and what this format validates; it additionally carries + /// tail padding and a round-up to 4 for the V=4/L=12 family. Conflating + /// them is how a bits-per-weight claim ends up four bytes wrong. + pub fn data_bytes(self, in_features: usize) -> usize { + match self { + QtipGeometry::K4V2L16 => self.num_symbols(in_features) / 2, + QtipGeometry::TrellisV4L12 { rung } => rung.data_bytes(rung.num_symbols(in_features)), + } + } + + /// The allocated length of a row of `in_features` weights — the tensor's + /// last dimension, and what this format validates. /// /// **Delegated, not restated.** `num_symbols / (8 / K)` is wrong at K=9 — /// there is no whole number of symbols per byte — and divides by zero, so @@ -7537,8 +7550,11 @@ mod tests { for k in trellis_v4l12::K_SUPPORTED { let geo = QtipGeometry::trellis_v4l12(k)?; let packed_per_row = geo.packed_len(k_in); - // Sanity: the row length really does track K. - assert_eq!(packed_per_row, (k_in / 4 * k as usize).div_ceil(8)); + // Sanity: the DATA length tracks K, and the allocated stride is + // that plus the padding the unclamped extraction relies on. + assert_eq!(geo.data_bytes(k_in), (k_in / 4 * k as usize).div_ceil(8)); + assert!(packed_per_row >= geo.data_bytes(k_in)); + assert!(packed_per_row - geo.data_bytes(k_in) <= 4); let blocks_data: Vec = (0..(n * packed_per_row)) .map(|i| (i * 37 % 251) as u8) @@ -7720,8 +7736,10 @@ mod tests { assert_eq!(k10.bpw_x100(), 250); for k_in in [512usize, 4096, 7168] { assert!(k9.packed_len(k_in) > k8.packed_len(k_in)); - // Exactly the bit-rate ratio, to the byte. - assert_eq!(k9.packed_len(k_in) * 8, k8.packed_len(k_in) * 9); + // Exactly the bit-rate ratio, to the byte — on DATA bytes. The + // stride is only approximately in that ratio, because K=9 pads for + // its multi-byte extraction and K=8 does not. + assert_eq!(k9.data_bytes(k_in) * 8, k8.data_bytes(k_in) * 9); // The whole family shares one table regardless of K. assert_eq!(k9.lut_values(), k8.lut_values()); assert_eq!(k9.lut_dtype(), k8.lut_dtype()); @@ -7738,20 +7756,34 @@ mod tests { #[test] fn packed_len_uses_the_ceiling_where_it_is_observable() { let k9 = QtipGeometry::trellis_v4l12(9).unwrap(); - // in_features=36 -> 9 symbols -> 81 bits -> 11 bytes, not 10. + // in_features=36 -> 9 symbols -> 81 bits -> 11 DATA bytes, not 10. assert_eq!(k9.num_symbols(36), 9); - assert_eq!(k9.packed_len(36), 11); - // in_features=4 -> 1 symbol -> 9 bits -> 2 bytes, not 1. - assert_eq!(k9.packed_len(4), 2); + assert_eq!(k9.data_bytes(36), 11); + // in_features=4 -> 1 symbol -> 9 bits -> 2 data bytes, not 1. + assert_eq!(k9.data_bytes(4), 2); let k10 = QtipGeometry::trellis_v4l12(10).unwrap(); - // 3 symbols -> 30 bits -> 4 bytes, not 3. - assert_eq!(k10.packed_len(12), 4); - // ...and it agrees with the rung that owns the formula. + // 3 symbols -> 30 bits -> 4 data bytes, not 3. + assert_eq!(k10.data_bytes(12), 4); + // The stride is the data plus tail padding, rounded up to 4. + assert_eq!(k9.packed_len(36), 12); + assert_eq!(k9.packed_len(4), 4); + assert_eq!(k10.packed_len(12), 8); + // ...and both agree with the rung that owns the formulas. for k in trellis_v4l12::K_SUPPORTED { let g = QtipGeometry::trellis_v4l12(k).unwrap(); let r = trellis_v4l12::Rung::new(k).unwrap(); for k_in in [4usize, 12, 36, 100, 260, 4096] { - assert_eq!(g.packed_len(k_in), r.packed_len(k_in), "K={k} k_in={k_in}"); + let n = r.num_symbols(k_in); + assert_eq!( + g.packed_len(k_in), + r.row_stride(n), + "K={k} k_in={k_in} stride" + ); + assert_eq!( + g.data_bytes(k_in), + r.data_bytes(n), + "K={k} k_in={k_in} data" + ); } } } From 8e81323ff4ae76fa65e6ae9601492c1db40eebe0 Mon Sep 17 00:00:00 2001 From: Nirupam Bhowmick <48842933+heydryft@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:59:20 +0100 Subject: [PATCH 4/4] fix(qtip): the CUDA bake fast path lost the geometry wire tag MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `quantize_with_options_cuda` builds a QtipLayer directly and was the one initializer the geometry discriminator missed. It is behind `#[cfg(feature = "cuda")]`, so the default Check jobs compiled fine and only the CUDA lane went red (E0063 at qtip/mod.rs:2318). Set the same K4V2L16 tag its CPU sibling sets — the CUDA bake kernels run the identical K=4/V=2/L=16 trellis, so a GPU-baked artifact must carry the identical tag or it would deserialize as a different geometry. Co-Authored-By: Claude Opus 5 (1M context) --- mistralrs-quant/src/qtip/mod.rs | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mistralrs-quant/src/qtip/mod.rs b/mistralrs-quant/src/qtip/mod.rs index 441eb5cf9..fdbd015a2 100644 --- a/mistralrs-quant/src/qtip/mod.rs +++ b/mistralrs-quant/src/qtip/mod.rs @@ -2370,6 +2370,11 @@ impl QtipLayer { // objective bit is false by construction, not by omission. search_detail: QtipSearchDetail::for_bake(mode, search, false), codebook, + // Same geometry as the CPU sibling above: the CUDA bake kernels are + // the K=4/V=2/L=16 trellis too, so a GPU-baked artifact must carry + // the identical wire tag. Omitting it here compiled fine without + // the `cuda` feature and only broke the CUDA lane. + geometry: QtipGeometry::K4V2L16, })) }