Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Fixed
- **Vector: SQ8/TQ code-size mis-dispatch CPU error-storm.** SQ8's `bits()` returns 8,
which falls outside TurboQuant's supported `1..=4` range; the free
`code_bytes_per_vector(padded, bits)` helper hit its `_ =>` arm and returned `0` while
logging a `tracing::error!` on **every** call. On the hot memory-accounting path
(`store.rs` via `bytes_per_code_per_vector()`) this produced a CPU-pegging error storm
and wrong resident-byte accounting for SQ8 indexes. Dispatch now computes the true SQ8
layout (`dim` unpadded u8 codes + 8-byte affine `(min,scale)` trailer) directly, and the
free-fn logs the unsupported-bit-width error at most once via an `AtomicBool` latch.

## [0.7.0] — 2026-07-15

**Replication GA for multi-shard masters.** Moon now supports real Redis-compatible
Expand Down
13 changes: 12 additions & 1 deletion src/vector/turbo_quant/codebook.rs
Original file line number Diff line number Diff line change
Expand Up @@ -191,7 +191,18 @@ pub fn code_bytes_per_vector(padded_dim: u32, bits: u8) -> usize {
3 => (pd * 3 + 7) / 8,
4 => pd / 2,
_ => {
tracing::error!("unsupported bit width {bits} for code_bytes_per_vector");
// Unsupported width (e.g. SQ8's bits()==8) must never reach here: SQ8
// is sized via CollectionMetadata::code_bytes_per_vector's SQ8 branch.
// If a stray caller ever does, log ONCE — a per-call error here on a
// hot accounting path (INFO memory / eviction) pegs a core with an
// error storm (observed 2026-07 on a lunaris-backed SQ8 store).
static WARNED: std::sync::atomic::AtomicBool =
std::sync::atomic::AtomicBool::new(false);
if !WARNED.swap(true, std::sync::atomic::Ordering::Relaxed) {
tracing::error!(
"unsupported bit width {bits} for code_bytes_per_vector (logged once)"
);
}
0
}
}
Expand Down
69 changes: 62 additions & 7 deletions src/vector/turbo_quant/collection.rs
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ use super::codebook::{
CODEBOOK_VERSION, code_bytes_per_vector, scaled_boundaries_n, scaled_centroids_n,
};
use super::encoder::padded_dimension;
use super::sq8::SQ8_PARAMS_BYTES;
use super::sub_centroid::SubCentroidTable;
use crate::vector::aligned_buffer::AlignedBuffer;
use crate::vector::types::DistanceMetric;
Expand Down Expand Up @@ -301,11 +302,14 @@ impl CollectionMetadata {
/// packed size is `padded_dim / 4` instead of the scalar TQ4's `padded_dim / 2`.
#[inline]
pub fn code_bytes_per_vector(&self) -> usize {
if self.quantization == QuantizationConfig::TurboQuant4A2 {
match self.quantization {
// SQ8 stores one u8 code per TRUE (unpadded) dimension — it does NOT
// FWHT-pad and its bits() is 8, outside the TurboQuant 1..=4 range the
// free `code_bytes_per_vector` supports. Mirror `MutableSegment::new`.
QuantizationConfig::Sq8 => self.dimension as usize,
// A2: padded_dim/2 pairs, each pair → 4-bit index, nibble-packed → padded_dim/4 bytes
self.padded_dimension as usize / 4
} else {
code_bytes_per_vector(self.padded_dimension, self.quantization.bits())
QuantizationConfig::TurboQuant4A2 => self.padded_dimension as usize / 4,
_ => code_bytes_per_vector(self.padded_dimension, self.quantization.bits()),
}
}

Expand Down Expand Up @@ -361,11 +365,19 @@ impl CollectionMetadata {
}
}

/// Bytes per TQ code including the 4-byte norm suffix.
/// = code_bytes_per_vector() + 4.
/// Full per-vector code slot size in bytes, including the trailer.
///
/// TurboQuant configs append a 4-byte f32 norm (`code_bytes_per_vector() + 4`).
/// SQ8 instead appends an 8-byte `(min, scale)` trailer over `dim` u8 codes
/// (`dim + SQ8_PARAMS_BYTES`), matching `MutableSegment::new`'s SQ8 branch and
/// `sq8_code_bytes`. Getting this wrong for SQ8 under-sizes memory accounting
/// (`store.rs`) and compaction slots (`compaction.rs`).
#[inline]
pub fn bytes_per_code_per_vector(&self) -> usize {
self.code_bytes_per_vector() + 4
match self.quantization {
QuantizationConfig::Sq8 => self.dimension as usize + SQ8_PARAMS_BYTES,
_ => self.code_bytes_per_vector() + 4,
}
}

/// Try to return the codebook as a `&[f32; 16]`, returning None for non-4-bit configs.
Expand Down Expand Up @@ -671,6 +683,49 @@ mod tests {
assert_eq!(meta4.code_bytes_per_vector(), 512);
}

/// SQ8 uses `bits() == 8`, which is OUTSIDE the TurboQuant 1..=4 range that
/// the free `code_bytes_per_vector(padded, bits)` supports. Before the fix,
/// `CollectionMetadata::code_bytes_per_vector()` fell through to that free fn
/// with bits=8, which returned 0 and logged an error on EVERY call — a hot
/// accounting path (INFO memory / eviction checks) turned that into a
/// CPU-pegging error storm (observed on a lunaris-backed store). SQ8's real
/// code layout is `dim` u8 codes (true dimension, NOT padded) + an 8-byte
/// (min,scale) trailer — mirroring `MutableSegment::new`'s SQ8 branch.
#[test]
fn test_code_bytes_per_vector_sq8() {
use crate::vector::turbo_quant::sq8::SQ8_PARAMS_BYTES;
let meta = CollectionMetadata::new(
9,
768, // pads to 1024, but SQ8 sizes by TRUE dim
DistanceMetric::L2,
QuantizationConfig::Sq8,
42,
);
// SQ8: one u8 code per (unpadded) dimension.
assert_eq!(
meta.code_bytes_per_vector(),
768,
"SQ8 code bytes must equal true dimension, not 0 (unsupported-bits fallthrough)"
);
// Full slot = codes + 8-byte (min,scale) trailer (NOT the TQ 4-byte norm).
assert_eq!(
meta.bytes_per_code_per_vector(),
768 + SQ8_PARAMS_BYTES,
"SQ8 slot must be dim + 8-byte params trailer, not dim/… + 4"
);

// A non-square dimension exercises the true-vs-padded distinction.
let meta2 = CollectionMetadata::new(
10,
384, // pads to 512
DistanceMetric::InnerProduct,
QuantizationConfig::Sq8,
42,
);
assert_eq!(meta2.code_bytes_per_vector(), 384);
assert_eq!(meta2.bytes_per_code_per_vector(), 384 + SQ8_PARAMS_BYTES);
}

#[test]
fn test_checksum_changes_when_quantization_changes() {
let meta1 = CollectionMetadata::new(
Expand Down
Loading