From 1e7a7e7bc395595a86b359b46b420ddfe1334e42 Mon Sep 17 00:00:00 2001 From: ajianaz Date: Fri, 28 Aug 2026 15:02:22 +0700 Subject: [PATCH] =?UTF-8?q?fix(core):=20persist=20the=20keyed=20map=20?= =?UTF-8?q?=E2=80=94=20file=20format=20v1.3=20(#32)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Keyed state used to vanish on save/reload: from_bytes rebuilt codes and scales but never the key->slot map, so search_keyed/remove_keyed were dead on every reloaded index (user-filed repro in #32). Format v1.3 appends a keyed-slot table after the codes block: [entries u32][slot u32 + key u64 each]. Slot ids are dense positions among the serialized (alive) slots — matching the reader's slot space — not in-memory indices, which can exceed the serialized count when tombstones are dropped (caught by the multi-key round-trip test). Entries are validated (slot bounds, duplicate keys) and restore_keys rebuilds both the single- and multi-key maps, so reloaded indexes are fully keyed-capable (search_keyed, remove_keyed(_at), relabel, compact all work). Readers accept v1/v1.1/v1.2/v1.3; writers emit v1.3. docs (SQLITE.md, README, BENCHMARK.md) updated; CHANGELOG Unreleased entry added. Tests (TDD, 2 new): the issue's exact repro (keys survive reload, reloaded index remains keyed-capable) and a multi-key round trip with remove_keyed_at + compact + re-serialization stability. Signed-off-by: ajianaz --- CHANGELOG.md | 5 ++ README.md | 4 +- crates/vecq-core/src/format.rs | 74 ++++++++++++++++++++++++------ crates/vecq-core/src/store.rs | 84 ++++++++++++++++++++++++++++++++-- docs/BENCHMARK.md | 8 ++-- docs/SQLITE.md | 21 +++------ 6 files changed, 157 insertions(+), 39 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1db120e..a736859 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,11 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). +## [Unreleased] + +### Fixed +- Keyed API now survives save/reload: file format v1.3 stores a keyed-slot table, and `from_bytes` restores the full key→slot map (#32) + ## [0.2.0] — 2026-08-28 ### Added diff --git a/README.md b/README.md index bcee83e..b390e08 100644 --- a/README.md +++ b/README.md @@ -51,7 +51,7 @@ index.remove_keyed(2002); // tombstone; `compact()` reclaims th let keyed_hits: Vec<(u64, f32)> = index.search_keyed(&query, 10); // Single-file persistence, deterministic across platforms. -// Tombstones are dropped on disk; keys live in your own metadata table. +// Tombstones are dropped on disk; keys persist (format v1.3 keyed table). let bytes = index.to_bytes(); let back = VecqIndex::from_bytes(&bytes).unwrap(); assert_eq!(index.search(&query, 10), back.search(&query, 10)); @@ -62,7 +62,7 @@ assert_eq!(index.search(&query, 10), back.search(&query, 10)); - **Deterministic**: same file + same query → identical result bits on any platform. The seed lives in the header, the scoring path has a fixed association order and no FMA contraction, and a unit test enforces SIMD/scalar bit-identity. - **Zero dependencies** in `vecq-core`'s quantization path. - **Recall@10 ≥ 0.95** on real embedding data at 6x compression (see `docs/BENCHMARK.md`). -- **Forward-compatible format**: readers accept v1 (f32 scales), v1.1 (f16 scales) and v1.2 (Matryoshka working_dim) files. +- **Forward-compatible format**: readers accept v1 (f32 scales), v1.1 (f16 scales), v1.2 (Matryoshka working_dim) and v1.3 (keyed-slot table) files. ## Matryoshka models diff --git a/crates/vecq-core/src/format.rs b/crates/vecq-core/src/format.rs index dc1ae2c..59ce3cb 100644 --- a/crates/vecq-core/src/format.rs +++ b/crates/vecq-core/src/format.rs @@ -14,10 +14,16 @@ //! `working_dim` (Matryoshka truncation; 0 means working_dim == dim). Codes //! and scales are laid out exactly like 1.1 — only the semantic of the //! reserved field changes. +//! Version 1.3 (stored as 259): after the codes block, a keyed-slot table +//! restores the keyed API across reloads: +//! ```text +//! [keyed_entries u32][entries: slot u32 + key u64 each, slots in order] +//! ``` //! -//! Readers accept v1, v1.1 and v1.2; writers emit 1.2. The seed is stored in -//! the header so the random sign diagonal can be regenerated identically on -//! any platform: identical file -> identical query results, bit for bit. +//! Readers accept v1, v1.1, v1.2 and v1.3; writers emit 1.3. The seed is +//! stored in the header so the random sign diagonal can be regenerated +//! identically on any platform: identical file -> identical query results, +//! bit for bit. use crate::store::VecqIndex; @@ -25,6 +31,7 @@ const MAGIC: u32 = u32::from_le_bytes(*b"VECQ"); const V1: u16 = 1; const V1_1: u16 = 257; pub const V1_2: u16 = 258; +const V1_3: u16 = 259; #[derive(Debug)] pub enum Error { @@ -33,6 +40,7 @@ pub enum Error { Truncated, DimMismatch { expected: usize, got: usize }, InvalidWorkingDim { dim: usize, working_dim: usize }, + InvalidKeyTable, } impl std::fmt::Display for Error { @@ -47,6 +55,7 @@ impl std::fmt::Display for Error { Error::InvalidWorkingDim { dim, working_dim } => { write!(f, "invalid working_dim {working_dim} for dim {dim}") } + Error::InvalidKeyTable => write!(f, "invalid keyed-slot table"), } } } @@ -124,14 +133,13 @@ fn f16_bits_to_f32(h: u16) -> f32 { } impl VecqIndex { - /// Serialize the index to bytes (format version 1.2, f16 scales). + /// Serialize the index to bytes (format version 1.3, f16 scales). /// /// Tombstoned slots are skipped: the output always holds the live vectors /// in slot order, so a round-trip through bytes has the same effect as /// [`VecqIndex::compact`] on disk without disturbing in-memory slot - /// indices. Keys are not part of the file format; persist a key→slot - /// table alongside (e.g. in SQLite) if you need keyed access across a - /// reload. + /// indices. Keys of live keyed slots are stored in the v1.3 keyed-slot + /// table and are fully restored by [`VecqIndex::from_bytes`]. pub fn to_bytes(&self) -> Vec { let bpv = self.padded_dim() / 2; let reserved: u16 = if self.working_dim() == self.dim() { @@ -141,7 +149,7 @@ impl VecqIndex { }; let mut out = Vec::with_capacity(24 + self.live_slots() * bpv + self.live_slots() * 2); out.extend_from_slice(&MAGIC.to_le_bytes()); - out.extend_from_slice(&V1_2.to_le_bytes()); + out.extend_from_slice(&V1_3.to_le_bytes()); out.extend_from_slice(&reserved.to_le_bytes()); out.extend_from_slice(&(self.dim() as u32).to_le_bytes()); out.extend_from_slice(&self.seed().to_le_bytes()); @@ -158,11 +166,25 @@ impl VecqIndex { } out.extend_from_slice(self.slot_codes(slot, bpv)); } + // Keyed-slot table (v1.3): restores the keyed API across reloads. + // Slot ids are DENSE positions among the serialized (alive) slots — + // the reader's slot space — not the in-memory slot indices, which + // may exceed the serialized count when tombstones are dropped. + let keyed: Vec<(u32, u64)> = (0..self.slots()) + .filter(|&s| self.slot_alive(s)) + .enumerate() + .filter_map(|(dense, s)| self.key_of(s).map(|k| (dense as u32, k))) + .collect(); + out.extend_from_slice(&(keyed.len() as u32).to_le_bytes()); + for (slot, key) in keyed { + out.extend_from_slice(&slot.to_le_bytes()); + out.extend_from_slice(&key.to_le_bytes()); + } out } - /// Parse an index from bytes produced by [`to_bytes`] (a v1.2 file) or a - /// legacy v1 / v1.1 file. + /// Parse an index from bytes produced by [`to_bytes`] (a v1.3 file) or a + /// legacy v1 / v1.1 / v1.2 file (which carry no key table). pub fn from_bytes(bytes: &[u8]) -> Result { let rd_u32 = |b: &[u8]| u32::from_le_bytes([b[0], b[1], b[2], b[3]]); let rd_u16 = |b: &[u8]| u16::from_le_bytes([b[0], b[1]]); @@ -173,16 +195,16 @@ impl VecqIndex { return Err(Error::NotAStableFile); } let version = rd_u16(&bytes[4..6]); - if version != V1 && version != V1_1 && version != V1_2 { + if version != V1 && version != V1_1 && version != V1_2 && version != V1_3 { return Err(Error::UnsupportedVersion(version)); } let dim = rd_u32(&bytes[8..12]) as usize; let seed = u64::from_le_bytes(bytes[12..20].try_into().unwrap()); let count = rd_u32(&bytes[20..24]) as usize; - // v1.2 carries working_dim in the reserved field (0 = full dim). + // v1.2 / v1.3 carry working_dim in the reserved field (0 = full dim). let working_dim = match version { - V1_2 => match rd_u16(&bytes[6..8]) as usize { + V1_2 | V1_3 => match rd_u16(&bytes[6..8]) as usize { 0 => dim, w if w <= dim => w, w => { @@ -214,11 +236,35 @@ impl VecqIndex { scales.push(s); off += scale_bytes; } + let code_end = off + count * codes_bytes; let mut index = VecqIndex::with_working_dim(dim, working_dim, seed); - index.codes = bytes[off..off + count * codes_bytes].to_vec(); + index.codes = bytes[off..code_end].to_vec(); index.scales = scales; index.n = count; index.init_dense(count); + if version == V1_3 { + // Keyed-slot table: [entries u32][slot u32 + key u64 each]. + if bytes.len() < code_end + 4 { + return Err(Error::Truncated); + } + let entries = rd_u32(&bytes[code_end..code_end + 4]) as usize; + let table_end = code_end + 4 + entries * 12; + if bytes.len() < table_end { + return Err(Error::Truncated); + } + let mut key_table: Vec> = vec![None; count]; + let mut e = code_end + 4; + for _ in 0..entries { + let slot = rd_u32(&bytes[e..e + 4]) as usize; + let key = u64::from_le_bytes(bytes[e + 4..e + 12].try_into().unwrap()); + e += 12; + if slot >= count || key_table[slot].is_some() { + return Err(Error::InvalidKeyTable); + } + key_table[slot] = Some(key); + } + index.restore_keys(key_table); + } Ok(index) } } diff --git a/crates/vecq-core/src/store.rs b/crates/vecq-core/src/store.rs index c21a6f5..ffbf6ab 100644 --- a/crates/vecq-core/src/store.rs +++ b/crates/vecq-core/src/store.rs @@ -160,6 +160,28 @@ impl VecqIndex { self.live = count; } + /// Restore a dense index's slot→key table (file format v1.3) and rebuild + /// the key maps from it. + pub(crate) fn restore_keys(&mut self, table: Vec>) { + let count = table.len(); + self.keys = table; + self.alive = vec![true; count]; + self.live = count; + self.key_to_slot.clear(); + self.key_to_slots.clear(); + for (slot, key) in self.keys.iter().enumerate() { + if let Some(key) = *key { + if let Some(slots) = self.key_to_slots.get_mut(&key) { + slots.push(slot); + } else if let Some(first) = self.key_to_slot.remove(&key) { + self.key_to_slots.insert(key, vec![first, slot]); + } else { + self.key_to_slot.insert(key, slot); + } + } + } + } + /// Quantize and add one vector (any norm; normalized internally). /// /// Returns the slot index holding the vector (stable until compaction). @@ -1329,6 +1351,61 @@ mod tests { assert!(idx.search_keyed(&rand_unit(dim, 4), 3).is_empty()); } + #[test] + fn keyed_keys_survive_file_round_trip() { + // Regression for issue #32: the keyed map must survive a save/reload. + let dim = 64; + let mut idx = VecqIndex::new(dim, 42); + let v = rand_unit(dim, 11); + idx.add_keyed(10, &v); + idx.add_keyed(20, &rand_unit(dim, 22)); + let bytes = idx.to_bytes(); + let mut back = VecqIndex::from_bytes(&bytes).expect("parse"); + assert_eq!(back.len(), 2); + assert!(back.contains_key(10)); + assert!(back.contains_key(20)); + assert_eq!(back.key_of(0), Some(10)); + assert_eq!(back.key_of(1), Some(20)); + let hits = back.search_keyed(&v, 5); + assert!(!hits.is_empty(), "keys must survive reload (issue #32)"); + assert_eq!(hits[0].0, 10); + // The reloaded index is fully keyed-capable. + assert!(back.remove_keyed(20)); + assert_eq!(back.len(), 1); + let slot = back.add_keyed_multi(10, &rand_unit(dim, 33)); + assert_eq!(back.key_of(slot), Some(10)); + assert!(back.relabel(10, 30)); + assert_eq!(back.key_of(slot), Some(30)); + } + + #[test] + fn multi_key_round_trip_and_compact() { + let dim = 64; + let mut idx = VecqIndex::new(dim, 7); + idx.add_keyed(1, &rand_unit(dim, 101)); + idx.add_keyed_multi(1, &rand_unit(dim, 202)); + idx.add_keyed_multi(1, &rand_unit(dim, 303)); + idx.add_keyed(2, &rand_unit(dim, 404)); + let bytes = idx.to_bytes(); + let mut back = VecqIndex::from_bytes(&bytes).expect("parse"); + assert_eq!(back.len(), 4); + // Multi structure restored: remove one slot, key survives with two. + assert!(back.remove_keyed_at(1, 1)); + assert!(back.contains_key(1)); + assert_eq!(back.tombstones(), 1); + // Compact keeps keys and multi grouping. + back.compact(); + assert_eq!(back.tombstones(), 0); + assert!(back.contains_key(1)); + assert!(back.contains_key(2)); + let probe = rand_unit(dim, 404); + assert_eq!(back.search_keyed(&probe, 5)[0].0, 2); + // Re-serialize stays stable. + let bytes2 = back.to_bytes(); + assert_eq!(bytes2, back.to_bytes()); + assert_eq!(VecqIndex::from_bytes(&bytes2).unwrap().len(), 3); + } + #[test] fn keyed_index_from_file_supports_keyed_adds() { let dim = 64; @@ -1547,11 +1624,8 @@ mod tests { for i in 0..8 { idx.add(&rand_unit(128, i + 50)); } - let bytes = idx.to_bytes(); // v1.2 now - assert_eq!( - u16::from_le_bytes([bytes[4], bytes[5]]), - crate::format::V1_2 - ); + let bytes = idx.to_bytes(); // v1.3 now + assert_eq!(u16::from_le_bytes([bytes[4], bytes[5]]), 259); let mut v11 = bytes.clone(); v11[4] = 257u16.to_le_bytes()[0]; v11[5] = 257u16.to_le_bytes()[1]; diff --git a/docs/BENCHMARK.md b/docs/BENCHMARK.md index 641089b..c9d58d7 100644 --- a/docs/BENCHMARK.md +++ b/docs/BENCHMARK.md @@ -62,10 +62,10 @@ contraction); a unit test enforces AVX2 == scalar and NEON == scalar on overlapping inputs. `search()` additionally batches 4 vectors per pass on both SIMD paths. -> **Format note:** writers emit format v1.2 since the Matryoshka `working_dim` -> feature (issue #24) — identical payload to v1.1 for full-dim indexes, with -> the reserved header field carrying `working_dim` for truncated indexes. -> Readers accept v1, v1.1 and v1.2. +> **Format note:** writers emit format v1.3 — v1.2 added the Matryoshka +> `working_dim` header field (issue #24), v1.3 appends a keyed-slot table so +> keyed APIs survive save/reload (issue #32). Readers accept v1, v1.1, v1.2 +> and v1.3. | architecture | path | selection | |---|---|---| diff --git a/docs/SQLITE.md b/docs/SQLITE.md index ed332c3..32a9b76 100644 --- a/docs/SQLITE.md +++ b/docs/SQLITE.md @@ -42,20 +42,13 @@ CREATE TABLE vecq_shard ( ); ``` -Keys from the keyed API (`add_keyed`/`remove_keyed`) are **not** part of the -file format — persist your own mapping table next to the index: - -```sql -CREATE TABLE vecq_keys ( - key INTEGER PRIMARY KEY, -- the u64 key used in add_keyed() - slot INTEGER NOT NULL -- slot index, valid until compact() -); -``` - -Because `to_bytes()` drops tombstones and re-serializes live slots in order, -**slot indices change whenever you save-then-reload after a `compact()`** — -rewrite `vecq_keys` in the same transaction whenever you save the index (see -below), or simply resolve keys → search results instead of storing slots. +Keys from the keyed API (`add_keyed`/`remove_keyed`) are **part of the file +format since v1.3**: `to_bytes()` stores them and `from_bytes()` restores the +keyed map, so a save/reload round-trip keeps `search_keyed`/`remove_keyed` +working. Slot indices change whenever you save-then-reload after a +`compact()` (the file always stores live slots densely) — rewrite +`vecq_keys` in the same transaction whenever you save the index, or simply +resolve keys → search results instead of storing slots. ## Canonical save/load (Rust + rusqlite)