From d9144648514b5eb18cacb46c933ba195539dd836 Mon Sep 17 00:00:00 2001 From: "Anaz S. Aji" Date: Tue, 22 Sep 2026 20:04:54 +0700 Subject: [PATCH 1/2] =?UTF-8?q?feat(recall):=20budgeted=20context=20pack?= =?UTF-8?q?=20=E2=80=94=20--pack=20on=20CLI,=20HTTP,=20and=20MCP=20(#1281?= =?UTF-8?q?=20Phase=201)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds a selection primitive over unified recall: rank-order preserving greedy fill of a caller-supplied character budget. - uteke-core: new pack_mode module (pack_context + ContextPack envelope {selected, skipped, budget_used, budget_chars}); Uteke:: recall_unified_packed wraps recall_unified. Skip reasons: excluded (caller-supplied memory IDs already injected this turn) and budget (no longer fits). Char-counted (multibyte-safe), per-item overhead constant. Ranking untouched — fusion stays the default strategy; deterministic, LLM-free. 5 unit tests. - CLI: recall --pack / --budget / --exclude-ids with human + --json output. Entity/category filters are honoured; the packed path loudly rejects --at/--related/--where instead of silently dropping them. - HTTP: POST /recall accepts pack, budget_chars, exclude_ids; pack honours the caller's search_type with the same 400 validation as the plain path; integration test covers envelope shape, exclusion reporting, and an impossible budget. - MCP: uteke_recall accepts pack/budget_chars/exclude_ids; tool schema updated. - Docs: cli-reference.md flags + examples, mcp.md tool description, CHANGELOG [Unreleased]. Phase 2 (MMR diversity) intentionally NOT included — experiment gated on LongMemEval + redundancy benchmarks per #1281. Testing: pack_mode 5/5; uteke-server 37 live tests green incl. the new pack API test; workspace suite green before test additions. Live E2E against a scratch UTEKE_HOME with real ONNX embeddings: remember x3, --pack human + JSON envelope (3 selected, 235/4000 chars), --exclude-ids reports reason=excluded for the right IDs, --budget 120 fills 1 item at 60 chars and skips 2 as budget. --- CHANGELOG.md | 4 + crates/uteke-cli/src/cli.rs | 12 ++ crates/uteke-cli/src/commands/mod.rs | 6 + crates/uteke-cli/src/commands/recall.rs | 41 ++++++ crates/uteke-cli/src/output.rs | 30 ++++ crates/uteke-core/src/lib.rs | 39 +++++ crates/uteke-core/src/pack_mode.rs | 183 ++++++++++++++++++++++++ crates/uteke-mcp/src/lib.rs | 54 ++++++- crates/uteke-server/src/handlers.rs | 183 ++++++++++++++++++++++++ crates/uteke-server/src/types.rs | 12 ++ docs/cli-reference.md | 6 + docs/mcp.md | 2 +- 12 files changed, 570 insertions(+), 2 deletions(-) create mode 100644 crates/uteke-core/src/pack_mode.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index f4b77732..4795febb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- **Budgeted context pack: `recall --pack` (#1281 Phase 1)** - `uteke recall --pack [--budget ] [--exclude-ids ]` returns a `{selected, skipped, budget_used, budget_chars}` envelope instead of a bare ranked list: rank-order preserving greedy fill of a character budget, with per-item skip reasons (`excluded` for caller-supplied memory IDs already injected this turn, `budget` for items that no longer fit). Deterministic and LLM-free — ranking is untouched (fusion RRF remains the default strategy); this is a selection primitive, not a re-ranker. Surfaces: CLI flags, HTTP `POST /recall` (`"pack": true`, `"budget_chars"`, `"exclude_ids"`), MCP `uteke_recall` (`pack`, `budget_chars`, `exclude_ids`). Phase 2 (MMR diversity) is an experiment gated on LongMemEval + redundancy benchmarks per the issue. + ## [0.18.2] - 2026-09-18 Patch release. Theme: **fix the unified doc recall flooding bug (#1270)** — diff --git a/crates/uteke-cli/src/cli.rs b/crates/uteke-cli/src/cli.rs index 5eecfd01..ae597883 100644 --- a/crates/uteke-cli/src/cli.rs +++ b/crates/uteke-cli/src/cli.rs @@ -161,6 +161,18 @@ pub enum Commands { /// on document results (#689). #[arg(long)] enrich: bool, + /// Return a budgeted context pack instead of a bare ranked list + /// (#1281): rank-order preserving greedy fill of a character budget, + /// deterministic and LLM-free. Pair with --budget and --exclude-ids. + #[arg(long)] + pack: bool, + /// Character budget for --pack (default: 4000). + #[arg(long, default_value = "4000")] + budget: usize, + /// Memory IDs to exclude from a --pack result (already injected this + /// turn), comma-separated (#1281). + #[arg(long, value_delimiter = ',')] + exclude_ids: Vec, }, /// Show project context summary (memory counts, top tags, recent activity) Context { diff --git a/crates/uteke-cli/src/commands/mod.rs b/crates/uteke-cli/src/commands/mod.rs index 41fc4a94..899a67cf 100644 --- a/crates/uteke-cli/src/commands/mod.rs +++ b/crates/uteke-cli/src/commands/mod.rs @@ -93,6 +93,9 @@ pub(crate) fn run_command(cli: &Cli, uteke: &mut Uteke, config: &Config) -> Resu r#type, enrich, explain, + pack, + budget, + exclude_ids, } => recall::run_recall( cli, uteke, @@ -117,6 +120,9 @@ pub(crate) fn run_command(cli: &Cli, uteke: &mut Uteke, config: &Config) -> Resu r#type.as_deref(), *enrich, *explain, + *pack, + *budget, + exclude_ids, ), Commands::Context { namespace } => { diff --git a/crates/uteke-cli/src/commands/recall.rs b/crates/uteke-cli/src/commands/recall.rs index f91093ee..76aaed39 100644 --- a/crates/uteke-cli/src/commands/recall.rs +++ b/crates/uteke-cli/src/commands/recall.rs @@ -31,6 +31,9 @@ pub(crate) fn run_recall( search_type: Option<&str>, enrich: bool, explain: bool, + pack: bool, + budget: usize, + exclude_ids: &[String], ) -> Result<(), String> { // Resolve search type: --type flag > default (All = unified) let resolved_search_type = match search_type { @@ -202,6 +205,44 @@ pub(crate) fn run_recall( return Ok(()); } + // Budgeted context pack (#1281 Phase 1): unified recall + greedy + // character-budget fill. Deterministic, LLM-free, rank-order preserving. + // Entity/category filters are honoured by the core post-filter; memory + // filters that have no packed-path equivalent are rejected loudly + // instead of being silently dropped. + if pack { + if at.is_some() || related || where_filter.is_some() { + return Err( + "--pack does not support --at/--related/--where; rerun without them".to_string(), + ); + } + let pack_result = uteke + .recall_unified_packed( + query, + limit, + tags_filter, + ns, + min_score, + resolved_search_type, + entity, + category, + enrich, + resolved_strategy, + budget, + exclude_ids, + ) + .map_err(|e| format!("Failed to recall: {e}"))?; + + uteke.reset_salience_recency_config(); + + if cli.json { + output::print_json(&pack_result); + } else { + output::print_pack_human(&pack_result); + } + return Ok(()); + } + if use_unified { // Unified search path (#531) let unified_results = uteke diff --git a/crates/uteke-cli/src/output.rs b/crates/uteke-cli/src/output.rs index c1cc5aad..d58576f2 100644 --- a/crates/uteke-cli/src/output.rs +++ b/crates/uteke-cli/src/output.rs @@ -1,10 +1,40 @@ //! Human-readable and JSON output helpers. +use uteke_core::pack_mode::ContextPack; + /// Print a value as JSON to stdout. pub(crate) fn print_json(value: &T) { println!("{}", serde_json::to_string(value).unwrap()); } +/// Print a context pack (#1281) in human-readable form. +pub(crate) fn print_pack_human(pack: &ContextPack) { + println!( + "Context pack: {} selected, {} skipped, budget {}/{} chars", + pack.selected.len(), + pack.skipped.len(), + pack.budget_used, + pack.budget_chars + ); + if pack.selected.is_empty() { + println!("No results fit the budget."); + return; + } + println!("\n── selected ──"); + print_unified_human(&pack.selected); + if !pack.skipped.is_empty() { + println!("\n── skipped ──"); + for s in &pack.skipped { + let id = s + .memory_id + .as_deref() + .map(|i| format!(" ({i})")) + .unwrap_or_default(); + println!("• [{}] {}{id}", s.reason, s.content); + } + } +} + /// Print tags in human-readable format. pub(crate) fn print_tags_human(tags: &[uteke_core::TagInfo], _by_count: bool) { if tags.is_empty() { diff --git a/crates/uteke-core/src/lib.rs b/crates/uteke-core/src/lib.rs index feaed372..f4376a60 100644 --- a/crates/uteke-core/src/lib.rs +++ b/crates/uteke-core/src/lib.rs @@ -28,6 +28,7 @@ mod maintenance; pub mod memory; pub mod offline_extraction; mod operations; +pub mod pack_mode; pub use operations::RememberOutcome; mod orphans; @@ -2224,6 +2225,44 @@ impl Uteke { Ok(results) } + /// Budgeted context pack over unified recall (#1281 Phase 1). + /// + /// Runs the normal unified recall, then greedily selects results into a + /// caller-supplied character budget (see [`pack_mode`]). Deterministic, + /// LLM-free, rank-order preserving. `exclude_ids` are memory IDs the + /// caller already injected this turn; they come back as `skipped` with + /// reason `excluded`. + #[allow(clippy::too_many_arguments)] + pub fn recall_unified_packed( + &self, + query: &str, + limit: usize, + tags_filter: Option<&[&str]>, + namespace: Option<&str>, + min_score: f32, + search_type: SearchType, + entity_filter: Option<&str>, + category_filter: Option<&str>, + enrich: bool, + strategy: RecallStrategy, + budget_chars: usize, + exclude_ids: &[String], + ) -> Result { + let results = self.recall_unified( + query, + limit, + tags_filter, + namespace, + min_score, + search_type, + entity_filter, + category_filter, + enrich, + strategy, + )?; + Ok(pack_mode::pack_context(results, budget_chars, exclude_ids)) + } + #[allow(clippy::too_many_arguments)] /// Unified search — memories only (backward-compatible path). fn recall_unified_memories( diff --git a/crates/uteke-core/src/pack_mode.rs b/crates/uteke-core/src/pack_mode.rs new file mode 100644 index 00000000..c3f37f56 --- /dev/null +++ b/crates/uteke-core/src/pack_mode.rs @@ -0,0 +1,183 @@ +//! Budgeted context packing for recall results (#1281 Phase 1). +//! +//! Deterministic and LLM-free: greedily fill a character budget from the +//! ranked recall list, honouring caller-supplied `exclude_ids`. Scores and +//! rank order are untouched — this is a selection primitive, not a +//! re-ranker. (MMR diversity re-ranking is the Phase 2 experiment, gated +//! on LongMemEval + redundancy benchmarks before it can ship.) + +use serde::{Deserialize, Serialize}; + +use crate::memory::types::UnifiedSearchResult; + +/// Per-item overhead estimate (JSON envelope fields, separators), in chars. +const ITEM_OVERHEAD_CHARS: usize = 24; + +/// Preview length for skipped items, so the report stays small. +const SKIPPED_PREVIEW_CHARS: usize = 160; + +/// Result of packing recall results into a character budget (#1281). +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ContextPack { + /// Items selected for the context window, in rank order. + pub selected: Vec, + /// Items not selected, in rank order, each with a reason. + pub skipped: Vec, + /// Estimated characters of selected content (incl. per-item overhead). + pub budget_used: usize, + /// The budget that was applied. + pub budget_chars: usize, +} + +/// A recall result left out of the pack, with the reason. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SkippedItem { + /// Content preview (truncated) of the skipped item. + pub content: String, + /// Why it was left out: `excluded` (caller exclude list) or `budget`. + pub reason: String, + /// Memory ID when the item is a memory. + #[serde(skip_serializing_if = "Option::is_none")] + pub memory_id: Option, +} + +/// Pack ranked recall results into `budget_chars` of content. +/// +/// `exclude_ids` are memory IDs the caller already injected this turn; +/// they are dropped first (reason `excluded`). Remaining items are taken +/// in rank order while they fit the remaining budget; items that do not +/// fit are reported as `skipped` (reason `budget`) and scanning continues, +/// so a cheaper lower-ranked item can still fill residual space. Rank +/// order is preserved end-to-end — no knapsack shuffling. +pub fn pack_context( + results: Vec, + budget_chars: usize, + exclude_ids: &[String], +) -> ContextPack { + let mut selected = Vec::new(); + let mut skipped = Vec::new(); + let mut used = 0usize; + + for r in results { + let id = r.memory_id.clone(); + if id + .as_deref() + .is_some_and(|id| exclude_ids.iter().any(|x| x == id)) + { + skipped.push(SkippedItem { + content: preview(&r.content), + reason: "excluded".to_string(), + memory_id: id, + }); + continue; + } + let cost = r.content.chars().count() + ITEM_OVERHEAD_CHARS; + if used + cost > budget_chars { + skipped.push(SkippedItem { + content: preview(&r.content), + reason: "budget".to_string(), + memory_id: id, + }); + continue; + } + used += cost; + selected.push(r); + } + + ContextPack { + selected, + skipped, + budget_used: used, + budget_chars, + } +} + +fn preview(s: &str) -> String { + if s.chars().count() <= SKIPPED_PREVIEW_CHARS { + s.to_string() + } else { + let t: String = s.chars().take(SKIPPED_PREVIEW_CHARS).collect(); + format!("{t}…") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn ur(id: &str, content: &str, score: f32) -> UnifiedSearchResult { + serde_json::from_value(serde_json::json!({ + "result_type": "memory", + "score": score, + "content": content, + "memory_id": id, + })) + .unwrap() + } + + #[test] + fn fills_in_rank_order_until_budget() { + let results = vec![ + ur("m1", &"a".repeat(50), 0.9), + ur("m2", &"b".repeat(50), 0.8), + ur("m3", &"c".repeat(50), 0.7), + ]; + // Budget fits exactly two items (2 × (50 + 24) = 148). + let pack = pack_context(results, 148, &[]); + assert_eq!(pack.selected.len(), 2); + assert_eq!(pack.selected[0].memory_id.as_deref(), Some("m1")); + assert_eq!(pack.selected[1].memory_id.as_deref(), Some("m2")); + assert_eq!(pack.skipped.len(), 1); + assert_eq!(pack.skipped[0].reason, "budget"); + assert_eq!(pack.skipped[0].memory_id.as_deref(), Some("m3")); + assert_eq!(pack.budget_used, 148); + assert!(pack.budget_used <= pack.budget_chars); + } + + #[test] + fn excluded_ids_are_dropped_with_reason() { + let results = vec![ur("m1", "alpha", 0.9), ur("m2", "beta", 0.8)]; + let pack = pack_context(results, 10_000, &["m1".to_string()]); + assert_eq!(pack.selected.len(), 1); + assert_eq!(pack.selected[0].memory_id.as_deref(), Some("m2")); + assert_eq!(pack.skipped.len(), 1); + assert_eq!(pack.skipped[0].reason, "excluded"); + assert_eq!(pack.skipped[0].memory_id.as_deref(), Some("m1")); + } + + #[test] + fn zero_budget_selects_nothing() { + let results = vec![ur("m1", "alpha", 0.9)]; + let pack = pack_context(results, 0, &[]); + assert!(pack.selected.is_empty()); + assert_eq!(pack.skipped.len(), 1); + assert_eq!(pack.skipped[0].reason, "budget"); + assert_eq!(pack.budget_used, 0); + } + + #[test] + fn residual_budget_allows_cheaper_lower_ranked_item() { + let results = vec![ + ur("big", &"x".repeat(90), 0.9), + ur("small", &"y".repeat(10), 0.5), + ]; + // Budget 120: big costs 114 and fits; small costs 34 → skipped. + let pack = pack_context(results.clone(), 120, &[]); + assert_eq!(pack.selected.len(), 1); + // Budget 115: big would exceed (0 + 114 fits? 114 ≤ 115 fits) — use + // 110 so big is skipped (114 > 110) and small (34) still fits. + let pack2 = pack_context(results, 110, &[]); + assert_eq!(pack2.selected.len(), 1); + assert_eq!(pack2.selected[0].memory_id.as_deref(), Some("small")); + assert_eq!(pack2.skipped[0].memory_id.as_deref(), Some("big")); + let _ = pack; // first assertion set already checked + } + + #[test] + fn multibyte_content_counted_by_chars() { + let results = vec![ur("m1", "é".repeat(30).as_str(), 0.9)]; + let pack = pack_context(results, 10_000, &[]); + assert_eq!(pack.selected.len(), 1); + assert_eq!(pack.budget_used, 30 + ITEM_OVERHEAD_CHARS); + } +} diff --git a/crates/uteke-mcp/src/lib.rs b/crates/uteke-mcp/src/lib.rs index a612c9ed..60337d0a 100644 --- a/crates/uteke-mcp/src/lib.rs +++ b/crates/uteke-mcp/src/lib.rs @@ -298,7 +298,10 @@ fn tool_recall() -> Value { "min_score": { "type": "number", "description": "Minimum similarity score 0..1 (default: 0.0)" }, "type": { "type": "string", "enum": ["all", "memory", "doc"], "description": "Search type: 'all' (default, unified), 'memory', or 'doc'" }, "strategy": { "type": "string", "enum": ["fusion", "hybrid", "vector", "fts5", "graph"], "description": "Recall strategy: 'fusion' (default since 0.16.0, weighted RRF of vector×1.7 + hybrid×1, #1123), 'hybrid' (vector+FTS5 via RRF), 'vector' (similarity only), 'fts5' (keyword only), or 'graph' (hybrid + graph-signal reranking)", "default": "fusion" }, - "explain": { "type": "boolean", "description": "Return per-result ranking signals (#1160): vector similarity/rank, RRF contributions, jaccard/salience/recency/graph boosts. Memory-only — omitted type is treated as memory; explicit type=all/doc is rejected." } + "explain": { "type": "boolean", "description": "Return per-result ranking signals (#1160): vector similarity/rank, RRF contributions, jaccard/salience/recency/graph boosts. Memory-only — omitted type is treated as memory; explicit type=all/doc is rejected." }, + "pack": { "type": "boolean", "description": "Return a budgeted context pack (#1281): {selected, skipped, budget_used, budget_chars} instead of a bare list. Deterministic, LLM-free, rank-order preserving. Pair with budget_chars and exclude_ids.", "default": false }, + "budget_chars": { "type": "integer", "description": "Character budget for pack mode (default 4000).", "default": 4000 }, + "exclude_ids": { "type": "array", "items": { "type": "string" }, "description": "Memory IDs already injected this turn; excluded from the pack and reported as skipped[reason=excluded]." } }, "required": ["query"] } @@ -1205,6 +1208,55 @@ fn exec_recall(uteke: &Uteke, args: &Value) -> Result { // Use unified search when type is specified or default (all). // Fall back to legacy recall only for backward compat with existing MCP consumers. + // Budgeted context pack (#1281 Phase 1): `pack: true` returns a + // ContextPack envelope (selected/skipped/budget_used/budget_chars) + // instead of a bare ranked list. Deterministic, LLM-free, rank-order + // preserving; `exclude_ids` are memory IDs already injected this turn. + if args["pack"].as_bool().unwrap_or(false) { + let budget = args["budget_chars"].as_u64().unwrap_or(4000) as usize; + let exclude_ids: Vec = args["exclude_ids"] + .as_array() + .map(|a| { + a.iter() + .filter_map(|v| v.as_str().map(|s| s.to_string())) + .collect() + }) + .unwrap_or_default(); + let pack = uteke + .recall_unified_packed( + query, + limit, + tags_ref, + namespace, + min_score, + search_type, + None, + None, + false, + strategy, + budget, + &exclude_ids, + ) + .map_err(|e| format!("Failed: {e}"))?; + if pack.selected.is_empty() { + return Ok(ToolResult { + content: vec![McpContent::Text { + r#type: "text".to_string(), + text: "No results fit the budget.".to_string(), + }], + is_error: false, + }); + } + let text = serde_json::to_string_pretty(&pack).unwrap_or_else(|_| "{}".to_string()); + return Ok(ToolResult { + content: vec![McpContent::Text { + r#type: "text".to_string(), + text, + }], + is_error: false, + }); + } + let results = uteke .recall_unified( query, diff --git a/crates/uteke-server/src/handlers.rs b/crates/uteke-server/src/handlers.rs index 08d1dcf0..b859a744 100644 --- a/crates/uteke-server/src/handlers.rs +++ b/crates/uteke-server/src/handlers.rs @@ -437,6 +437,48 @@ pub fn route(uteke: &Mutex, ctx: &ReqCtx, req: &mut Request) -> Response< }; } + // Budgeted context pack (#1281 Phase 1): unified recall + + // greedy character-budget fill; deterministic, LLM-free, + // rank-order preserving. Honours the caller's search_type + // (validated with the same 400 as the plain path). + if req_data.pack { + let parsed_search_type = match req_data.search_type.as_deref() { + Some("memory") => uteke_core::SearchType::Memory, + Some("doc") => uteke_core::SearchType::Document, + Some("all") | None => uteke_core::SearchType::All, + Some(other) => { + return ctx.error_response_for( + req, + 400, + format!( + "Invalid search_type: '{other}'. Use 'all', 'memory', or 'doc'." + ), + ); + } + }; + let exclude_ids = req_data.exclude_ids.clone().unwrap_or_default(); + return match uteke.recall_unified_packed( + &req_data.query, + limit, + tags_filter, + ns(&req_data.namespace), + min_score, + parsed_search_type, + entity_filter, + category_filter, + req_data.enrich, + strategy, + req_data.budget_chars.unwrap_or(4000), + &exclude_ids, + ) { + Ok(pack) => ctx.ok_response_for(req, &pack), + Err(e) => { + error!("Pack recall error: {e}"); + ctx.error_response_for(req, 500, "Internal server error") + } + }; + } + // Unified search path (#531): when search_type is specified, // use recall_unified. Entity/category filters are passed // through to the core recall candidate loop (#663). @@ -3625,6 +3667,147 @@ mod contradiction_api_tests { // ── Explain recall API (#1160) ────────────────────────────────────── +#[cfg(test)] +mod pack_recall_api_tests { + //! #1281 Phase 1: `pack` recall returns a budgeted ContextPack envelope. + use super::*; + use tiny_http::TestRequest; + + struct PackApp { + uteke: Mutex, + } + + impl PackApp { + fn new() -> Self { + // No embedder: tests use the fts5 strategy, which needs no + // query embedding (CI-safe without ONNX). + Self { + uteke: Mutex::new( + Uteke::open_with_backend(":memory:", None) + .expect("open in-memory uteke without embedder"), + ), + } + } + + fn call( + &self, + method: Method, + url: &str, + body: Option, + ) -> (u16, serde_json::Value) { + let mut req = match body { + Some(b) => { + let leaked: &'static str = Box::leak(b.into_boxed_str()); + TestRequest::new() + .with_method(method) + .with_path(url) + .with_body(leaked) + .into() + } + None => TestRequest::new().with_method(method).with_path(url).into(), + }; + let ctx = ReqCtx { + auth_token_hash: None, + read_only_token_hash: None, + cors_origins: Vec::new(), + recall_config: None, + extraction_config: None, + }; + let resp = route(&self.uteke, &ctx, &mut req); + let status = resp.status_code().0; + let bytes = resp.into_reader().into_inner(); + let json = serde_json::from_slice(&bytes).unwrap_or(serde_json::Value::Null); + (status, json) + } + + fn remember_in(&self, content: &str, namespace: &str) -> String { + let body = + serde_json::json!({ "content": content, "namespace": namespace }).to_string(); + let (status, resp) = self.call(Method::Post, "/remember", Some(body)); + assert_eq!(status, 200, "remember must succeed: {resp}"); + resp["id"] + .as_str() + .unwrap_or_else(|| panic!("remember response must carry id: {resp}")) + .to_string() + } + } + + #[test] + fn pack_returns_envelope_honours_budget_and_excludes() { + let app = PackApp::new(); + let fox_id = app.remember_in("The quick brown fox jumps over the lazy dog", "pack-ns"); + app.remember_in( + "Completely unrelated content about gardening tools", + "pack-ns", + ); + + // Normal pack: envelope shape with a generous budget. + let body = serde_json::json!({ + "query": "quick brown fox", + "limit": 5, + "namespace": "pack-ns", + "strategy": "fts5", + "pack": true, + "budget_chars": 10000 + }) + .to_string(); + let (status, resp) = app.call(Method::Post, "/recall", Some(body)); + assert_eq!(status, 200, "{resp}"); + assert!( + resp["selected"].is_array(), + "pack must carry selected: {resp}" + ); + assert!( + !resp["selected"].as_array().unwrap().is_empty(), + "fts5 must find the fox" + ); + assert_eq!(resp["budget_chars"], serde_json::json!(10000)); + assert!(resp["budget_used"].as_u64().unwrap() > 0, "{resp}"); + assert!(resp["skipped"].is_array(), "{resp}"); + + // exclude_ids: the fox is already injected → skipped, reason excluded. + let body2 = serde_json::json!({ + "query": "quick brown fox", + "limit": 5, + "namespace": "pack-ns", + "strategy": "fts5", + "pack": true, + "budget_chars": 10000, + "exclude_ids": [fox_id] + }) + .to_string(); + let (status, resp2) = app.call(Method::Post, "/recall", Some(body2)); + assert_eq!(status, 200, "{resp2}"); + let skipped = resp2["skipped"].as_array().unwrap(); + assert!( + skipped + .iter() + .any(|s| s["memory_id"] == serde_json::json!(fox_id) + && s["reason"] == serde_json::json!("excluded")), + "excluded fox must be reported: {resp2}" + ); + + // Impossible budget → nothing selected, everything skipped as budget. + let body3 = serde_json::json!({ + "query": "quick brown fox", + "limit": 5, + "namespace": "pack-ns", + "strategy": "fts5", + "pack": true, + "budget_chars": 1 + }) + .to_string(); + let (status, resp3) = app.call(Method::Post, "/recall", Some(body3)); + assert_eq!(status, 200, "{resp3}"); + assert!(resp3["selected"].as_array().unwrap().is_empty()); + assert_eq!(resp3["budget_used"], serde_json::json!(0)); + assert!( + !resp3["skipped"].as_array().unwrap().is_empty(), + "oversized items must be reported: {resp3}" + ); + } +} + #[cfg(test)] mod explain_recall_api_tests { use super::*; diff --git a/crates/uteke-server/src/types.rs b/crates/uteke-server/src/types.rs index 71217fe1..00e779c0 100644 --- a/crates/uteke-server/src/types.rs +++ b/crates/uteke-server/src/types.rs @@ -249,6 +249,18 @@ pub struct RecallRequest { /// `linked_memory_ids` on document results. #[serde(default)] pub enrich: bool, + /// Budgeted context pack (#1281 Phase 1): return a ContextPack + /// (selected/skipped/budget_used) instead of a bare ranked list. + /// Deterministic, LLM-free, rank-order preserving. + #[serde(default)] + pub pack: bool, + /// Character budget for `pack` (default 4000). + #[serde(default)] + pub budget_chars: Option, + /// Memory IDs already injected this turn; excluded from pack results + /// (reported as `skipped` with reason `excluded`). + #[serde(default)] + pub exclude_ids: Option>, /// Recall strategy: "fusion" (default since 0.16.0), "vector", "fts5", /// "hybrid", or "graph" (#900, #1034, #1123). /// When absent, the server falls back to `[recall] default_strategy` from diff --git a/docs/cli-reference.md b/docs/cli-reference.md index e308755b..9ffd7a1d 100644 --- a/docs/cli-reference.md +++ b/docs/cli-reference.md @@ -122,6 +122,9 @@ uteke recall "API architecture" --type doc uteke recall "deployment" --type memory # Explain mode (#1160): show the ranking signals behind each result uteke recall "database caching" --explain +# Budgeted context pack (#1281): bounded selection for prompt injection +uteke recall "database caching" --pack --budget 4000 +uteke recall "database caching" --pack --exclude-ids abc123,def456 ``` | Flag | Description | @@ -133,6 +136,9 @@ uteke recall "database caching" --explain | `--content-format ` | Content display: `auto` (detect), `text`, `json` (pretty-print JSON memories) | | `--where ` | Filter by JSON field on structured memories (e.g. `--where role=CTO`) | | `--explain` | Show the ranking signals behind each result (#1160): final score, strategy, vector similarity + rank, FTS rank, RRF score with per-channel fusion contributions, and jaccard/salience/recency/graph boost deltas. Memory-only — not available with `--type doc` | +| `--pack` | Return a budgeted context pack (#1281) instead of a bare ranked list: `{selected, skipped, budget_used, budget_chars}`. Rank-order preserving greedy fill of the character budget; deterministic and LLM-free | +| `--budget ` | Character budget for `--pack` (default: 4000) | +| `--exclude-ids ` | Memory IDs already injected this turn, comma-separated — excluded from the pack and reported as `skipped[reason=excluded]` | | `--json` | Output as JSON array | ## uteke search diff --git a/docs/mcp.md b/docs/mcp.md index d14462f5..647f2d45 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -137,7 +137,7 @@ Both transports expose the same 46 tools (MCP protocol version `2025-06-18`): | Tool | Description | |------|-------------| | `uteke_remember` | Store a memory (supports type, room, author, tags) | -| `uteke_recall` | Semantic search (supports tags filter, min_score, strategy: fusion/vector/fts5/hybrid/graph — default `fusion` since 0.16.0, and the `explain` flag showing why each result ranked where it did, #1160) | +| `uteke_recall` | Semantic search (supports tags filter, min_score, strategy: fusion/vector/fts5/hybrid/graph — default `fusion` since 0.16.0, and the `explain` flag showing why each result ranked where it did, #1160). `pack: true` returns a budgeted context pack `{selected, skipped, budget_used, budget_chars}` (#1281) — deterministic, LLM-free, rank-order preserving; pair with `budget_chars` and `exclude_ids` (IDs already injected this turn) | | `uteke_search` | Text search with optional tag filter | | `uteke_list` | List memories (supports pagination via offset, or `include_meta: true` for the `{memories, total, has_more, next_offset}` envelope, #1188) | | `uteke_get` | Fetch a single memory's full record by id — no truncation (accepts UUID or unambiguous prefix) | From 84a5cb2a7e743fe7e4fa65c681a0af86f1814eb9 Mon Sep 17 00:00:00 2001 From: "Anaz S. Aji" Date: Tue, 22 Sep 2026 20:24:51 +0700 Subject: [PATCH 2/2] fix(server): pack recall loudly rejects at/after/before + regen api docs CodeCora review on #1284 flagged that the pack branch validated only search_type: a request with pack+at silently returned non-time-travel results while the plain path honours the filters (and the CLI pack path already rejects --at loudly). - POST /recall with pack + at/after/before now returns the same 400 as the CLI contract; integration test extended. - docs/api-reference.md regenerated via docgen for the new public recall_unified_packed surface (API Docs Fresh gate). --- crates/uteke-server/src/handlers.rs | 33 +++++++++++++++++++++++++++++ docs/api-reference.md | 6 ++++++ 2 files changed, 39 insertions(+) diff --git a/crates/uteke-server/src/handlers.rs b/crates/uteke-server/src/handlers.rs index b859a744..e12050db 100644 --- a/crates/uteke-server/src/handlers.rs +++ b/crates/uteke-server/src/handlers.rs @@ -442,6 +442,19 @@ pub fn route(uteke: &Mutex, ctx: &ReqCtx, req: &mut Request) -> Response< // rank-order preserving. Honours the caller's search_type // (validated with the same 400 as the plain path). if req_data.pack { + // Loud reject like the CLI: pack is not composable with + // point-in-time / range filters (the plain unified path + // consumes them further down, after this early return). + if req_data.at.is_some() + || req_data.after.is_some() + || req_data.before.is_some() + { + return ctx.error_response_for( + req, + 400, + "pack is not supported with at/after/before filters; rerun without time filters", + ); + } let parsed_search_type = match req_data.search_type.as_deref() { Some("memory") => uteke_core::SearchType::Memory, Some("doc") => uteke_core::SearchType::Document, @@ -3805,6 +3818,26 @@ mod pack_recall_api_tests { !resp3["skipped"].as_array().unwrap().is_empty(), "oversized items must be reported: {resp3}" ); + + // pack + time filters → 400 (loud, same contract as the CLI). + let body4 = serde_json::json!({ + "query": "quick brown fox", + "namespace": "pack-ns", + "strategy": "fts5", + "pack": true, + "budget_chars": 10000, + "at": "2026-01-01T00:00:00Z" + }) + .to_string(); + let (status, resp4) = app.call(Method::Post, "/recall", Some(body4)); + assert_eq!(status, 400, "{resp4}"); + assert!( + resp4["error"] + .as_str() + .unwrap_or_default() + .contains("pack is not supported"), + "{resp4}" + ); } } diff --git a/docs/api-reference.md b/docs/api-reference.md index 4bf22eb3..2cc05d47 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -729,11 +729,14 @@ RFC3339 timestamp (#902). | | `at` | any | No | Time-travel: query memories that existed at this RFC3339 timestamp. | | `before` | any | No | Temporal range filter: only return memories created at or before this RFC3339 timestamp (#902). | +| `budget_chars` | any | No | Character budget for `pack` (default 4000). | | `category` | any | No | Filter by category metadata. | | `enrich` | `boolean` | No | Enrich results with cross-entity links (doc↔memory) (#689). When true, populates `linked_doc_slugs` on memory results and `linked_memory_ids` on document results. | | `entity` | any | No | Filter by entity metadata. | +| `exclude_ids` | any | No | Memory IDs already injected this turn; excluded from pack results +(reported as `skipped` with reason `excluded`). | | `explain` | `boolean` | No | Explain mode (#1160): return per-result ranking signals alongside each memory. Memory-only recall — rejected (400) together with search_type/unified, at, before/after. | @@ -741,6 +744,9 @@ search_type/unified, at, before/after. | | `min_score` | any | No | Minimum similarity score (0.0-1.0). Results below are filtered. Default: 0.0 (no filtering). Use `strict=true` for 0.5 default (#995). | | `namespace` | any | No | | +| `pack` | `boolean` | No | Budgeted context pack (#1281 Phase 1): return a ContextPack +(selected/skipped/budget_used) instead of a bare ranked list. +Deterministic, LLM-free, rank-order preserving. | | `query` | `string` | Yes | | | `search_type` | any | No | Search type filter: "all" (default, unified), "memory", or "doc" (#531). | | `strategy` | any | No | Recall strategy: "fusion" (default since 0.16.0), "vector", "fts5",