From 928f190a2e083c56f127e2c0abc10a0decd68773 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 19:10:09 +0200 Subject: [PATCH 01/18] =?UTF-8?q?optimize:=20=20=20=20=201.=20&str=20bodie?= =?UTF-8?q?s=20=E2=80=94=20no=20heap=20String=20per=20symbol=20in=20pass1?= =?UTF-8?q?=20=20=20=20=202.=20Worker=20prep=20=E2=80=94=20line=20offsets?= =?UTF-8?q?=20+=20BLAKE3=20+=20bloom=20on=20extract=20workers=20(SymbolPas?= =?UTF-8?q?s1Prep)=20=20=20=20=203.=20Tracker=20mapping=20=E2=80=94=20accu?= =?UTF-8?q?mulated=20at=20commit=5Fnode;=20skips=20full=20mmap=20scan/sort?= =?UTF-8?q?=20=20=20=20=204.=20CodeIndex=20off=20by=20default=20=E2=80=94?= =?UTF-8?q?=20no=20code=5Findex.json=20(was=201.1=20GB);=20code=5Fhash=20s?= =?UTF-8?q?till=20on=20nodes=20=20=20=20=205.=20Spill=20sort=20runs=20256?= =?UTF-8?q?=20MiB=20(was=2064=20MiB)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- Cargo.toml | 2 +- README.md | 4 +- crates/rgctl-extraction/src/extractor.rs | 87 ++++++++++++++---- crates/rgctl-extraction/src/graph_builder.rs | 60 +++++++++++-- crates/rgctl-extraction/src/lib.rs | 2 +- crates/rgctl-graph/src/code_index.rs | 76 ++++++++++++++-- crates/rgctl-graph/src/segmented_spill.rs | 7 +- crates/rgctl-pipeline/Cargo.toml | 1 + crates/rgctl-pipeline/src/lib.rs | 4 +- crates/rgctl-pipeline/src/pipeline.rs | 57 +++++++++--- crates/rgctl-pipeline/src/stream.rs | 93 +++++++++++++++++++- docs/installation.md | 6 +- docs/internal/profile.md | 16 +++- docs/user-guide.md | 2 +- src/cli/discover_impl.rs | 50 +++++++---- src/cli/discover_output.rs | 1 + src/cli/stage_profile.rs | 32 +++++++ tests/cli_output/discover.rs | 1 + 18 files changed, 421 insertions(+), 80 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 249ec690..98ca8b39 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -48,7 +48,7 @@ resolver = "2" [workspace.package] edition = "2024" -rust-version = "1.88" +rust-version = "1.99" [workspace.dependencies] rgctl-plugin-api = { path = "crates/rgctl-plugin-api", version = "0.4.17" } diff --git a/README.md b/README.md index 800874a1..85f75b24 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ [![Docs](https://img.shields.io/badge/docs-shaaf.dev%2Frgctl-2563eb?style=flat-square&logo=readthedocs&logoColor=white)](https://shaaf.dev/rgctl) [![Website](https://img.shields.io/github/actions/workflow/status/sshaaf/rgctl/website.yml?branch=main&style=flat-square&label=website)](https://shaaf.dev/rgctl) -[![Rust](https://img.shields.io/badge/rust-1.88%2B-orange?style=flat-square&logo=rust)](https://www.rust-lang.org/) +[![Rust](https://img.shields.io/badge/rust-1.99%2B-orange?style=flat-square&logo=rust)](https://www.rust-lang.org/) [![Platforms](https://img.shields.io/badge/platform-macOS%20%7C%20Linux%20%7C%20Windows-555?style=flat-square)](https://github.com/sshaaf/rgctl/releases/latest) [![tree-sitter](https://img.shields.io/badge/parser-tree--sitter-brightgreen?style=flat-square)](https://tree-sitter.github.io/tree-sitter/) [![Agents](https://img.shields.io/badge/agents-Cursor%20%7C%20Claude%20%7C%20Codex-111827?style=flat-square)](https://shaaf.dev/rgctl/docs/guides/agent-commands/) @@ -57,7 +57,7 @@ https://github.com/user-attachments/assets/15ec6d91-f716-4cbd-a873-e982ba3c6dca rgctl --version ``` -**Or build from source** (Rust **1.88+**): +**Or build from source** (Rust **1.99+**): ```bash git clone https://github.com/sshaaf/rgctl.git diff --git a/crates/rgctl-extraction/src/extractor.rs b/crates/rgctl-extraction/src/extractor.rs index 5c556b4e..d3991f9a 100644 --- a/crates/rgctl-extraction/src/extractor.rs +++ b/crates/rgctl-extraction/src/extractor.rs @@ -4,7 +4,9 @@ use crate::discovery::{DiscoveryConfig, FileDiscoverer}; use crate::graph_builder::GraphBuilder; use crate::usage_detector::{ConfigUsage, ConfigUsageDetector}; use rgctl_error::{Error, Result}; -use rgctl_plugin_api::{ConfigKey, Relation, Symbol}; +use rgctl_graph::code_index::hash_code; +use rgctl_graph::structural_sketch::{TokenBloom, build_token_bloom}; +use rgctl_plugin_api::{ConfigKey, Relation, Symbol, SymbolType}; use rgctl_registry::LanguageRegistry; use std::collections::HashMap; use std::path::{Path, PathBuf}; @@ -16,6 +18,15 @@ pub struct Extractor { registry: Arc, } +/// Precomputed pass-1 work for one symbol (filled on extract workers). +#[derive(Debug, Clone, Default)] +pub struct SymbolPass1Prep { + /// BLAKE3 hex digest of the symbol body when present. + pub code_hash: Option, + /// Token bloom for functions (when sketched). + pub token_bloom: Option, +} + /// Result of extracting a single file. #[derive(Debug, Default, Clone)] pub struct FileExtraction { @@ -23,6 +34,8 @@ pub struct FileExtraction { pub path: PathBuf, /// Extracted code symbols pub symbols: Vec, + /// Parallel to [`Self::symbols`]: hash/bloom precomputed on the worker. + pub symbol_preps: Vec, /// Extracted symbol relations pub relations: Vec, /// Extracted configuration keys @@ -95,9 +108,11 @@ impl Extractor { if let Ok(plugin) = self.registry.get_plugin_for_file(path) { let extracted = plugin.extract_all(path, &source)?; let config_usages = ConfigUsageDetector::detect(plugin.language_id(), &source, path); + let symbol_preps = prepare_symbol_pass1(&source, &extracted.symbols); return Ok(FileExtraction { path: path.to_path_buf(), symbols: extracted.symbols, + symbol_preps, relations: extracted.relations, config_keys: Vec::new(), config_usages, @@ -109,9 +124,11 @@ impl Extractor { // Manifests: Dependency extractors (section 3). if self.registry.is_manifest_file(path) { let (symbols, relations) = crate::manifests::extract_manifest(path, &source); + let symbol_preps = prepare_symbol_pass1(&source, &symbols); return Ok(FileExtraction { path: path.to_path_buf(), symbols, + symbol_preps, relations, config_keys: Vec::new(), config_usages: Vec::new(), @@ -120,11 +137,12 @@ impl Extractor { }); } - if let Ok(plugin) = self.registry.get_config_plugin_for_file(path) { - let config_keys = plugin.extract_config_keys(path, &source)?; + if let Ok(config_plugin) = self.registry.get_config_plugin_for_file(path) { + let config_keys = config_plugin.extract_config_keys(path, &source)?; return Ok(FileExtraction { path: path.to_path_buf(), symbols: Vec::new(), + symbol_preps: Vec::new(), relations: Vec::new(), config_keys, config_usages: Vec::new(), @@ -161,20 +179,25 @@ impl Extractor { (!extraction.source.is_empty()).then_some(extraction.source.as_slice()), ); builder.merge_content_blobs(&extraction.content_blobs); - let source = (!extraction.source.is_empty()).then_some(extraction.source.as_slice()); - let line_offsets = source.map(line_start_offsets); if !extraction.symbols.is_empty() { let symbol_start = Instant::now(); - for symbol in &extraction.symbols { - let body = source.and_then(|bytes| { - let offsets = line_offsets.as_ref()?; - symbol_body_from_source(bytes, offsets, symbol) - }); - if let Some(body) = body.as_deref() { - builder.add_symbol_with_body(symbol, file_id, Some(body)); - } else { - builder.add_symbol(symbol, file_id); + let use_preps = extraction.symbol_preps.len() == extraction.symbols.len(); + if use_preps { + for (symbol, prep) in extraction.symbols.iter().zip(extraction.symbol_preps.iter()) + { + builder.add_symbol_with_prep(symbol, file_id, None, Some(prep)); + } + } else { + let source = + (!extraction.source.is_empty()).then_some(extraction.source.as_slice()); + let line_offsets = source.map(line_start_offsets); + for symbol in &extraction.symbols { + let body = source.and_then(|bytes| { + let offsets = line_offsets.as_ref()?; + symbol_body_from_source(bytes, offsets, symbol) + }); + builder.add_symbol_with_prep(symbol, file_id, body, None); } } profile.symbol_processing += symbol_start.elapsed(); @@ -191,6 +214,7 @@ impl Extractor { extraction.content_blobs.clear(); extraction.source.clear(); extraction.symbols.clear(); + extraction.symbol_preps.clear(); extraction.config_keys.clear(); Ok(( @@ -319,11 +343,11 @@ fn line_start_offsets(source: &[u8]) -> Vec { offsets } -fn symbol_body_from_source( - source: &[u8], +fn symbol_body_from_source<'a>( + source: &'a [u8], line_offsets: &[usize], symbol: &Symbol, -) -> Option { +) -> Option<&'a str> { let start = symbol.location.start_line.saturating_sub(1); let end_line = symbol.location.end_line.max(symbol.location.start_line); if start >= line_offsets.len() { @@ -340,8 +364,35 @@ fn symbol_body_from_source( if text.is_empty() { None } else { - Some(text.to_string()) + Some(text) + } +} + +fn prepare_symbol_pass1(source: &[u8], symbols: &[Symbol]) -> Vec { + if symbols.is_empty() { + return Vec::new(); + } + let offsets = line_start_offsets(source); + let mut preps = Vec::with_capacity(symbols.len()); + for symbol in symbols { + let body = symbol_body_from_source(source, &offsets, symbol); + let code_hash = body.map(hash_code); + let token_bloom = if matches!(symbol.symbol_type, SymbolType::Function) { + Some(build_token_bloom( + &symbol.name, + symbol.qualified_name.as_deref(), + symbol.signature.as_deref(), + body, + )) + } else { + None + }; + preps.push(SymbolPass1Prep { + code_hash, + token_bloom, + }); } + preps } #[cfg(test)] diff --git a/crates/rgctl-extraction/src/graph_builder.rs b/crates/rgctl-extraction/src/graph_builder.rs index c8aac20c..7c94917b 100644 --- a/crates/rgctl-extraction/src/graph_builder.rs +++ b/crates/rgctl-extraction/src/graph_builder.rs @@ -1,5 +1,6 @@ //! Maps extracted symbols and relations into graph nodes and edges. +use crate::extractor::SymbolPass1Prep; use rgctl_error::{Error, Result}; use rgctl_graph::code_index::{CodeIndex, hash_code}; use rgctl_graph::content_store::{ContentStore, INLINE_BODY_MAX_BYTES, hash_bytes}; @@ -59,6 +60,8 @@ pub struct GraphBuilder { suffix_resolve_cache: HashMap, /// When false (default discover), `Symbol.fields` stay on symbols only — no Variable nodes. materialize_fields: bool, + /// Path → node ids accumulated during commit (feeds FileTracker without a full mmap scan). + tracker_mapping: HashMap>, } #[derive(Debug, Default)] @@ -196,6 +199,7 @@ impl GraphBuilder { fn commit_node(&mut self, node: Node) { self.record_line_span(&node); + self.record_tracker_mapping(&node); if let Some(spill) = self.spill.as_mut() { if let Err(e) = spill.append_node(&node) { self.spill_error = Some(e.to_string()); @@ -207,6 +211,21 @@ impl GraphBuilder { } } + fn record_tracker_mapping(&mut self, node: &Node) { + let path = node.file_path.as_deref().or_else(|| { + if matches!(node.node_type, NodeType::File) { + Some(node.name.as_str()) + } else { + None + } + }); + let Some(path) = path else { + return; + }; + let key = normalize_path_str(path).into_owned(); + self.tracker_mapping.entry(key).or_default().push(node.id); + } + fn commit_edge(&mut self, edge: Edge) { if let Some(spill) = self.spill.as_mut() { if let Err(e) = spill.append_edge(&edge) { @@ -288,9 +307,14 @@ impl GraphBuilder { self.code_index.take() } + /// Take the path → node-id mapping accumulated during commit (for FileTracker). + pub fn take_tracker_mapping(&mut self) -> HashMap> { + std::mem::take(&mut self.tracker_mapping) + } + /// Add a symbol node linked to its file. pub fn add_symbol(&mut self, symbol: &Symbol, file_id: Uuid) -> Uuid { - self.add_symbol_with_body(symbol, file_id, None) + self.add_symbol_with_prep(symbol, file_id, None, None) } /// Add a symbol node and optionally hash its body for change detection. @@ -299,6 +323,17 @@ impl GraphBuilder { symbol: &Symbol, file_id: Uuid, body: Option<&str>, + ) -> Uuid { + self.add_symbol_with_prep(symbol, file_id, body, None) + } + + /// Add a symbol, preferring worker-precomputed hash/bloom when `prep` is set. + pub fn add_symbol_with_prep( + &mut self, + symbol: &Symbol, + file_id: Uuid, + body: Option<&str>, + prep: Option<&SymbolPass1Prep>, ) -> Uuid { let mut key = symbol_key( &symbol.location.file, @@ -350,16 +385,25 @@ impl GraphBuilder { .collect(), ); } - if let Some(body) = body { - let code_hash = if let Some(index) = self.code_index.as_mut() { - index.add_code(body, &symbol.location) - } else { - hash_code(body) - }; + + let code_hash = prep + .and_then(|p| p.code_hash.clone()) + .or_else(|| { + body.map(|b| { + if let Some(index) = self.code_index.as_mut() { + index.add_code(b, &symbol.location) + } else { + hash_code(b) + } + }) + }); + if let Some(code_hash) = code_hash { node = node.with_code_hash(code_hash); } - if should_sketch_symbol(symbol.symbol_type) { + if let Some(bloom) = prep.and_then(|p| p.token_bloom) { + node = node.with_token_bloom(bloom); + } else if should_sketch_symbol(symbol.symbol_type) { let bloom = build_token_bloom( &symbol.name, symbol.qualified_name.as_deref(), diff --git a/crates/rgctl-extraction/src/lib.rs b/crates/rgctl-extraction/src/lib.rs index b320f909..10dfff0b 100644 --- a/crates/rgctl-extraction/src/lib.rs +++ b/crates/rgctl-extraction/src/lib.rs @@ -8,6 +8,6 @@ pub mod manifests; pub mod usage_detector; pub use discovery::{DiscoveryConfig, FileDiscoverer}; -pub use extractor::{ExtractionTail, Extractor, FileExtraction}; +pub use extractor::{ExtractionTail, Extractor, FileExtraction, SymbolPass1Prep}; pub use graph_builder::GraphBuilder; pub use manifests::{DependencyDeclaration, extract_manifest}; diff --git a/crates/rgctl-graph/src/code_index.rs b/crates/rgctl-graph/src/code_index.rs index e0df747e..a93e2f97 100644 --- a/crates/rgctl-graph/src/code_index.rs +++ b/crates/rgctl-graph/src/code_index.rs @@ -16,7 +16,12 @@ pub struct CodeLocation { pub start_line: usize, /// End line (1-based) pub end_line: usize, - /// Code text that was hashed + /// Optional code text that was hashed. + /// + /// Default discover leaves this empty to avoid multi-GB RAM / `code_index.json` + /// payloads. Enable [`CodeIndex::store_bodies`] only when a caller needs + /// `get_code` lookups. + #[serde(default)] pub code: String, } @@ -26,23 +31,40 @@ pub struct CodeIndex { hash_to_code: HashMap, #[serde(skip)] cache_file: Option, + /// When true, [`Self::add_code`] retains full body text (expensive at scale). + #[serde(skip)] + store_bodies: bool, } impl CodeIndex { - /// Create an empty in-memory index. + /// Create an empty in-memory index (span metadata only; no body text). pub fn new() -> Self { Self::default() } - /// Create an index backed by a cache file path. + /// Create an index backed by a cache file path (span metadata only). pub fn with_cache_file(cache_file: PathBuf) -> Self { Self { hash_to_code: HashMap::new(), cache_file: Some(cache_file), + store_bodies: false, } } + /// Retain full function bodies in the index (opt-in; multi-GB on large repos). + pub fn store_bodies(mut self, enabled: bool) -> Self { + self.store_bodies = enabled; + self + } + + /// Whether full body text is retained. + pub fn stores_bodies(&self) -> bool { + self.store_bodies + } + /// Hash code and record its location. Returns the hex digest. + /// + /// By default only path/line span metadata is stored (empty `code`). pub fn add_code(&mut self, code: &str, location: &SourceLocation) -> String { let hash = hash_code(code); self.hash_to_code.insert( @@ -51,7 +73,11 @@ impl CodeIndex { file_path: location.file.clone(), start_line: location.start_line, end_line: location.end_line, - code: code.to_string(), + code: if self.store_bodies { + code.to_string() + } else { + String::new() + }, }, ); hash @@ -62,9 +88,12 @@ impl CodeIndex { hash_code(current_code) != stored_hash } - /// Look up code by hash. + /// Look up code by hash (only populated when [`Self::store_bodies`] is enabled). pub fn get_code(&self, hash: &str) -> Option<&str> { - self.hash_to_code.get(hash).map(|loc| loc.code.as_str()) + self.hash_to_code + .get(hash) + .map(|loc| loc.code.as_str()) + .filter(|code| !code.is_empty()) } /// Number of indexed fragments. @@ -78,10 +107,15 @@ impl CodeIndex { } /// Persist the index to the configured cache file. + /// + /// Skips writing when the map is empty (default discover path). pub fn save(&self) -> Result<()> { let Some(path) = &self.cache_file else { return Ok(()); }; + if self.hash_to_code.is_empty() { + return Ok(()); + } if let Some(parent) = path.parent() { std::fs::create_dir_all(parent)?; } @@ -91,6 +125,9 @@ impl CodeIndex { } /// Load an index from disk, or return empty if missing. + /// + /// Does **not** enable body storage; loaded `code` fields are kept as-is for + /// callers that opt in later via [`Self::store_bodies`]. pub fn load(cache_file: PathBuf) -> Result { if cache_file.exists() { let json = std::fs::read_to_string(&cache_file)?; @@ -99,12 +136,21 @@ impl CodeIndex { Ok(Self { hash_to_code, cache_file: Some(cache_file), + store_bodies: false, }) } else { Ok(Self::with_cache_file(cache_file)) } } + /// Open a cache path without reading an existing on-disk index into RAM. + /// + /// Use for default discover so a stale multi-GB `code_index.json` cannot + /// inflate cold RSS. + pub fn open_empty(cache_file: PathBuf) -> Self { + Self::with_cache_file(cache_file) + } + /// Default cache path under a repository root. pub fn default_cache_path(repo_root: &Path) -> PathBuf { repo_root.join(".rgctl").join("code_index.json") @@ -130,7 +176,7 @@ mod tests { #[test] fn test_code_index_change_detection() { - let mut index = CodeIndex::new(); + let mut index = CodeIndex::new().store_bodies(true); let loc = SourceLocation { file: "main.rs".to_string(), start_line: 1, @@ -141,5 +187,21 @@ mod tests { let hash = index.add_code("fn old() {}", &loc); assert!(!CodeIndex::has_changed(&hash, "fn old() {}")); assert!(CodeIndex::has_changed(&hash, "fn new() {}")); + assert_eq!(index.get_code(&hash), Some("fn old() {}")); + } + + #[test] + fn default_add_code_omits_body_text() { + let mut index = CodeIndex::new(); + let loc = SourceLocation { + file: "main.rs".to_string(), + start_line: 1, + end_line: 1, + start_column: 0, + end_column: 0, + }; + let hash = index.add_code("fn body() {}", &loc); + assert!(index.get_code(&hash).is_none()); + assert!(index.hash_to_code[&hash].code.is_empty()); } } diff --git a/crates/rgctl-graph/src/segmented_spill.rs b/crates/rgctl-graph/src/segmented_spill.rs index 46b229f7..c1dc8d5d 100644 --- a/crates/rgctl-graph/src/segmented_spill.rs +++ b/crates/rgctl-graph/src/segmented_spill.rs @@ -22,8 +22,11 @@ use std::io::{BufReader, BufWriter, Read, Write}; use std::path::{Path, PathBuf}; use uuid::Uuid; -/// Default run size for external merge-sort (~64 MiB of record payload). -pub const DEFAULT_SORT_RUN_BYTES: usize = 64 * 1024 * 1024; +/// Default run size for external merge-sort (~256 MiB of record payload). +/// +/// Larger runs cut multi-way merge I/O on kernel-scale spills (nodes/edges +/// segs are hundreds of MiB). Peak RSS during sort grows by one run buffer. +pub const DEFAULT_SORT_RUN_BYTES: usize = 256 * 1024 * 1024; const NODE_KEY_LEN: usize = 16; const EDGE_KEY_LEN: usize = 16 + 16 + 8; // from + to + type/pad diff --git a/crates/rgctl-pipeline/Cargo.toml b/crates/rgctl-pipeline/Cargo.toml index 94c7024a..c957a7f6 100644 --- a/crates/rgctl-pipeline/Cargo.toml +++ b/crates/rgctl-pipeline/Cargo.toml @@ -15,6 +15,7 @@ indicatif = "0.17" rayon = "1.7" crossbeam = "0.8" tracing = "0.1" +uuid = { version = "1", features = ["v4", "serde"] } [dev-dependencies] tempfile = { workspace = true } diff --git a/crates/rgctl-pipeline/src/lib.rs b/crates/rgctl-pipeline/src/lib.rs index cac70460..a31c446d 100644 --- a/crates/rgctl-pipeline/src/lib.rs +++ b/crates/rgctl-pipeline/src/lib.rs @@ -9,7 +9,9 @@ pub use parallel::{ with_large_stack, with_pool, }; pub use pipeline::{PipelineConfig, PipelineStats, ProcessingPipeline}; -pub use stream::{DEFAULT_STREAM_CHANNEL_CAPACITY, stream_into_graph}; +pub use stream::{ + DEFAULT_STREAM_CHANNEL_CAPACITY, ExtractPhaseTimings, StreamStats, stream_into_graph, +}; use rgctl_error::Result; use rgctl_graph::CodeGraph; diff --git a/crates/rgctl-pipeline/src/pipeline.rs b/crates/rgctl-pipeline/src/pipeline.rs index 11b652bb..9969c3a0 100644 --- a/crates/rgctl-pipeline/src/pipeline.rs +++ b/crates/rgctl-pipeline/src/pipeline.rs @@ -8,14 +8,15 @@ use rgctl_error::Result; use rgctl_extraction::discovery::{DiscoveryConfig, FileDiscoverer}; use rgctl_extraction::{Extractor, GraphBuilder}; use rgctl_graph::code_graph::CodeGraph; -use rgctl_graph::code_index::CodeIndex; use rgctl_graph::content_store::ContentStore; use rgctl_graph::schema::{Edge, Node}; use rgctl_graph::write_columnar_from_spill; use rgctl_registry::LanguageRegistry; +use std::collections::HashMap; use std::path::Path; use std::sync::Arc; use std::time::{Duration, Instant}; +use uuid::Uuid; /// Options for the processing pipeline. #[derive(Debug, Clone)] @@ -62,10 +63,24 @@ pub struct PipelineStats { pub edges_created: usize, /// Total processing duration pub duration: Duration, - /// Time spent in parallel file extraction (tree-sitter) + /// Wall time for parallel extract + sequential pass-1 (`stream_into_graph`) pub extract_duration: Duration, - /// Time spent merging extractions into the graph + /// Sum of `fs::read` across extract workers (may exceed [`Self::extract_duration`]) + pub extract_read_cpu: Duration, + /// Sum of plugin parse/extract across workers (may exceed [`Self::extract_duration`]) + pub extract_parse_cpu: Duration, + /// Sequential pass-1 merge wall inside extract + pub extract_pass1_wall: Duration, + /// Time spent merging extractions into the graph (indexes + pass-2 + spill/columnar) pub graph_build_duration: Duration, + /// Build symbol resolution indexes (subset of [`Self::graph_build_duration`]) + pub graph_resolution_index: Duration, + /// Pass-2 relation / config-usage resolution + pub graph_pass2: Duration, + /// Spill finish + columnar snapshot compile (snapshot path only) + pub graph_spill_columnar: Duration, + /// Path → node ids collected during extract (skips full mmap scan in save_tracker) + pub node_path_mapping: HashMap>, } /// End-to-end repository processing pipeline. @@ -131,7 +146,8 @@ impl ProcessingPipeline { std::fs::create_dir_all(&spill_dir)?; let mut builder = GraphBuilder::with_spill(&spill_dir)?; builder.set_materialize_fields(self.config.materialize_fields); - builder.set_code_index(CodeIndex::load(CodeIndex::default_cache_path(store))?); + // Default discover: do not load/store CodeIndex bodies (multi-GB on linux). + // Nodes still get `code_hash` via worker-precomputed prep / `hash_code`. builder.set_content_store(ContentStore::load(ContentStore::default_path(store))?); let progress_for_stream = progress.clone(); @@ -156,6 +172,7 @@ impl ProcessingPipeline { let files_processed = stream_stats.files_processed; let files_failed = stream_stats.extraction_failures.len(); + let extract_phases = stream_stats.extract_phases; let graph_start = Instant::now(); let index_start = Instant::now(); @@ -168,7 +185,7 @@ impl ProcessingPipeline { let nodes_created = builder.node_count(); let edges_created = builder.edge_count(); let content_store = builder.take_content_store(); - let code_index = builder.take_code_index(); + let node_path_mapping = builder.take_tracker_mapping(); let spill_start = Instant::now(); let finished = builder.finish_spill()?; let digest = write_columnar_from_spill(finished, snapshot_path)?; @@ -177,14 +194,14 @@ impl ProcessingPipeline { resolution_index_secs = index_elapsed.as_secs_f64(), pass2_relation_resolution_secs = pass2_elapsed.as_secs_f64(), spill_and_columnar_secs = spill_elapsed.as_secs_f64(), + extract_read_cpu_secs = extract_phases.read_cpu.as_secs_f64(), + extract_parse_cpu_secs = extract_phases.parse_cpu.as_secs_f64(), + extract_pass1_wall_secs = extract_phases.pass1_wall.as_secs_f64(), "graph build sub-phase timings" ); if let Some(store) = content_store { store.save()?; } - if let Some(index) = code_index { - index.save()?; - } let graph_build_duration = graph_start.elapsed(); Ok(( @@ -196,7 +213,14 @@ impl ProcessingPipeline { edges_created, duration: start.elapsed(), extract_duration, + extract_read_cpu: extract_phases.read_cpu, + extract_parse_cpu: extract_phases.parse_cpu, + extract_pass1_wall: extract_phases.pass1_wall, graph_build_duration, + graph_resolution_index: index_elapsed, + graph_pass2: pass2_elapsed, + graph_spill_columnar: spill_elapsed, + node_path_mapping, }, digest, )) @@ -231,7 +255,6 @@ impl ProcessingPipeline { let extract_start = Instant::now(); let mut builder = GraphBuilder::new(); builder.set_materialize_fields(self.config.materialize_fields); - builder.set_code_index(CodeIndex::load(CodeIndex::default_cache_path(root))?); builder.set_content_store(ContentStore::load(ContentStore::default_path(root))?); let progress_for_stream = progress.clone(); let (stream_stats, tails) = stream_into_graph( @@ -255,16 +278,19 @@ impl ProcessingPipeline { let files_processed = stream_stats.files_processed; let files_failed = stream_stats.extraction_failures.len(); + let extract_phases = stream_stats.extract_phases; let graph_start = Instant::now(); + let index_start = Instant::now(); builder.build_resolution_indexes(); + let index_elapsed = index_start.elapsed(); + let pass2_start = Instant::now(); extractor.populate_pass2(&tails, &mut builder)?; + let pass2_elapsed = pass2_start.elapsed(); if let Some(store) = builder.take_content_store() { store.save()?; } - if let Some(index) = builder.take_code_index() { - index.save()?; - } + let node_path_mapping = builder.take_tracker_mapping(); let (nodes, edges): (Vec, Vec) = builder.into_graph(); let graph_build_duration = graph_start.elapsed(); @@ -281,7 +307,14 @@ impl ProcessingPipeline { edges_created, duration: start.elapsed(), extract_duration, + extract_read_cpu: extract_phases.read_cpu, + extract_parse_cpu: extract_phases.parse_cpu, + extract_pass1_wall: extract_phases.pass1_wall, graph_build_duration, + graph_resolution_index: index_elapsed, + graph_pass2: pass2_elapsed, + graph_spill_columnar: Duration::ZERO, + node_path_mapping, }, )) } diff --git a/crates/rgctl-pipeline/src/stream.rs b/crates/rgctl-pipeline/src/stream.rs index 9851426f..29d7b98b 100644 --- a/crates/rgctl-pipeline/src/stream.rs +++ b/crates/rgctl-pipeline/src/stream.rs @@ -8,6 +8,8 @@ use rgctl_extraction::{ExtractionTail, Extractor, FileExtraction, GraphBuilder}; use rgctl_registry::LanguageRegistry; use std::path::PathBuf; use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::{Duration, Instant}; /// Default in-flight extraction cap (~1024 file buffers max between extract and merge). pub const DEFAULT_STREAM_CHANNEL_CAPACITY: usize = 1024; @@ -18,10 +20,22 @@ pub struct ExtractionFailure { pub error: String, } +/// CPU-sum / wall sub-timings for [`stream_into_graph`] (Instant only; no extra I/O). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct ExtractPhaseTimings { + /// Sum of `fs::read` across worker threads (can exceed wall). + pub read_cpu: Duration, + /// Sum of plugin extract (`extract_file_with_source`) across workers (can exceed wall). + pub parse_cpu: Duration, + /// Sequential pass-1 merge wall on the consumer thread. + pub pass1_wall: Duration, +} + #[derive(Debug, Clone, PartialEq, Eq, Default)] pub struct StreamStats { pub files_processed: usize, pub extraction_failures: Vec, + pub extract_phases: ExtractPhaseTimings, } /// Run parallel extractors into a bounded channel while the caller consumes on the main thread. @@ -31,15 +45,36 @@ pub fn start_parallel_extraction( files: Arc>, capacity: usize, on_file_done: impl Fn() + Send + Sync + 'static, + read_ns: Arc, + parse_ns: Arc, ) -> Receiver> { let (tx, rx) = bounded(capacity); std::thread::spawn(move || { with_pool(thread_count, || { files.par_iter().for_each(|path| { let extractor = Extractor::new(Arc::clone(®istry)); - match extractor.extract_file(path) { - Ok(extraction) => { - let _ = tx.send(Ok(extraction)); + let read_start = Instant::now(); + let read_result = std::fs::read(path); + read_ns.fetch_add(duration_as_nanos_u64(read_start.elapsed()), Ordering::Relaxed); + match read_result { + Ok(source) => { + let parse_start = Instant::now(); + let extracted = extractor.extract_file_with_source(path, source); + parse_ns.fetch_add( + duration_as_nanos_u64(parse_start.elapsed()), + Ordering::Relaxed, + ); + match extracted { + Ok(extraction) => { + let _ = tx.send(Ok(extraction)); + } + Err(err) => { + let _ = tx.send(Err(ExtractionFailure { + path: path.clone(), + error: err.to_string(), + })); + } + } } Err(err) => { let _ = tx.send(Err(ExtractionFailure { @@ -67,14 +102,27 @@ pub fn stream_into_graph( ) -> Result<(StreamStats, Vec)> { let files = Arc::new(files.to_vec()); let file_count = files.len(); - let rx = start_parallel_extraction(thread_count, registry, files, capacity, on_file_done); + let read_ns = Arc::new(AtomicU64::new(0)); + let parse_ns = Arc::new(AtomicU64::new(0)); + let rx = start_parallel_extraction( + thread_count, + registry, + files, + capacity, + on_file_done, + Arc::clone(&read_ns), + Arc::clone(&parse_ns), + ); let mut tails = Vec::with_capacity(file_count); let mut stats = StreamStats::default(); + let mut pass1_wall = Duration::ZERO; while let Ok(result) = rx.recv() { match result { Ok(mut extraction) => { + let pass1_start = Instant::now(); tails.push(extractor.populate_pass1(&mut extraction, builder)?); + pass1_wall += pass1_start.elapsed(); stats.files_processed += 1; } Err(failure) => { @@ -83,9 +131,19 @@ pub fn stream_into_graph( } } + stats.extract_phases = ExtractPhaseTimings { + read_cpu: Duration::from_nanos(read_ns.load(Ordering::Relaxed)), + parse_cpu: Duration::from_nanos(parse_ns.load(Ordering::Relaxed)), + pass1_wall, + }; + Ok((stats, tails)) } +fn duration_as_nanos_u64(d: Duration) -> u64 { + u64::try_from(d.as_nanos()).unwrap_or(u64::MAX) +} + #[cfg(test)] mod tests { use super::*; @@ -114,4 +172,31 @@ mod tests { assert_eq!(stats.extraction_failures[0].path, missing); assert!(tails.is_empty()); } + + #[test] + fn stream_records_extract_phase_timings() { + let temp = TempDir::new().unwrap(); + let path = temp.path().join("ok.rs"); + std::fs::write(&path, "fn hello() {}\n").unwrap(); + + let registry = Arc::new(rgctl_languages::default_registry()); + let extractor = Extractor::new(Arc::clone(®istry)); + let mut builder = GraphBuilder::new(); + + let (stats, tails) = stream_into_graph( + Some(1), + &extractor, + registry, + &[path], + 8, + &mut builder, + || {}, + ) + .unwrap(); + + assert_eq!(stats.files_processed, 1); + assert_eq!(tails.len(), 1); + assert!(stats.extract_phases.parse_cpu > Duration::ZERO); + assert!(stats.extract_phases.pass1_wall > Duration::ZERO); + } } diff --git a/docs/installation.md b/docs/installation.md index de922c86..210224b9 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -27,7 +27,7 @@ Everything you need to install rgctl (`rgctl`), choose the right operating mode, | Requirement | Notes | |-------------|-------| | **OS** | macOS (Apple Silicon or Intel), Linux (x86_64), Windows (x86_64) | -| **Rust 1.88+** | Only for building from source ([rustup.rs](https://rustup.rs/)). Pre-built binaries need no Rust toolchain. | +| **Rust 1.99+** | Only for building from source ([rustup.rs](https://rustup.rs/)). Pre-built binaries need no Rust toolchain. | | **Git** | For cloning the repository (source builds) | | **Git LFS** | Optional. Only required if you use `semantic index --embedder code-daemon` (~206 MB ONNX weights). The default `vocab` embedder needs no LFS. | @@ -71,7 +71,7 @@ Expand-Archive rgctl-*-x86_64-pc-windows-msvc.zip -DestinationPath . ### Option B -- Build from source -Requires **Rust 1.88+** (workspace `rust-version`; edition 2024). Check with `rustc --version`. +Requires **Rust 1.99+** (workspace `rust-version`; edition 2024). Check with `rustc --version`. ```bash git clone https://github.com/sshaaf/rgctl.git @@ -372,7 +372,7 @@ Start with the default mode (no extra flags). Add `--with-cfg`, `--with-taint`, ### Build from source fails -- Confirm **Rust 1.88+** (workspace `rust-version`): `rustc --version` +- Confirm **Rust 1.99+** (workspace `rust-version`): `rustc --version` - Update Rust: `rustup update` - Clean build: `cargo clean && cargo build --release --bin rgctl` - ONNX / `ort-sys` link errors: `cargo build --release --bin rgctl --no-default-features` (disables default `semantic-onnx`; default `vocab` semantic search still works) diff --git a/docs/internal/profile.md b/docs/internal/profile.md index af4e3fea..752106b9 100644 --- a/docs/internal/profile.md +++ b/docs/internal/profile.md @@ -127,11 +127,25 @@ rm -rf .rgctl | Line | Meaning | |------|---------| | `[profile] discover summary` | Wall time, `index_secs`, `post_index_secs`, peak RSS (`peak_rss_mb`, `ingest_peak_rss_mb`, `analysis_peak_rss_mb`), node/function counts | -| `[profile] stage` | Per-stage wall seconds and `%` of discover wall (`index_extract`, `index_graph_build`, `centrality`, `cfg_total`, `save_dashboard`, `kantra_eval`, `kantra_index`, …) | +| `[profile] stage` | Per-stage wall seconds and `%` of discover wall (`index_extract`, `extract_pass1`, `index_graph_build`, `graph_resolution_index`, `graph_pass2`, `graph_spill_columnar`, `centrality`, …) | +| `[profile] extract cpu stage` | CPU-sum across extract workers (`extract_read`, `extract_parse`); may exceed `index_extract` wall | | `[profile] centrality breakdown` | PageRank / betweenness / harmonic sub-times | | `[profile] save_dashboard stage` | Dashboard export substeps (e.g. `export_cfg_slice`) | | `[profile] cfg cpu stage` | CFG thread CPU sums (can exceed wall on parallel passes) | +`index_extract` wall = parallel file extract + sequential pass-1 merge. Sub-breakdown: + +| Stage | Kind | Meaning | +|-------|------|---------| +| `extract_read` | CPU sum | `fs::read` across workers | +| `extract_parse` | CPU sum | plugin `extract_all` / tree-sitter across workers | +| `extract_pass1` | Wall | sequential symbol/config commit on the merge thread | +| `graph_resolution_index` | Wall | build resolution indexes | +| `graph_pass2` | Wall | relation / config-usage resolution | +| `graph_spill_columnar` | Wall | spill finish + columnar snapshot compile (`DEFAULT_SORT_RUN_BYTES` = 256 MiB) | + +Default discover does **not** attach a body-storing `CodeIndex` (avoids multi-GB `code_index.json` / RAM). Nodes still receive `code_hash` from worker-precomputed prep. + Harmonic runs only when `--with-harmonic` or **`discover --full`** (deep stage). Default linux discover skips harmonic and dashboard export. ### Kantra stages (`--with-kantra`) diff --git a/docs/user-guide.md b/docs/user-guide.md index bf301940..796c60a1 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -76,7 +76,7 @@ If no release is published yet for your platform, use [Option B](#option-b--buil ### Option B — Build from source -Requires **Rust 1.88+** (Edition 2024; [rustup.rs](https://rustup.rs/)). +Requires **Rust 1.99+** (Edition 2024; [rustup.rs](https://rustup.rs/)). ```bash git clone https://github.com/sshaaf/rgctl.git diff --git a/src/cli/discover_impl.rs b/src/cli/discover_impl.rs index 5cc4368e..d435973e 100644 --- a/src/cli/discover_impl.rs +++ b/src/cli/discover_impl.rs @@ -180,7 +180,7 @@ pub(crate) fn run_full_analysis( let index_start = Instant::now(); let graph_from_snapshot = !force_reindex && file_changes.is_empty() && snapshot_path.is_file(); let mut cold_reused: Option = None; - let (index_stats, graph_digest) = if graph_from_snapshot { + let (mut index_stats, graph_digest) = if graph_from_snapshot { let load_start = Instant::now(); let cold = crate::analysis::ColdMetadataDb::open(&snapshot_path)?; let digest = cold.store().content_digest()?.to_string(); @@ -202,6 +202,7 @@ pub(crate) fn run_full_analysis( duration: load_elapsed, extract_duration: Duration::default(), graph_build_duration: load_elapsed, + ..Default::default() }; cold_reused = Some(cold); (stats, digest) @@ -218,7 +219,13 @@ pub(crate) fn run_full_analysis( }; profile.index_pipeline.secs = secs(index_start.elapsed()); profile.index_extract.secs = secs(index_stats.extract_duration); + profile.extract_pass1.secs = secs(index_stats.extract_pass1_wall); + profile.extract_read_cpu.secs = secs(index_stats.extract_read_cpu); + profile.extract_parse_cpu.secs = secs(index_stats.extract_parse_cpu); profile.index_graph_build.secs = secs(index_stats.graph_build_duration); + profile.graph_resolution_index.secs = secs(index_stats.graph_resolution_index); + profile.graph_pass2.secs = secs(index_stats.graph_pass2); + profile.graph_spill_columnar.secs = secs(index_stats.graph_spill_columnar); profile.nodes = index_stats.nodes_created; // Snapshot write is folded into index_graph_build (Lever 1: no separate backend rewrite). profile.save_snapshot.secs = 0.0; @@ -1028,26 +1035,31 @@ pub(crate) fn run_full_analysis( // Save graph topology (no analysis properties!) let save_tracker_start = Instant::now(); - let mut node_path_pairs: Vec<(String, uuid::Uuid)> = Vec::with_capacity(cold.node_count()); - cold.for_each_node(&mut |node| { - let raw_path = node.file_path.as_deref().or_else(|| { - if matches!(node.node_type, NodeType::File) { - Some(node.name.as_str()) - } else { - None + let node_mapping = if !index_stats.node_path_mapping.is_empty() { + std::mem::take(&mut index_stats.node_path_mapping) + } else { + let mut node_path_pairs: Vec<(String, uuid::Uuid)> = + Vec::with_capacity(cold.node_count()); + cold.for_each_node(&mut |node| { + let raw_path = node.file_path.as_deref().or_else(|| { + if matches!(node.node_type, NodeType::File) { + Some(node.name.as_str()) + } else { + None + } + }); + if let Some(path) = raw_path { + node_path_pairs.push((crate::incremental::normalize_path_str(path), node.id)); } - }); - if let Some(path) = raw_path { - node_path_pairs.push((crate::incremental::normalize_path_str(path), node.id)); + })?; + const PAR_SORT_NODE_PATHS_MIN: usize = 32_768; + if node_path_pairs.len() >= PAR_SORT_NODE_PATHS_MIN { + node_path_pairs.par_sort_unstable_by(|a, b| a.0.cmp(&b.0)); + } else { + node_path_pairs.sort_unstable_by(|a, b| a.0.cmp(&b.0)); } - })?; - const PAR_SORT_NODE_PATHS_MIN: usize = 32_768; - if node_path_pairs.len() >= PAR_SORT_NODE_PATHS_MIN { - node_path_pairs.par_sort_unstable_by(|a, b| a.0.cmp(&b.0)); - } else { - node_path_pairs.sort_unstable_by(|a, b| a.0.cmp(&b.0)); - } - let node_mapping = crate::incremental::group_sorted_node_paths(node_path_pairs); + crate::incremental::group_sorted_node_paths(node_path_pairs) + }; file_tracker.index_files_with_mapping(&files, node_mapping)?; file_tracker.save()?; profile.save_tracker.secs = secs(save_tracker_start.elapsed()); diff --git a/src/cli/discover_output.rs b/src/cli/discover_output.rs index ba269772..37a81ad3 100644 --- a/src/cli/discover_output.rs +++ b/src/cli/discover_output.rs @@ -66,6 +66,7 @@ pub fn fixture_discover_response() -> DiscoverJsonResponse { duration: std::time::Duration::from_millis(18_200), extract_duration: std::time::Duration::from_millis(12_000), graph_build_duration: std::time::Duration::from_millis(6_200), + ..Default::default() }, 18_200, ) diff --git a/src/cli/stage_profile.rs b/src/cli/stage_profile.rs index c6c4a200..c729408a 100644 --- a/src/cli/stage_profile.rs +++ b/src/cli/stage_profile.rs @@ -14,7 +14,19 @@ pub struct DiscoverStageReport { pub wall_total: StageTiming, pub index_pipeline: StageTiming, pub index_extract: StageTiming, + /// Sequential pass-1 merge wall (subset of `index_extract`). + pub extract_pass1: StageTiming, + /// Sum of file-read CPU across extract workers (may exceed wall). + pub extract_read_cpu: StageTiming, + /// Sum of plugin parse CPU across extract workers (may exceed wall). + pub extract_parse_cpu: StageTiming, pub index_graph_build: StageTiming, + /// Resolution-index build (subset of `index_graph_build`). + pub graph_resolution_index: StageTiming, + /// Pass-2 relation resolution (subset of `index_graph_build`). + pub graph_pass2: StageTiming, + /// Spill + columnar compile (subset of `index_graph_build`). + pub graph_spill_columnar: StageTiming, pub topology: StageTiming, pub community: StageTiming, pub complexity: StageTiming, @@ -83,7 +95,11 @@ impl DiscoverStageReport { let stages: &[(&str, f64)] = &[ ("index_extract", self.index_extract.secs), + ("extract_pass1", self.extract_pass1.secs), ("index_graph_build", self.index_graph_build.secs), + ("graph_resolution_index", self.graph_resolution_index.secs), + ("graph_pass2", self.graph_pass2.secs), + ("graph_spill_columnar", self.graph_spill_columnar.secs), ("topology", self.topology.secs), ("community", self.community.secs), ("complexity", self.complexity.secs), @@ -145,6 +161,22 @@ impl DiscoverStageReport { ); } + for (name, secs) in [ + ("extract_read", self.extract_read_cpu.secs), + ("extract_parse", self.extract_parse_cpu.secs), + ] { + if secs <= 0.0 { + continue; + } + tracing::info!( + target: "profile", + stage = name, + cpu_secs = secs, + extract_wall_secs = self.index_extract.secs, + "[profile] extract cpu stage (sum across workers; may exceed index_extract wall)" + ); + } + for (name, secs) in [ ("cfg_build", self.cfg_build.secs), ("cfg_dominator", self.cfg_dominator.secs), diff --git a/tests/cli_output/discover.rs b/tests/cli_output/discover.rs index 8885c3d1..9f8cc163 100644 --- a/tests/cli_output/discover.rs +++ b/tests/cli_output/discover.rs @@ -43,6 +43,7 @@ fn test_discover_build_maps_pipeline_stats() { duration: std::time::Duration::from_millis(500), extract_duration: std::time::Duration::from_millis(300), graph_build_duration: std::time::Duration::from_millis(200), + ..Default::default() }; let response = build_discover_response(&stats, 750); assert_eq!(response.metrics.files_discovered, 100); From 341628f849b3fd5ce7e95922d897e3848465b692 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 19:59:38 +0200 Subject: [PATCH 02/18] =?UTF-8?q?=20=20What=20landed:=20=20=20=E2=80=A2=20?= =?UTF-8?q?Hash=20reuse=20=E2=80=94=20workers=20set=20FileExtraction.file?= =?UTF-8?q?=5Fhash;=20stream=20=E2=86=92=20PipelineStats=20=E2=86=92=20=20?= =?UTF-8?q?=20=20=20index=5Ffiles=5Fwith=5Fmapping(...,=20Some(hashes))=20?= =?UTF-8?q?(no=2071k=20re-read)=20=20=20=E2=80=A2=20Empty=20tracker=20?= =?UTF-8?q?=E2=80=94=20detect=5Fchanges=20marks=20all=20as=20added=20witho?= =?UTF-8?q?ut=20hashing=20(still=20non-empty=20so=20stale=20=20=20=20=20sn?= =?UTF-8?q?apshots=20aren=E2=80=99t=20reused)=20=20=20=E2=80=A2=20Spill=20?= =?UTF-8?q?=E2=80=94=20reusable=20scratch=20+=20bincode::serialize=5Finto?= =?UTF-8?q?=20=20=20=E2=80=A2=20Pass-1=20batch=20=E2=80=94=20begin=5Ffile?= =?UTF-8?q?=5Fbatch=20/=20get=5Fmut=20on=20tracker=20keys?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 19 +++++++ crates/rgctl-extraction/src/extractor.rs | 9 ++++ crates/rgctl-extraction/src/graph_builder.rs | 40 +++++++++++--- crates/rgctl-graph/src/segmented_spill.rs | 19 ++++--- crates/rgctl-incremental/src/file_tracker.rs | 57 +++++++++++++++++++- crates/rgctl-pipeline/src/pipeline.rs | 6 +++ crates/rgctl-pipeline/src/stream.rs | 9 ++++ src/cli/discover_impl.rs | 6 ++- 8 files changed, 149 insertions(+), 16 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 562f0485..1f56033c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -16,6 +16,7 @@ - **Parallel ingest:** Per-file plugin extraction runs on the discover worker pool. Do not replace with a serial whole-repo walk when parallel ingest exists. - **Streaming commits:** Emit symbols/relations file-by-file; avoid unbounded `Vec` / whole-repo ASTs before commit (`rgctl-extraction` spill patterns). - **Clone hygiene:** Prefer `&[u8]` / `Cow` / borrows in tree-sitter walkers; `Vec::with_capacity` when sizes are known; no `unwrap()` in library paths. +- **Ingest hot path:** Follow **Ingest hot-path practices** below (no per-symbol heap strings, hash/prep once on workers, spill scratch reuse, tracker mapping without re-scan). - **Typed graph:** Respect `EdgeType` / node kinds; do not invent ad-hoc string edges for hot paths. - **Artifacts:** Session data lives in `{repo}/.rgctl/`. Warm caches invalidate wall-time claims. - **Features:** Default semantic embedder is compiled **vocab**. Do not require ONNX / Python ML unless behind an explicit feature (e.g. `semantic-onnx` / code-daemon + Git LFS). @@ -44,6 +45,24 @@ Applies to all extraction / language / discover hot-path work (and OpenSpec `*-e 3. **Streaming** — incremental graph commit; match extraction spill/channel patterns. 4. **Idiomatic Rust** — `Result` + `thiserror`; follow `rgctl-lang-java` / `rgctl-extraction` conventions. +### Ingest hot-path practices + +Rules distilled from linux cold-discover work (`index_extract` / pass-1 / spill / `save_tracker`). Breaking these usually shows up as Gate A wall or RSS regressions — treat O(files)×O(symbols) heap work as a bug. + +| Practice | Do | Don't | +|----------|----|-------| +| **No heap strings per symbol** | Pass `&str` / slices into pass-1 (`add_symbol_with_prep`); borrow file bytes | `String::from` / `to_string()` for every symbol body or path key on the merge thread | +| **Prep on workers** | Compute line offsets, BLAKE3 `code_hash`, token bloom in `SymbolPass1Prep` on extract workers | Re-walk source / re-hash on the sequential pass-1 thread | +| **Hash once** | Set `FileExtraction.file_hash` from bytes already in memory; thread through `StreamStats` / `PipelineStats` into `FileTracker::index_files_with_mapping` | Re-`fs::read` + BLAKE3 all files in `save_tracker` after extract already hashed them | +| **Empty-tracker short-circuit** | When `file_hashes.json` is empty, `detect_changes` marks all paths **added** without hashing (keeps ChangeSet non-empty so a stale snapshot is not reused) | Hash the whole tree twice on cold discover (detect + index) | +| **Normalize / map once per file** | `GraphBuilder::begin_file_batch` → `get_mut` on the active tracker key; accumulate `tracker_mapping` at `commit_node` | `normalize_path_str(...).into_owned()` per symbol; full mmap node scan/sort just to rebuild file→node ids | +| **Spill alloc reuse** | `SegmentedSpill` scratch `Vec` + `bincode::serialize_into`; keep sort runs at `DEFAULT_SORT_RUN_BYTES` (256 MiB) unless profiling says otherwise | Fresh `bincode::serialize` → new `Vec` per node/edge; shrinking sort runs without a cold gate | +| **CodeIndex bodies off by default** | Default discover: no body-storing `CodeIndex` (no multi-GB `code_index.json`); nodes still get `code_hash` from prep | Attach a full CodeIndex on the cold path “for convenience” | + +When adding extract or graph-commit code, ask: *does this allocate or re-read once per symbol/file on the sequential merge thread?* If yes, move it to workers or reuse an existing buffer/key. + +Stage meanings and current linux notes: [docs/internal/profile.md](docs/internal/profile.md). + ### Cold profile (mandatory for scale / perf claims) 1. **Release binary only:** `cargo build --release --bin rgctl` diff --git a/crates/rgctl-extraction/src/extractor.rs b/crates/rgctl-extraction/src/extractor.rs index d3991f9a..6beb900c 100644 --- a/crates/rgctl-extraction/src/extractor.rs +++ b/crates/rgctl-extraction/src/extractor.rs @@ -5,6 +5,7 @@ use crate::graph_builder::GraphBuilder; use crate::usage_detector::{ConfigUsage, ConfigUsageDetector}; use rgctl_error::{Error, Result}; use rgctl_graph::code_index::hash_code; +use rgctl_graph::content_store::hash_bytes; use rgctl_graph::structural_sketch::{TokenBloom, build_token_bloom}; use rgctl_plugin_api::{ConfigKey, Relation, Symbol, SymbolType}; use rgctl_registry::LanguageRegistry; @@ -32,6 +33,8 @@ pub struct SymbolPass1Prep { pub struct FileExtraction { /// Path to the source file pub path: PathBuf, + /// BLAKE3 hex of `source` (computed once on the extract worker). + pub file_hash: Option, /// Extracted code symbols pub symbols: Vec, /// Parallel to [`Self::symbols`]: hash/bloom precomputed on the worker. @@ -105,12 +108,14 @@ impl Extractor { /// recorded on the returned symbols and relations, but it does not have to /// exist on disk. pub fn extract_file_with_source(&self, path: &Path, source: Vec) -> Result { + let file_hash = Some(hash_bytes(&source)); if let Ok(plugin) = self.registry.get_plugin_for_file(path) { let extracted = plugin.extract_all(path, &source)?; let config_usages = ConfigUsageDetector::detect(plugin.language_id(), &source, path); let symbol_preps = prepare_symbol_pass1(&source, &extracted.symbols); return Ok(FileExtraction { path: path.to_path_buf(), + file_hash, symbols: extracted.symbols, symbol_preps, relations: extracted.relations, @@ -127,6 +132,7 @@ impl Extractor { let symbol_preps = prepare_symbol_pass1(&source, &symbols); return Ok(FileExtraction { path: path.to_path_buf(), + file_hash, symbols, symbol_preps, relations, @@ -141,6 +147,7 @@ impl Extractor { let config_keys = config_plugin.extract_config_keys(path, &source)?; return Ok(FileExtraction { path: path.to_path_buf(), + file_hash, symbols: Vec::new(), symbol_preps: Vec::new(), relations: Vec::new(), @@ -174,6 +181,7 @@ impl Extractor { use std::time::Instant; let mut profile = Pass1Profile::default(); + builder.begin_file_batch(&extraction.path); let file_id = builder.ensure_file_node_with_source( &extraction.path, (!extraction.source.is_empty()).then_some(extraction.source.as_slice()), @@ -216,6 +224,7 @@ impl Extractor { extraction.symbols.clear(); extraction.symbol_preps.clear(); extraction.config_keys.clear(); + builder.end_file_batch(); Ok(( ExtractionTail { diff --git a/crates/rgctl-extraction/src/graph_builder.rs b/crates/rgctl-extraction/src/graph_builder.rs index 7c94917b..792b51e5 100644 --- a/crates/rgctl-extraction/src/graph_builder.rs +++ b/crates/rgctl-extraction/src/graph_builder.rs @@ -62,6 +62,8 @@ pub struct GraphBuilder { materialize_fields: bool, /// Path → node ids accumulated during commit (feeds FileTracker without a full mmap scan). tracker_mapping: HashMap>, + /// Normalized path for the current pass-1 file (avoids re-normalize + re-hash per symbol). + active_tracker_key: Option, } #[derive(Debug, Default)] @@ -125,6 +127,17 @@ impl GraphBuilder { } } + /// Pin tracker mapping key for the duration of one file's pass-1 insert. + pub fn begin_file_batch(&mut self, path: &Path) { + let key = normalize_path_str(&path.to_string_lossy()).into_owned(); + self.active_tracker_key = Some(key); + } + + /// Clear the active pass-1 file key. + pub fn end_file_batch(&mut self) { + self.active_tracker_key = None; + } + fn record_line_span(&mut self, node: &Node) { let Some(file) = node.file_path.as_deref() else { return; @@ -133,14 +146,16 @@ impl GraphBuilder { return; }; let end = node.end_line.unwrap_or(start); - self.file_line_spans - .entry(file.to_string()) - .or_default() - .push(LineSpan { - start, - end, - id: node.id, - }); + let span = LineSpan { + start, + end, + id: node.id, + }; + if let Some(spans) = self.file_line_spans.get_mut(file) { + spans.push(span); + } else { + self.file_line_spans.insert(file.to_string(), vec![span]); + } } fn index_symbol_resolution(&mut self, key: &str, node: &Node) { @@ -212,6 +227,15 @@ impl GraphBuilder { } fn record_tracker_mapping(&mut self, node: &Node) { + if let Some(key) = self.active_tracker_key.as_ref() { + if let Some(ids) = self.tracker_mapping.get_mut(key) { + ids.push(node.id); + } else { + let key = key.clone(); + self.tracker_mapping.insert(key, vec![node.id]); + } + return; + } let path = node.file_path.as_deref().or_else(|| { if matches!(node.node_type, NodeType::File) { Some(node.name.as_str()) diff --git a/crates/rgctl-graph/src/segmented_spill.rs b/crates/rgctl-graph/src/segmented_spill.rs index c1dc8d5d..187fb51e 100644 --- a/crates/rgctl-graph/src/segmented_spill.rs +++ b/crates/rgctl-graph/src/segmented_spill.rs @@ -38,6 +38,8 @@ pub struct SegmentedSpill { edges: BufWriter, node_count: usize, edge_count: usize, + /// Reused bincode buffer (avoids a fresh `Vec` per append). + scratch: Vec, } /// Closed spill ready for external sort + columnar compile. @@ -60,6 +62,7 @@ impl SegmentedSpill { edges, node_count: 0, edge_count: 0, + scratch: Vec::with_capacity(64 * 1024), }) } @@ -80,11 +83,13 @@ impl SegmentedSpill { /// Append a node as length-prefixed bincode with UUID key prefix. pub fn append_node(&mut self, node: &Node) -> Result<()> { - let blob = bincode::serialize(node) + self.scratch.clear(); + bincode::serialize_into(&mut self.scratch, node) .map_err(|e| Error::SerdeError(format!("segmented spill node serialize: {e}")))?; self.nodes.write_all(node.id.as_bytes())?; - self.nodes.write_all(&(blob.len() as u64).to_le_bytes())?; - self.nodes.write_all(&blob)?; + self.nodes + .write_all(&(self.scratch.len() as u64).to_le_bytes())?; + self.nodes.write_all(&self.scratch)?; self.node_count += 1; Ok(()) } @@ -95,15 +100,17 @@ impl SegmentedSpill { /// columnar rows after rematerialize/compact. pub fn append_edge(&mut self, edge: &Edge) -> Result<()> { let canonical = edge.for_columnar_digest(); - let blob = bincode::serialize(&canonical) + self.scratch.clear(); + bincode::serialize_into(&mut self.scratch, &canonical) .map_err(|e| Error::SerdeError(format!("segmented spill edge serialize: {e}")))?; let mut key = [0u8; EDGE_KEY_LEN]; key[..16].copy_from_slice(canonical.from.as_bytes()); key[16..32].copy_from_slice(canonical.to.as_bytes()); key[32] = edge_type_to_u8(canonical.edge_type); self.edges.write_all(&key)?; - self.edges.write_all(&(blob.len() as u64).to_le_bytes())?; - self.edges.write_all(&blob)?; + self.edges + .write_all(&(self.scratch.len() as u64).to_le_bytes())?; + self.edges.write_all(&self.scratch)?; self.edge_count += 1; Ok(()) } diff --git a/crates/rgctl-incremental/src/file_tracker.rs b/crates/rgctl-incremental/src/file_tracker.rs index eccead38..54d3b4bd 100644 --- a/crates/rgctl-incremental/src/file_tracker.rs +++ b/crates/rgctl-incremental/src/file_tracker.rs @@ -96,17 +96,25 @@ impl FileTracker { } /// Index files using a precomputed node→file mapping (cold discover path). + /// + /// When `precomputed_hashes` is provided, keys are absolute path strings (as from + /// extract workers). Missing entries fall back to reading/hashing the file. pub fn index_files_with_mapping( &mut self, files: &[PathBuf], node_mapping: HashMap>, + precomputed_hashes: Option<&HashMap>, ) -> Result<()> { self.metadata.files.clear(); self.metadata.node_mapping = node_mapping; for file in files { let rel = relative_path(&self.repo_root, file)?; - self.metadata.files.insert(rel, Self::hash_file(file)?); + let hash = precomputed_hashes + .and_then(|m| m.get(file.to_string_lossy().as_ref()).cloned()) + .map(Ok) + .unwrap_or_else(|| Self::hash_file(file))?; + self.metadata.files.insert(rel, hash); } self.metadata.indexed_at = chrono_lite_now(); @@ -117,6 +125,22 @@ impl FileTracker { /// Compare current file hashes against stored metadata. pub fn detect_changes(&self, files: &[PathBuf]) -> Result { + // Cold / empty tracker: treat every file as added without hashing. + // Returning a non-empty ChangeSet prevents false snapshot reuse when + // `file_hashes.json` is missing but a stale snapshot file remains. + if self.metadata.files.is_empty() { + let added: Vec = files + .iter() + .filter_map(|path| relative_path(&self.repo_root, path).ok()) + .collect(); + return Ok(ChangeSet { + added, + changed: Vec::new(), + deleted: Vec::new(), + renamed: Vec::new(), + }); + } + let current: HashMap = files .iter() .filter_map(|path| { @@ -448,6 +472,37 @@ mod tests { use std::fs; use tempfile::TempDir; + #[test] + fn test_detect_changes_skips_hash_when_tracker_empty() { + let temp = TempDir::new().unwrap(); + let file = temp.path().join("main.rs"); + fs::write(&file, "fn main() {}\n").unwrap(); + + let tracker = FileTracker::new(temp.path()); + let changes = tracker.detect_changes(&[file]).unwrap(); + assert_eq!(changes.added.len(), 1); + assert!(changes.changed.is_empty()); + assert!(changes.deleted.is_empty()); + } + + #[test] + fn test_index_files_with_precomputed_hashes() { + let temp = TempDir::new().unwrap(); + let file = temp.path().join("lib.rs"); + fs::write(&file, "fn hello() {}\n").unwrap(); + let expected = FileTracker::hash_file(&file).unwrap(); + + let mut precomputed = HashMap::new(); + precomputed.insert(file.to_string_lossy().into_owned(), expected.clone()); + + let mut tracker = FileTracker::new(temp.path()); + tracker + .index_files_with_mapping(&[file.clone()], HashMap::new(), Some(&precomputed)) + .unwrap(); + + assert_eq!(tracker.file_hashes().get("lib.rs"), Some(&expected)); + } + #[test] fn test_file_change_detection() { let temp = TempDir::new().unwrap(); diff --git a/crates/rgctl-pipeline/src/pipeline.rs b/crates/rgctl-pipeline/src/pipeline.rs index 9969c3a0..0f5dbf83 100644 --- a/crates/rgctl-pipeline/src/pipeline.rs +++ b/crates/rgctl-pipeline/src/pipeline.rs @@ -81,6 +81,8 @@ pub struct PipelineStats { pub graph_spill_columnar: Duration, /// Path → node ids collected during extract (skips full mmap scan in save_tracker) pub node_path_mapping: HashMap>, + /// Absolute path → BLAKE3 hex from extract workers (skips re-hash in save_tracker) + pub file_hashes: HashMap, } /// End-to-end repository processing pipeline. @@ -173,6 +175,7 @@ impl ProcessingPipeline { let files_processed = stream_stats.files_processed; let files_failed = stream_stats.extraction_failures.len(); let extract_phases = stream_stats.extract_phases; + let file_hashes = stream_stats.file_hashes; let graph_start = Instant::now(); let index_start = Instant::now(); @@ -221,6 +224,7 @@ impl ProcessingPipeline { graph_pass2: pass2_elapsed, graph_spill_columnar: spill_elapsed, node_path_mapping, + file_hashes, }, digest, )) @@ -279,6 +283,7 @@ impl ProcessingPipeline { let files_processed = stream_stats.files_processed; let files_failed = stream_stats.extraction_failures.len(); let extract_phases = stream_stats.extract_phases; + let file_hashes = stream_stats.file_hashes; let graph_start = Instant::now(); let index_start = Instant::now(); @@ -315,6 +320,7 @@ impl ProcessingPipeline { graph_pass2: pass2_elapsed, graph_spill_columnar: Duration::ZERO, node_path_mapping, + file_hashes, }, )) } diff --git a/crates/rgctl-pipeline/src/stream.rs b/crates/rgctl-pipeline/src/stream.rs index 29d7b98b..90184833 100644 --- a/crates/rgctl-pipeline/src/stream.rs +++ b/crates/rgctl-pipeline/src/stream.rs @@ -6,6 +6,7 @@ use rayon::prelude::*; use rgctl_error::Result; use rgctl_extraction::{ExtractionTail, Extractor, FileExtraction, GraphBuilder}; use rgctl_registry::LanguageRegistry; +use std::collections::HashMap; use std::path::PathBuf; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; @@ -36,6 +37,8 @@ pub struct StreamStats { pub files_processed: usize, pub extraction_failures: Vec, pub extract_phases: ExtractPhaseTimings, + /// Absolute path string → BLAKE3 hex from extract workers (feeds FileTracker). + pub file_hashes: HashMap, } /// Run parallel extractors into a bounded channel while the caller consumes on the main thread. @@ -116,10 +119,16 @@ pub fn stream_into_graph( let mut tails = Vec::with_capacity(file_count); let mut stats = StreamStats::default(); + stats.file_hashes.reserve(file_count); let mut pass1_wall = Duration::ZERO; while let Ok(result) = rx.recv() { match result { Ok(mut extraction) => { + if let Some(hash) = extraction.file_hash.take() { + stats + .file_hashes + .insert(extraction.path.to_string_lossy().into_owned(), hash); + } let pass1_start = Instant::now(); tails.push(extractor.populate_pass1(&mut extraction, builder)?); pass1_wall += pass1_start.elapsed(); diff --git a/src/cli/discover_impl.rs b/src/cli/discover_impl.rs index d435973e..59af6f96 100644 --- a/src/cli/discover_impl.rs +++ b/src/cli/discover_impl.rs @@ -1060,7 +1060,11 @@ pub(crate) fn run_full_analysis( } crate::incremental::group_sorted_node_paths(node_path_pairs) }; - file_tracker.index_files_with_mapping(&files, node_mapping)?; + file_tracker.index_files_with_mapping( + &files, + node_mapping, + Some(&index_stats.file_hashes), + )?; file_tracker.save()?; profile.save_tracker.secs = secs(save_tracker_start.elapsed()); From 5c7f80017a4fdba99395b9a28d502f85555a18f9 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 20:21:46 +0200 Subject: [PATCH 03/18] =?UTF-8?q?Skip=20duplicate=20::=20suffix=20loop=20w?= =?UTF-8?q?hen=20parts.len()<=3D2=20=20=20=20=20=E2=9C=94=20active=5Ftrack?= =?UTF-8?q?er=5Fids=20Vec=20push;=20flush=20in=20end=5Ffile=5Fbatch?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 3 +- crates/rgctl-extraction/src/graph_builder.rs | 61 +++++++++++++++----- 2 files changed, 47 insertions(+), 17 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 1f56033c..9e16a10c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -55,7 +55,8 @@ Rules distilled from linux cold-discover work (`index_extract` / pass-1 / spill | **Prep on workers** | Compute line offsets, BLAKE3 `code_hash`, token bloom in `SymbolPass1Prep` on extract workers | Re-walk source / re-hash on the sequential pass-1 thread | | **Hash once** | Set `FileExtraction.file_hash` from bytes already in memory; thread through `StreamStats` / `PipelineStats` into `FileTracker::index_files_with_mapping` | Re-`fs::read` + BLAKE3 all files in `save_tracker` after extract already hashed them | | **Empty-tracker short-circuit** | When `file_hashes.json` is empty, `detect_changes` marks all paths **added** without hashing (keeps ChangeSet non-empty so a stale snapshot is not reused) | Hash the whole tree twice on cold discover (detect + index) | -| **Normalize / map once per file** | `GraphBuilder::begin_file_batch` → `get_mut` on the active tracker key; accumulate `tracker_mapping` at `commit_node` | `normalize_path_str(...).into_owned()` per symbol; full mmap node scan/sort just to rebuild file→node ids | +| **Normalize / map once per file** | `begin_file_batch` → push into `active_tracker_ids`; flush once in `end_file_batch`; accumulate mapping at commit | Per-symbol `HashMap::get_mut` / `normalize_path_str(...).into_owned()`; full mmap node scan/sort just to rebuild file→node ids | +| **No duplicate suffix index** | Skip `key.split("::")` suffix loop when the key is only `file::name` (already indexed as bare name / QN) | Push the same bare name twice into `symbols_by_suffix` for every C-like symbol | | **Spill alloc reuse** | `SegmentedSpill` scratch `Vec` + `bincode::serialize_into`; keep sort runs at `DEFAULT_SORT_RUN_BYTES` (256 MiB) unless profiling says otherwise | Fresh `bincode::serialize` → new `Vec` per node/edge; shrinking sort runs without a cold gate | | **CodeIndex bodies off by default** | Default discover: no body-storing `CodeIndex` (no multi-GB `code_index.json`); nodes still get `code_hash` from prep | Attach a full CodeIndex on the cold path “for convenience” | diff --git a/crates/rgctl-extraction/src/graph_builder.rs b/crates/rgctl-extraction/src/graph_builder.rs index 792b51e5..612a6e01 100644 --- a/crates/rgctl-extraction/src/graph_builder.rs +++ b/crates/rgctl-extraction/src/graph_builder.rs @@ -64,6 +64,8 @@ pub struct GraphBuilder { tracker_mapping: HashMap>, /// Normalized path for the current pass-1 file (avoids re-normalize + re-hash per symbol). active_tracker_key: Option, + /// Node ids for the active file batch — flushed once in [`Self::end_file_batch`]. + active_tracker_ids: Vec, } #[derive(Debug, Default)] @@ -129,13 +131,24 @@ impl GraphBuilder { /// Pin tracker mapping key for the duration of one file's pass-1 insert. pub fn begin_file_batch(&mut self, path: &Path) { + if self.active_tracker_key.is_some() { + self.end_file_batch(); + } let key = normalize_path_str(&path.to_string_lossy()).into_owned(); self.active_tracker_key = Some(key); + self.active_tracker_ids.clear(); } - /// Clear the active pass-1 file key. + /// Flush active file node ids into `tracker_mapping` (once per file, not per symbol). pub fn end_file_batch(&mut self) { - self.active_tracker_key = None; + if let Some(key) = self.active_tracker_key.take() { + if !self.active_tracker_ids.is_empty() { + let ids = std::mem::take(&mut self.active_tracker_ids); + self.tracker_mapping.entry(key).or_default().extend(ids); + } + } else { + self.active_tracker_ids.clear(); + } } fn record_line_span(&mut self, node: &Node) { @@ -202,13 +215,33 @@ impl GraphBuilder { .or_default() .push(node.id); } - let parts: Vec<&str> = key.split("::").collect(); - for i in 1..parts.len() { - let suffix = parts[i..].join("::"); - self.symbols_by_suffix - .entry(suffix) - .or_default() - .push(node.id); + // `symbol_key` is `file::name` or `file::a::b::…` / `file::Qualified.Name`. + // For unqualified `file::name` the bare-name insert above already covers + // resolution — skip to avoid a duplicate push + alloc per C-like symbol. + match key.matches("::").count() { + 0 => {} + 1 => { + if let Some((_, tail)) = key.split_once("::") { + let duplicate_bare = + node.qualified_name.is_none() && tail == node.name.as_str(); + if !duplicate_bare { + self.symbols_by_suffix + .entry(tail.to_string()) + .or_default() + .push(node.id); + } + } + } + _ => { + let parts: Vec<&str> = key.split("::").collect(); + for i in 1..parts.len() { + let suffix = parts[i..].join("::"); + self.symbols_by_suffix + .entry(suffix) + .or_default() + .push(node.id); + } + } } } @@ -227,13 +260,8 @@ impl GraphBuilder { } fn record_tracker_mapping(&mut self, node: &Node) { - if let Some(key) = self.active_tracker_key.as_ref() { - if let Some(ids) = self.tracker_mapping.get_mut(key) { - ids.push(node.id); - } else { - let key = key.clone(); - self.tracker_mapping.insert(key, vec![node.id]); - } + if self.active_tracker_key.is_some() { + self.active_tracker_ids.push(node.id); return; } let path = node.file_path.as_deref().or_else(|| { @@ -333,6 +361,7 @@ impl GraphBuilder { /// Take the path → node-id mapping accumulated during commit (for FileTracker). pub fn take_tracker_mapping(&mut self) -> HashMap> { + self.end_file_batch(); std::mem::take(&mut self.tracker_mapping) } From bb67262273f5338ae00c1bdff683e214867cde1f Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 20:54:07 +0200 Subject: [PATCH 04/18] tweak plugin parsing Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 2 +- crates/rgctl-lang-c/src/plugin.rs | 22 +++++---------- crates/rgctl-lang-cpp/src/plugin.rs | 21 ++++---------- .../rgctl-plugin-helpers/src/tree_sitter.rs | 28 +++++++++++++------ 4 files changed, 34 insertions(+), 39 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 9e16a10c..5cf6c23f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,7 +56,7 @@ Rules distilled from linux cold-discover work (`index_extract` / pass-1 / spill | **Hash once** | Set `FileExtraction.file_hash` from bytes already in memory; thread through `StreamStats` / `PipelineStats` into `FileTracker::index_files_with_mapping` | Re-`fs::read` + BLAKE3 all files in `save_tracker` after extract already hashed them | | **Empty-tracker short-circuit** | When `file_hashes.json` is empty, `detect_changes` marks all paths **added** without hashing (keeps ChangeSet non-empty so a stale snapshot is not reused) | Hash the whole tree twice on cold discover (detect + index) | | **Normalize / map once per file** | `begin_file_batch` → push into `active_tracker_ids`; flush once in `end_file_batch`; accumulate mapping at commit | Per-symbol `HashMap::get_mut` / `normalize_path_str(...).into_owned()`; full mmap node scan/sort just to rebuild file→node ids | -| **No duplicate suffix index** | Skip `key.split("::")` suffix loop when the key is only `file::name` (already indexed as bare name / QN) | Push the same bare name twice into `symbols_by_suffix` for every C-like symbol | +| **Reuse tree-sitter parsers** | Call `rgctl_plugin_helpers::parse_source` (thread-local `Parser` per Rayon worker); never `Parser::new()` per file on the extract hot path | Fresh `Parser::new` + `set_language` inside `extract_*` / `parse` for every file (linux: ~71k×) | | **Spill alloc reuse** | `SegmentedSpill` scratch `Vec` + `bincode::serialize_into`; keep sort runs at `DEFAULT_SORT_RUN_BYTES` (256 MiB) unless profiling says otherwise | Fresh `bincode::serialize` → new `Vec` per node/edge; shrinking sort runs without a cold gate | | **CodeIndex bodies off by default** | Default discover: no body-storing `CodeIndex` (no multi-GB `code_index.json`); nodes still get `code_hash` from prep | Attach a full CodeIndex on the cold path “for convenience” | diff --git a/crates/rgctl-lang-c/src/plugin.rs b/crates/rgctl-lang-c/src/plugin.rs index ded99add..b929c3df 100644 --- a/crates/rgctl-lang-c/src/plugin.rs +++ b/crates/rgctl-lang-c/src/plugin.rs @@ -1,8 +1,9 @@ //! C language plugin using Tree-sitter. use rgctl_plugin_api::*; +use rgctl_plugin_helpers::parse_source; use std::path::Path; -use tree_sitter::{Node, Parser}; +use tree_sitter::Node; /// File-scoped qualified name: `{file_stem}::{symbol}`. /// @@ -40,30 +41,21 @@ fn include_path_from_node(node: Node, source: &[u8]) -> Result { } /// C language plugin. -pub struct CPlugin { - _parser: Parser, -} +pub struct CPlugin; impl CPlugin { /// Create a new C plugin. pub fn new() -> Result { - let mut parser = Parser::new(); + // Validate grammar at registration; per-file parse reuses a thread-local parser. + let mut parser = tree_sitter::Parser::new(); parser .set_language(&tree_sitter_c::LANGUAGE.into()) .map_err(|e| Error::PluginError(format!("Failed to set C grammar: {e}")))?; - Ok(Self { _parser: parser }) + Ok(Self) } fn parse(&self, file_path: &Path, source: &[u8]) -> Result { - let mut parser = Parser::new(); - parser - .set_language(&tree_sitter_c::LANGUAGE.into()) - .map_err(|e| Error::PluginError(format!("Failed to set C grammar: {e}")))?; - parser.parse(source, None).ok_or_else(|| Error::ParseError { - file: file_path.to_path_buf(), - line: 0, - message: "Failed to parse C source".to_string(), - }) + parse_source(source, file_path, tree_sitter_c::LANGUAGE.into()) } fn extract_function( diff --git a/crates/rgctl-lang-cpp/src/plugin.rs b/crates/rgctl-lang-cpp/src/plugin.rs index 0f27d1c9..685ca876 100644 --- a/crates/rgctl-lang-cpp/src/plugin.rs +++ b/crates/rgctl-lang-cpp/src/plugin.rs @@ -1,34 +1,25 @@ //! C++ language plugin using Tree-sitter. use rgctl_plugin_api::*; +use rgctl_plugin_helpers::parse_source; use std::path::Path; -use tree_sitter::{Node, Parser}; +use tree_sitter::Node; /// C++ language plugin. -pub struct CppPlugin { - _parser: Parser, -} +pub struct CppPlugin; impl CppPlugin { /// Create a new C++ plugin. pub fn new() -> Result { - let mut parser = Parser::new(); + let mut parser = tree_sitter::Parser::new(); parser .set_language(&tree_sitter_cpp::LANGUAGE.into()) .map_err(|e| Error::PluginError(format!("Failed to set C++ grammar: {e}")))?; - Ok(Self { _parser: parser }) + Ok(Self) } fn parse(&self, file_path: &Path, source: &[u8]) -> Result { - let mut parser = Parser::new(); - parser - .set_language(&tree_sitter_cpp::LANGUAGE.into()) - .map_err(|e| Error::PluginError(format!("Failed to set C++ grammar: {e}")))?; - parser.parse(source, None).ok_or_else(|| Error::ParseError { - file: file_path.to_path_buf(), - line: 0, - message: "Failed to parse C++ source".to_string(), - }) + parse_source(source, file_path, tree_sitter_cpp::LANGUAGE.into()) } fn extract_function( diff --git a/crates/rgctl-plugin-helpers/src/tree_sitter.rs b/crates/rgctl-plugin-helpers/src/tree_sitter.rs index ced08161..55f320ab 100644 --- a/crates/rgctl-plugin-helpers/src/tree_sitter.rs +++ b/crates/rgctl-plugin-helpers/src/tree_sitter.rs @@ -257,20 +257,32 @@ fn walk_extract( } /// Parse source with the given grammar and return the tree. +/// +/// Reuses a thread-local [`Parser`] (one per Rayon worker). `Parser::new` + grammar +/// setup is expensive; cold discover parses tens of thousands of files on the same +/// workers, so this matters more than per-file micro-opts in pass-1. pub fn parse_source( source: &[u8], file_path: &Path, grammar: tree_sitter::Language, ) -> Result { use rgctl_plugin_api::Error; - let mut parser = Parser::new(); - parser - .set_language(&grammar) - .map_err(|e| Error::PluginError(format!("Failed to set grammar: {e}")))?; - parser.parse(source, None).ok_or_else(|| Error::ParseError { - file: file_path.to_string_lossy().to_string().into(), - line: 0, - message: "Failed to parse source".to_string(), + use std::cell::RefCell; + + thread_local! { + static PARSER: RefCell = RefCell::new(Parser::new()); + } + + PARSER.with(|cell| { + let mut parser = cell.borrow_mut(); + parser + .set_language(&grammar) + .map_err(|e| Error::PluginError(format!("Failed to set grammar: {e}")))?; + parser.parse(source, None).ok_or_else(|| Error::ParseError { + file: file_path.to_path_buf(), + line: 0, + message: "Failed to parse source".to_string(), + }) }) } From 449c09d9a5d6bcd477e83b2dd7b994b42394b797 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 21:12:28 +0200 Subject: [PATCH 05/18] add easier release handling Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 19 +++- CONTRIBUTING.md | 7 ++ Cargo.toml | 3 +- crates/rgctl-agent-pack-codegen/Cargo.toml | 2 +- crates/rgctl-analysis/Cargo.toml | 2 +- crates/rgctl-ast-coverage/Cargo.toml | 2 +- crates/rgctl-config-formats/Cargo.toml | 2 +- crates/rgctl-core/Cargo.toml | 2 +- crates/rgctl-dashboard/Cargo.toml | 2 +- crates/rgctl-error/Cargo.toml | 2 +- crates/rgctl-export/Cargo.toml | 2 +- crates/rgctl-extraction/Cargo.toml | 2 +- crates/rgctl-gql/Cargo.toml | 2 +- crates/rgctl-graph/Cargo.toml | 2 +- crates/rgctl-incremental/Cargo.toml | 2 +- crates/rgctl-kantra/Cargo.toml | 2 +- crates/rgctl-lang-c/Cargo.toml | 2 +- crates/rgctl-lang-cpp/Cargo.toml | 2 +- crates/rgctl-lang-csharp/Cargo.toml | 2 +- crates/rgctl-lang-erb/Cargo.toml | 2 +- crates/rgctl-lang-go/Cargo.toml | 2 +- crates/rgctl-lang-groovy/Cargo.toml | 2 +- crates/rgctl-lang-java/Cargo.toml | 2 +- crates/rgctl-lang-javascript/Cargo.toml | 2 +- crates/rgctl-lang-kotlin/Cargo.toml | 2 +- crates/rgctl-lang-markdown/Cargo.toml | 2 +- crates/rgctl-lang-php/Cargo.toml | 2 +- crates/rgctl-lang-puppet/Cargo.toml | 2 +- crates/rgctl-lang-python/Cargo.toml | 2 +- crates/rgctl-lang-ruby/Cargo.toml | 2 +- crates/rgctl-lang-runtime/Cargo.toml | 2 +- crates/rgctl-lang-rust/Cargo.toml | 2 +- crates/rgctl-lang-typescript/Cargo.toml | 2 +- crates/rgctl-languages/Cargo.toml | 2 +- crates/rgctl-pipeline/Cargo.toml | 2 +- crates/rgctl-plugin-api/Cargo.toml | 2 +- crates/rgctl-plugin-helpers/Cargo.toml | 2 +- crates/rgctl-project-config/Cargo.toml | 2 +- crates/rgctl-registry/Cargo.toml | 2 +- crates/rgctl-rules/Cargo.toml | 2 +- crates/rgctl-security/Cargo.toml | 2 +- crates/rgctl-semantic/Cargo.toml | 2 +- crates/rgctl-service/Cargo.toml | 2 +- crates/rgctl-wasm/Cargo.toml | 2 +- docs/releasing.md | 35 ++++++- release.toml | 40 ++++++++ rgctl-macros/Cargo.toml | 2 +- scripts/bump-version.sh | 106 +++++++++++++++++++++ 48 files changed, 246 insertions(+), 48 deletions(-) create mode 100644 release.toml create mode 100755 scripts/bump-version.sh diff --git a/AGENTS.md b/AGENTS.md index 5cf6c23f..2710668e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -22,7 +22,7 @@ - **Features:** Default semantic embedder is compiled **vocab**. Do not require ONNX / Python ML unless behind an explicit feature (e.g. `semantic-onnx` / code-daemon + Git LFS). - **OpenSpec language work:** Still cite [openspec/changes/_shared/starting-context.md](openspec/changes/_shared/starting-context.md) (pointer here); follow the sections below. - **Grammar bumps:** When you bump a tree-sitter grammar pin, update that language’s `*-ast-coverage.json` (and add the language to `rgctl-ast-coverage::bundled_specs` for new languages). Unit tests hard-fail the same drift; `cargo check -p rgctl-languages` warns (`RGCTL_AST_COVERAGE_STRICT=1` fails). The website `/docs/languages/` pages are generated from those JSON files — do not maintain parallel tables under `docs/languages/`. - +- **Releases:** Follow **Releases** below (and [docs/releasing.md](docs/releasing.md)). Do not hand-edit dozens of crate `version =` lines. --- ## Context & architecture @@ -158,10 +158,27 @@ Baselines and notes: [docs/internal/profile.md](docs/internal/profile.md#snapsho | [CONTRIBUTING.md](CONTRIBUTING.md) | Setup, tests, PR norms | | [docs/contributor-checklist.md](docs/contributor-checklist.md) | Language / feature checklist | | [docs/guides/semantic-search.md](docs/guides/semantic-search.md) | Embedders (if touching semantic) | +| [docs/releasing.md](docs/releasing.md) | Version bump + GitHub Release tags | | [openspec/changes/_shared/starting-context.md](openspec/changes/_shared/starting-context.md) | OpenSpec pointer (canonical policy is this file) | --- +## Releases + +When asked to cut or bump a release, use the lockstep tooling — full detail: [docs/releasing.md](docs/releasing.md). + +| Rule | Detail | +|------|--------| +| **One version** | SSOT is `[workspace.package] version` in root `Cargo.toml`. Crates use `version.workspace = true`. Do **not** sed/`version =` across every crate by hand. | +| **Bump TOMLs only** | `./scripts/bump-version.sh patch` (or `minor` / `major` / `X.Y.Z`). Syncs workspace version, `[workspace.dependencies]` path pins, and README release links. | +| **Bump + tag + push** | `cargo release patch --workspace` (dry-run), then `--execute` when the user wants commit/tag/push. Config: [`release.toml`](release.toml) (`shared-version`, `publish = false`, tag `v{{version}}`). Needs a **clean** git tree. | +| **Tools** | `cargo install cargo-edit cargo-release --locked` if missing. | +| **GitHub Release** | Pushing `v*` runs [`.github/workflows/release.yml`](.github/workflows/release.yml) (binaries). Add `docs/releases/vX.Y.Z.md` for curated notes. | +| **No crates.io** | `publish = false` — do not `cargo publish` unless the user explicitly asks to enable it. | +| **Commits / tags / push** | Only when the user explicitly requests them (same standing rule as other git ops). Prefer preparing the bump + release notes and stopping for the user to commit/sign/tag if they GPG-sign locally. | + +--- + ## Build and day-to-day commands ```bash diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f2293ef0..7c96a6e5 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -89,6 +89,13 @@ Full map: [docs/Code_structure.md](docs/Code_structure.md) --- +## Releasing (version bump) + +Lockstep workspace version via `[workspace.package]` + `version.workspace = true`. +See **[docs/releasing.md](docs/releasing.md)** for `./scripts/bump-version.sh` and `cargo release`. + +--- + ## Adding or improving a language / feature Use the hub checklist for path choice, test matrices, and pre-PR commands: diff --git a/Cargo.toml b/Cargo.toml index 98ca8b39..770b0f98 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -49,6 +49,7 @@ resolver = "2" [workspace.package] edition = "2024" rust-version = "1.99" +version = "0.4.17" [workspace.dependencies] rgctl-plugin-api = { path = "crates/rgctl-plugin-api", version = "0.4.17" } @@ -124,7 +125,7 @@ expect_used = "warn" [package] name = "rgctl" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true authors = ["rgctl Contributors"] diff --git a/crates/rgctl-agent-pack-codegen/Cargo.toml b/crates/rgctl-agent-pack-codegen/Cargo.toml index 73effe03..5383fbed 100644 --- a/crates/rgctl-agent-pack-codegen/Cargo.toml +++ b/crates/rgctl-agent-pack-codegen/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-agent-pack-codegen" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true publish = false diff --git a/crates/rgctl-analysis/Cargo.toml b/crates/rgctl-analysis/Cargo.toml index 607bdcef..c29e9849 100644 --- a/crates/rgctl-analysis/Cargo.toml +++ b/crates/rgctl-analysis/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-analysis" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Graph analysis algorithms for rgctl" diff --git a/crates/rgctl-ast-coverage/Cargo.toml b/crates/rgctl-ast-coverage/Cargo.toml index 3a546a18..8fdf4ab5 100644 --- a/crates/rgctl-ast-coverage/Cargo.toml +++ b/crates/rgctl-ast-coverage/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-ast-coverage" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "AST coverage manifest checks vs pinned tree-sitter grammars" diff --git a/crates/rgctl-config-formats/Cargo.toml b/crates/rgctl-config-formats/Cargo.toml index a2528f90..7761dd0e 100644 --- a/crates/rgctl-config-formats/Cargo.toml +++ b/crates/rgctl-config-formats/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-config-formats" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Configuration format plugins for rgctl" diff --git a/crates/rgctl-core/Cargo.toml b/crates/rgctl-core/Cargo.toml index 8adc5bbe..0ce0b401 100644 --- a/crates/rgctl-core/Cargo.toml +++ b/crates/rgctl-core/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-core" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Facade crate for rgctl library consumers" diff --git a/crates/rgctl-dashboard/Cargo.toml b/crates/rgctl-dashboard/Cargo.toml index e09edfc8..451e8a1f 100644 --- a/crates/rgctl-dashboard/Cargo.toml +++ b/crates/rgctl-dashboard/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-dashboard" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Static dashboard bundle export for rgctl" diff --git a/crates/rgctl-error/Cargo.toml b/crates/rgctl-error/Cargo.toml index 0632535c..6a1cc6eb 100644 --- a/crates/rgctl-error/Cargo.toml +++ b/crates/rgctl-error/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-error" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Error types for rgctl" diff --git a/crates/rgctl-export/Cargo.toml b/crates/rgctl-export/Cargo.toml index 2a0620d3..19cc2a95 100644 --- a/crates/rgctl-export/Cargo.toml +++ b/crates/rgctl-export/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-export" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Graph export and visualization for rgctl" diff --git a/crates/rgctl-extraction/Cargo.toml b/crates/rgctl-extraction/Cargo.toml index 839725c0..cbf1cee7 100644 --- a/crates/rgctl-extraction/Cargo.toml +++ b/crates/rgctl-extraction/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-extraction" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Code extraction layer for rgctl" diff --git a/crates/rgctl-gql/Cargo.toml b/crates/rgctl-gql/Cargo.toml index 23b13b84..fb4d52c7 100644 --- a/crates/rgctl-gql/Cargo.toml +++ b/crates/rgctl-gql/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-gql" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Graph query language for rgctl" diff --git a/crates/rgctl-graph/Cargo.toml b/crates/rgctl-graph/Cargo.toml index 7cffbd3e..f0b536e9 100644 --- a/crates/rgctl-graph/Cargo.toml +++ b/crates/rgctl-graph/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-graph" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Graph storage and query layer for rgctl" diff --git a/crates/rgctl-incremental/Cargo.toml b/crates/rgctl-incremental/Cargo.toml index 3570b4a3..7f2a5299 100644 --- a/crates/rgctl-incremental/Cargo.toml +++ b/crates/rgctl-incremental/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-incremental" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Incremental graph updates for rgctl" diff --git a/crates/rgctl-kantra/Cargo.toml b/crates/rgctl-kantra/Cargo.toml index daac467a..27cd8329 100644 --- a/crates/rgctl-kantra/Cargo.toml +++ b/crates/rgctl-kantra/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-kantra" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Native Konveyor Kantra rule evaluator for rgctl" diff --git a/crates/rgctl-lang-c/Cargo.toml b/crates/rgctl-lang-c/Cargo.toml index f69fc931..6dd196b2 100644 --- a/crates/rgctl-lang-c/Cargo.toml +++ b/crates/rgctl-lang-c/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-c" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: c" diff --git a/crates/rgctl-lang-cpp/Cargo.toml b/crates/rgctl-lang-cpp/Cargo.toml index 12170f46..80213ce7 100644 --- a/crates/rgctl-lang-cpp/Cargo.toml +++ b/crates/rgctl-lang-cpp/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-cpp" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: cpp" diff --git a/crates/rgctl-lang-csharp/Cargo.toml b/crates/rgctl-lang-csharp/Cargo.toml index 88bd5b9e..cb9e6279 100644 --- a/crates/rgctl-lang-csharp/Cargo.toml +++ b/crates/rgctl-lang-csharp/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-csharp" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: csharp" diff --git a/crates/rgctl-lang-erb/Cargo.toml b/crates/rgctl-lang-erb/Cargo.toml index 30fac20d..95e1d1e7 100644 --- a/crates/rgctl-lang-erb/Cargo.toml +++ b/crates/rgctl-lang-erb/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-erb" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: ERB embedded templates (tree-sitter-embedded-template)" diff --git a/crates/rgctl-lang-go/Cargo.toml b/crates/rgctl-lang-go/Cargo.toml index 5a1c6db9..86d7e2fc 100644 --- a/crates/rgctl-lang-go/Cargo.toml +++ b/crates/rgctl-lang-go/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-go" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: go" diff --git a/crates/rgctl-lang-groovy/Cargo.toml b/crates/rgctl-lang-groovy/Cargo.toml index c7490136..6ec8f2c2 100644 --- a/crates/rgctl-lang-groovy/Cargo.toml +++ b/crates/rgctl-lang-groovy/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-groovy" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: groovy (tree-sitter-groovy)" diff --git a/crates/rgctl-lang-java/Cargo.toml b/crates/rgctl-lang-java/Cargo.toml index 5c9d9f3f..4da88b29 100644 --- a/crates/rgctl-lang-java/Cargo.toml +++ b/crates/rgctl-lang-java/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-java" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: java" diff --git a/crates/rgctl-lang-javascript/Cargo.toml b/crates/rgctl-lang-javascript/Cargo.toml index a96ad48e..43c1d5d0 100644 --- a/crates/rgctl-lang-javascript/Cargo.toml +++ b/crates/rgctl-lang-javascript/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-javascript" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: javascript" diff --git a/crates/rgctl-lang-kotlin/Cargo.toml b/crates/rgctl-lang-kotlin/Cargo.toml index 1c63190b..54307890 100644 --- a/crates/rgctl-lang-kotlin/Cargo.toml +++ b/crates/rgctl-lang-kotlin/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-kotlin" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: kotlin" diff --git a/crates/rgctl-lang-markdown/Cargo.toml b/crates/rgctl-lang-markdown/Cargo.toml index 78276a4c..8739977a 100644 --- a/crates/rgctl-lang-markdown/Cargo.toml +++ b/crates/rgctl-lang-markdown/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-markdown" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: markdown (context / documentation graph)" diff --git a/crates/rgctl-lang-php/Cargo.toml b/crates/rgctl-lang-php/Cargo.toml index 4f9cb524..dcc6f69e 100644 --- a/crates/rgctl-lang-php/Cargo.toml +++ b/crates/rgctl-lang-php/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-php" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: php" diff --git a/crates/rgctl-lang-puppet/Cargo.toml b/crates/rgctl-lang-puppet/Cargo.toml index 899a2775..c3b7d649 100644 --- a/crates/rgctl-lang-puppet/Cargo.toml +++ b/crates/rgctl-lang-puppet/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-puppet" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: puppet (tree-sitter-puppet)" diff --git a/crates/rgctl-lang-python/Cargo.toml b/crates/rgctl-lang-python/Cargo.toml index ed844029..e56d3769 100644 --- a/crates/rgctl-lang-python/Cargo.toml +++ b/crates/rgctl-lang-python/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-python" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: python" diff --git a/crates/rgctl-lang-ruby/Cargo.toml b/crates/rgctl-lang-ruby/Cargo.toml index eb405489..d5f91a46 100644 --- a/crates/rgctl-lang-ruby/Cargo.toml +++ b/crates/rgctl-lang-ruby/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-ruby" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: ruby" diff --git a/crates/rgctl-lang-runtime/Cargo.toml b/crates/rgctl-lang-runtime/Cargo.toml index 9f1c60fb..7868f79d 100644 --- a/crates/rgctl-lang-runtime/Cargo.toml +++ b/crates/rgctl-lang-runtime/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-runtime" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Generic tree-sitter and regex language plugins for rgctl" diff --git a/crates/rgctl-lang-rust/Cargo.toml b/crates/rgctl-lang-rust/Cargo.toml index 072218f2..ee4e0e71 100644 --- a/crates/rgctl-lang-rust/Cargo.toml +++ b/crates/rgctl-lang-rust/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-rust" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: rust" diff --git a/crates/rgctl-lang-typescript/Cargo.toml b/crates/rgctl-lang-typescript/Cargo.toml index b1f52823..7ed04224 100644 --- a/crates/rgctl-lang-typescript/Cargo.toml +++ b/crates/rgctl-lang-typescript/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-lang-typescript" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "rgctl language plugin: typescript" diff --git a/crates/rgctl-languages/Cargo.toml b/crates/rgctl-languages/Cargo.toml index 47d1fbc2..b577c4c0 100644 --- a/crates/rgctl-languages/Cargo.toml +++ b/crates/rgctl-languages/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-languages" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Built-in Tier 1 language plugins for rgctl" diff --git a/crates/rgctl-pipeline/Cargo.toml b/crates/rgctl-pipeline/Cargo.toml index c957a7f6..8b24bbeb 100644 --- a/crates/rgctl-pipeline/Cargo.toml +++ b/crates/rgctl-pipeline/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-pipeline" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Parallel processing pipeline for rgctl" diff --git a/crates/rgctl-plugin-api/Cargo.toml b/crates/rgctl-plugin-api/Cargo.toml index d813fcb3..06a8ab60 100644 --- a/crates/rgctl-plugin-api/Cargo.toml +++ b/crates/rgctl-plugin-api/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-plugin-api" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Stable plugin API for rgctl language and config format plugins" diff --git a/crates/rgctl-plugin-helpers/Cargo.toml b/crates/rgctl-plugin-helpers/Cargo.toml index 603e6a85..53cef220 100644 --- a/crates/rgctl-plugin-helpers/Cargo.toml +++ b/crates/rgctl-plugin-helpers/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-plugin-helpers" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Tree-sitter extraction helpers for rgctl plugins" diff --git a/crates/rgctl-project-config/Cargo.toml b/crates/rgctl-project-config/Cargo.toml index 3a5568a0..351a06c0 100644 --- a/crates/rgctl-project-config/Cargo.toml +++ b/crates/rgctl-project-config/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-project-config" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Project configuration analysis for rgctl" diff --git a/crates/rgctl-registry/Cargo.toml b/crates/rgctl-registry/Cargo.toml index 58c75334..cc6c678e 100644 --- a/crates/rgctl-registry/Cargo.toml +++ b/crates/rgctl-registry/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-registry" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Language plugin registry for rgctl" diff --git a/crates/rgctl-rules/Cargo.toml b/crates/rgctl-rules/Cargo.toml index 6fb11941..2547f146 100644 --- a/crates/rgctl-rules/Cargo.toml +++ b/crates/rgctl-rules/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-rules" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Rule engine for rgctl" diff --git a/crates/rgctl-security/Cargo.toml b/crates/rgctl-security/Cargo.toml index 7a341675..8b3048c7 100644 --- a/crates/rgctl-security/Cargo.toml +++ b/crates/rgctl-security/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-security" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Security analysis for rgctl" diff --git a/crates/rgctl-semantic/Cargo.toml b/crates/rgctl-semantic/Cargo.toml index 3f0998b1..e8d2ba67 100644 --- a/crates/rgctl-semantic/Cargo.toml +++ b/crates/rgctl-semantic/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-semantic" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Semantic analysis and IDL generation for rgctl" diff --git a/crates/rgctl-service/Cargo.toml b/crates/rgctl-service/Cargo.toml index 08b1c5f7..830aceec 100644 --- a/crates/rgctl-service/Cargo.toml +++ b/crates/rgctl-service/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-service" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Shared command execution for rgctl CLI, HTTP, and MCP" diff --git a/crates/rgctl-wasm/Cargo.toml b/crates/rgctl-wasm/Cargo.toml index 90bfcecf..34d62232 100644 --- a/crates/rgctl-wasm/Cargo.toml +++ b/crates/rgctl-wasm/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-wasm" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "WASM graph engine for rgctl dashboard" diff --git a/docs/releasing.md b/docs/releasing.md index 52397508..c2b773f5 100644 --- a/docs/releasing.md +++ b/docs/releasing.md @@ -6,11 +6,38 @@ How maintainers publish versioned binaries and GitHub Releases. ## Version numbers -- **Crate / CLI version** lives in root [`Cargo.toml`](../Cargo.toml) (`[package].version`). -- **Workspace crates** share the same version in their `Cargo.toml` files and `[workspace.dependencies]` pins. -- **Git tags** use a `v` prefix: `v0.2.0` (not `0.2.0` alone). +- **Single source of truth:** `[workspace.package] version` in root [`Cargo.toml`](../Cargo.toml). +- **All workspace crates** use `version.workspace = true` (including the root `rgctl` package). +- **`[workspace.dependencies]`** path pins still carry an explicit `version = "…"` (crates.io metadata); keep them in lockstep with the workspace version. +- **Git tags** use a `v` prefix: `v0.4.18` (not `0.4.18` alone). -Bump all workspace versions together before tagging. +### Automating the bump + +Install once: + +```bash +cargo install cargo-edit --locked # cargo set-version +cargo install cargo-release --locked # cargo release +``` + +**Bump TOMLs only** (no commit/tag) — updates workspace package version, path pins, and README release link: + +```bash +./scripts/bump-version.sh patch # or: minor | major | 0.4.18 +git diff # review, commit yourself, then tag (below) +``` + +**Bump + commit + tag + push** (see [`release.toml`](../release.toml); dry-run by default). +Requires a **clean git tree** (commit the inheritance / docs changes first): + +```bash +cargo release patch --workspace # preview +cargo release patch --workspace --execute # commit, tag vX.Y.Z, push +``` + +`publish = false`: we do **not** `cargo publish` to crates.io yet. Pushing the `v*` tag still triggers the GitHub Release binary workflow below. + +Write `docs/releases/vX.Y.Z.md` before or right after the bump so CI can attach curated notes. --- diff --git a/release.toml b/release.toml new file mode 100644 index 00000000..b9ab78e3 --- /dev/null +++ b/release.toml @@ -0,0 +1,40 @@ +# cargo-release config — lockstep workspace version + git tag for GH Release workflow. +# Docs: https://github.com/crate-ci/cargo-release +# +# Dry-run (default): cargo release patch --workspace +# Execute: cargo release patch --workspace --execute +# Explicit version: cargo release 0.4.18 --workspace --execute +# +# Binary releases are built by `.github/workflows/release.yml` on `v*` tags. +# Crates.io publish is off until we intentionally enable it. + +# All publishable members share one version (matches `[workspace.package].version`). +shared-version = true + +# When bumping, upgrade in-workspace path dependency `version =` requirements. +dependent-version = "upgrade" + +# One commit for the whole workspace bump. +consolidate-commits = true + +# Tag shape matches existing GH Actions trigger (`on.push.tags: v*`). +# Shared-version workspaces tag once as `v{{version}}`. +tag-name = "v{{version}}" +tag-prefix = "" + +# Do not cargo-publish on release (GH binary artifacts only for now). +publish = false + +# Push tags so the Release workflow runs (requires --execute). +push = true +push-remote = "origin" + +# Branches allowed for --execute (globs ok). +allow-branch = ["*"] + +# Exclude dogfood / fixture crates (not workspace members, listed for safety). +# exclude = [] + +# README latest-release link: updated by ./scripts/bump-version.sh (not +# cargo-release replacements — those apply per-crate and break on missing +# crates/*/README.md). diff --git a/rgctl-macros/Cargo.toml b/rgctl-macros/Cargo.toml index 853d6e85..22e36a8c 100644 --- a/rgctl-macros/Cargo.toml +++ b/rgctl-macros/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "rgctl-macros" -version = "0.4.17" +version.workspace = true edition.workspace = true rust-version.workspace = true description = "Procedural macros for rgctl language plugins" diff --git a/scripts/bump-version.sh b/scripts/bump-version.sh new file mode 100755 index 00000000..d7eeaecf --- /dev/null +++ b/scripts/bump-version.sh @@ -0,0 +1,106 @@ +#!/usr/bin/env bash +# Bump the lockstep workspace version (option 2: cargo set-version). +# +# Usage: +# ./scripts/bump-version.sh patch # 0.4.17 -> 0.4.18 +# ./scripts/bump-version.sh minor # 0.4.17 -> 0.5.0 +# ./scripts/bump-version.sh 0.4.18 # set exact version +# +# Updates: +# - [workspace.package] version (crates use version.workspace = true) +# - path crate version= pins under [workspace.dependencies] +# +# Does NOT commit or tag. For bump+commit+tag use: +# cargo release --workspace --execute +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +cd "$ROOT" + +if ! cargo set-version -h >/dev/null 2>&1; then + echo "error: cargo-edit (cargo set-version) is required" >&2 + echo " cargo install cargo-edit --locked" >&2 + exit 1 +fi + +ARG="${1:-}" +if [[ -z "$ARG" ]]; then + echo "usage: $0 " >&2 + exit 1 +fi + +read_workspace_version() { + python3 - <<'PY' +import re +from pathlib import Path +text = Path("Cargo.toml").read_text() +m = re.search(r'(?ms)^\[workspace\.package\].*?^version = "([^"]+)"', text) +if not m: + raise SystemExit("could not find [workspace.package] version") +print(m.group(1)) +PY +} + +prev="$(read_workspace_version)" + +if [[ "$ARG" =~ ^[0-9]+\.[0-9]+\.[0-9]+([.-].*)?$ ]]; then + cargo set-version --workspace "$ARG" +else + case "$ARG" in + patch|minor|major) + cargo set-version --bump "$ARG" --workspace + ;; + *) + echo "error: unknown bump '$ARG' (use patch|minor|major|X.Y.Z)" >&2 + exit 1 + ;; + esac +fi + +new="$(read_workspace_version)" + +# Keep [workspace.dependencies] path pins in lockstep (needed for publish metadata). +python3 - "$prev" "$new" <<'PY' +import re, sys +from pathlib import Path +prev, new = sys.argv[1], sys.argv[2] +path = Path("Cargo.toml") +text = path.read_text() + +def repl_block(match: re.Match) -> str: + return match.group(0).replace(f'version = "{prev}"', f'version = "{new}"') + +text2, n = re.subn( + r'(?ms)^\[workspace\.dependencies\]\n.*?(?=^\[|\Z)', + repl_block, + text, + count=1, +) +if n != 1: + print("warning: could not locate [workspace.dependencies] block to sync", file=sys.stderr) +else: + path.write_text(text2) + print(f"synced [workspace.dependencies] path versions {prev} -> {new}") +PY + +echo "workspace version: $prev -> $new" + +# Keep README latest-release link in sync when present. +if [[ -f README.md ]]; then + python3 - "$prev" "$new" <<'PY' +import sys +from pathlib import Path +prev, new = sys.argv[1], sys.argv[2] +path = Path("README.md") +text = path.read_text() +updated = text.replace(f"v{prev}", f"v{new}").replace( + f"docs/releases/v{prev}.md", f"docs/releases/v{new}.md" +) +if updated != text: + path.write_text(updated) + print(f"updated README.md release links {prev} -> {new}") +PY +fi + +echo "next: review git diff, then either commit manually or:" +echo " cargo release $new --workspace --execute # commit + tag v$new + push" From 17651d811cc1bb5ba8331914a10acfbce1afb735 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:32:51 +0200 Subject: [PATCH 06/18] add limits for running in constrained environments e.g. containers etc Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 1 + crates/rgctl-graph/src/lib.rs | 3 +- crates/rgctl-graph/src/segmented_spill.rs | 27 +++- docs/guides/discovering-and-indexing.md | 8 + docs/internal/integration-tests.md | 1 + docs/user-guide.md | 11 ++ scripts/run-container-with-limits-smoke.sh | 58 +++++++ src/cli/discover.rs | 10 +- src/cli/discover_impl.rs | 45 +++++- src/cli/discover_limits.rs | 169 +++++++++++++++++++++ src/cli/mod.rs | 14 ++ src/cli/pipeline_session.rs | 9 +- tests/Containerfile | 50 ++++++ tests/container_with_limits_smoke.sh | 82 ++++++++++ 14 files changed, 479 insertions(+), 9 deletions(-) create mode 100755 scripts/run-container-with-limits-smoke.sh create mode 100644 src/cli/discover_limits.rs create mode 100644 tests/Containerfile create mode 100755 tests/container_with_limits_smoke.sh diff --git a/AGENTS.md b/AGENTS.md index 2710668e..013ec7d1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -19,6 +19,7 @@ - **Ingest hot path:** Follow **Ingest hot-path practices** below (no per-symbol heap strings, hash/prep once on workers, spill scratch reuse, tracker mapping without re-scan). - **Typed graph:** Respect `EdgeType` / node kinds; do not invent ad-hoc string edges for hot paths. - **Artifacts:** Session data lives in `{repo}/.rgctl/`. Warm caches invalidate wall-time claims. +- **Constrained discover (opt-in):** Prefer `rgctl discover . --with-limits max-mem-mb=4096,threads=1` (or `RGCTL_WITH_LIMITS`) in containers / cgroups — do **not** change default desktop discover for memory. Soft RSS tripwire at ~95%; smaller spill sort-runs and stream channel when `max-mem-mb` is set. Container smoke: `./scripts/run-container-with-limits-smoke.sh` (`tests/Containerfile`, mounts `example/linux`). - **Features:** Default semantic embedder is compiled **vocab**. Do not require ONNX / Python ML unless behind an explicit feature (e.g. `semantic-onnx` / code-daemon + Git LFS). - **OpenSpec language work:** Still cite [openspec/changes/_shared/starting-context.md](openspec/changes/_shared/starting-context.md) (pointer here); follow the sections below. - **Grammar bumps:** When you bump a tree-sitter grammar pin, update that language’s `*-ast-coverage.json` (and add the language to `rgctl-ast-coverage::bundled_specs` for new languages). Unit tests hard-fail the same drift; `cargo check -p rgctl-languages` warns (`RGCTL_AST_COVERAGE_STRICT=1` fails). The website `/docs/languages/` pages are generated from those JSON files — do not maintain parallel tables under `docs/languages/`. diff --git a/crates/rgctl-graph/src/lib.rs b/crates/rgctl-graph/src/lib.rs index 41b300b3..083fad79 100644 --- a/crates/rgctl-graph/src/lib.rs +++ b/crates/rgctl-graph/src/lib.rs @@ -57,7 +57,8 @@ pub use graph_compactor::{ pub use migration::{migrate_snapshot, migrate_v1_to_v2}; pub use schema::{AccessType, CallType, GRAPH_SCHEMA_VERSION, GraphParameter, SharedStr}; pub use segmented_spill::{ - DEFAULT_SORT_RUN_BYTES, FinishedSpill, SegmentedSpill, write_columnar_from_spill, + DEFAULT_SORT_RUN_BYTES, FinishedSpill, SegmentedSpill, set_sort_run_bytes_override, + write_columnar_from_spill, }; pub use snapshot::{ MmappedGraphSnapshot, PreparedGraphSnapshot, PreparedIndexes, SNAPSHOT_FILE, SnapshotNodeStore, diff --git a/crates/rgctl-graph/src/segmented_spill.rs b/crates/rgctl-graph/src/segmented_spill.rs index 187fb51e..e6ac46f0 100644 --- a/crates/rgctl-graph/src/segmented_spill.rs +++ b/crates/rgctl-graph/src/segmented_spill.rs @@ -20,6 +20,7 @@ use std::collections::{BinaryHeap, HashMap}; use std::fs::{self, File}; use std::io::{BufReader, BufWriter, Read, Write}; use std::path::{Path, PathBuf}; +use std::sync::atomic::{AtomicUsize, Ordering as AtomicOrdering}; use uuid::Uuid; /// Default run size for external merge-sort (~256 MiB of record payload). @@ -28,6 +29,24 @@ use uuid::Uuid; /// segs are hundreds of MiB). Peak RSS during sort grows by one run buffer. pub const DEFAULT_SORT_RUN_BYTES: usize = 256 * 1024 * 1024; +/// Process-wide sort-run override (`0` = use [`DEFAULT_SORT_RUN_BYTES`]). +/// Set by discover `--with-limits` for constrained containers; leave unset on desktop. +static SORT_RUN_BYTES_OVERRIDE: AtomicUsize = AtomicUsize::new(0); + +/// Cap external-sort run buffers for this process (discover `--with-limits`). +pub fn set_sort_run_bytes_override(bytes: Option) { + SORT_RUN_BYTES_OVERRIDE.store(bytes.unwrap_or(0), AtomicOrdering::Relaxed); +} + +fn effective_sort_run_bytes() -> usize { + let o = SORT_RUN_BYTES_OVERRIDE.load(AtomicOrdering::Relaxed); + if o == 0 { + DEFAULT_SORT_RUN_BYTES + } else { + o + } +} + const NODE_KEY_LEN: usize = 16; const EDGE_KEY_LEN: usize = 16 + 16 + 8; // from + to + type/pad @@ -169,7 +188,7 @@ pub fn materialize_sorted_graph(spill: &FinishedSpill) -> Result<(Vec, Vec &nodes_unsorted, &nodes_sorted, NODE_KEY_LEN, - DEFAULT_SORT_RUN_BYTES, + effective_sort_run_bytes(), spill.node_count, ) }); @@ -178,7 +197,7 @@ pub fn materialize_sorted_graph(spill: &FinishedSpill) -> Result<(Vec, Vec &edges_unsorted, &edges_sorted, EDGE_KEY_LEN, - DEFAULT_SORT_RUN_BYTES, + effective_sort_run_bytes(), spill.edge_count, ) }); @@ -244,7 +263,7 @@ pub fn write_columnar_from_spill(spill: FinishedSpill, path: &Path) -> Result Result/dev/null 2>&1; then + ENGINE=podman + elif command -v docker >/dev/null 2>&1; then + ENGINE=docker + else + echo "ERROR: need podman or docker" >&2 + exit 1 + fi +fi + +if [[ ! -d "$LINUX" ]]; then + echo "ERROR: linux corpus missing at $LINUX" >&2 + echo " Run: ./scripts/fetch-profile-repos.sh # or set RGCTL_LINUX_REPO" >&2 + exit 1 +fi + +if [[ "$ENGINE" == podman ]]; then + if ! podman info >/dev/null 2>&1; then + echo "==> starting podman machine" + podman machine start + # Wait until the API is actually reachable (start can race ahead of the socket). + for _ in $(seq 1 60); do + if podman info >/dev/null 2>&1; then + break + fi + sleep 1 + done + podman info >/dev/null || { + echo "ERROR: podman machine started but API is unreachable" >&2 + exit 1 + } + fi +fi + +echo "==> building $IMAGE (engine=$ENGINE)" +"$ENGINE" build -f "$ROOT/tests/Containerfile" -t "$IMAGE" "$ROOT" + +echo "==> running smoke (mount $LINUX -> /corpus, memory=$MEMORY, scope=$SMOKE_SCOPE)" +"$ENGINE" run --rm \ + --memory="$MEMORY" \ + -e "SMOKE_SCOPE=$SMOKE_SCOPE" \ + -e "LIMITS_SPEC=${LIMITS_SPEC:-max-mem-mb=4096,threads=1}" \ + -v "$LINUX:/corpus:Z" \ + "$IMAGE" + +echo "==> container with-limits smoke OK" diff --git a/src/cli/discover.rs b/src/cli/discover.rs index 5b831136..8851918f 100644 --- a/src/cli/discover.rs +++ b/src/cli/discover.rs @@ -48,6 +48,8 @@ pub struct DiscoverArgs { pub kantra_index_only: bool, /// Staged full pipeline (`--full`). pub full: bool, + /// Resource limits (`--with-limits SPEC` / `RGCTL_WITH_LIMITS`). + pub with_limits: Option, /// Preset strategy for `--export-migration-hints` (default: hybrid_default). pub migration_preset: String, /// Roadmap row order: `scheduled` (deps) or `priority` (score rank). @@ -84,6 +86,7 @@ pub fn run(ctx: &CliContext, args: DiscoverArgs) -> Result<()> { &args.kantra_rules, &args.kantra_catalog, )?; + let limits = super::discover_limits::DiscoverLimits::from_cli(args.with_limits.as_deref())?; let path = resolve_session_root(ctx, args.path.as_deref()); if let Some(files) = &args.files { @@ -91,7 +94,11 @@ pub fn run(ctx: &CliContext, args: DiscoverArgs) -> Result<()> { } if args.full { - run_full_pipeline(ctx, &path, FullPipelineArgs::from_discover(&args))?; + run_full_pipeline( + ctx, + &path, + FullPipelineArgs::from_discover(&args, limits.clone()), + )?; return Ok(()); } @@ -122,6 +129,7 @@ pub fn run(ctx: &CliContext, args: DiscoverArgs) -> Result<()> { force_reindex: false, emit_cli_summary: true, artifact_root: args.artifact_root.as_deref(), + limits, }, )?; Ok(()) diff --git a/src/cli/discover_impl.rs b/src/cli/discover_impl.rs index 59af6f96..87f5bcf3 100644 --- a/src/cli/discover_impl.rs +++ b/src/cli/discover_impl.rs @@ -6,6 +6,7 @@ use rgctl_pipeline::with_large_pool; use super::discover_cfg::{ CfgAnalysisOptions, FileSourceCache, preload_file_sources, run_cfg_analysis_batch, }; +use super::discover_limits::{check_memory_budget, DiscoverLimits}; use super::discover_output::build_discover_response; use super::stage_profile::{DiscoverStageReport, secs}; use crate::analysis::graph_utils::PetGraphView; @@ -59,6 +60,8 @@ pub(crate) struct AnalysisOptions<'a> { pub emit_cli_summary: bool, /// Persist snapshots under this root (defaults to the scanned `path`). pub artifact_root: Option<&'a Path>, + /// Opt-in resource limits (`--with-limits`). + pub limits: Option, } /// Result of one `run_full_analysis` pass. @@ -99,6 +102,7 @@ pub(crate) fn run_full_analysis( force_reindex, emit_cli_summary, artifact_root, + limits, } = opts; let verbose = ctx.verbose; @@ -112,6 +116,28 @@ pub(crate) fn run_full_analysis( profile.cfg_enabled = run_cfg_pass; profile.security_enabled = with_security; + if let Some(ref lim) = limits { + lim.log_active(); + if let Some(bytes) = lim.sort_run_bytes() { + rgctl_graph::set_sort_run_bytes_override(Some(bytes)); + } + if let Some(n) = lim.threads { + // Help any code paths that still use the global Rayon pool. + if std::env::var_os("RAYON_NUM_THREADS").is_none() { + // SAFETY: single-threaded init before worker pools start. + unsafe { std::env::set_var("RAYON_NUM_THREADS", n.to_string()) }; + } + } + } + // Clear spill sort-run override even on early return / error. + struct ClearSortRunOverride; + impl Drop for ClearSortRunOverride { + fn drop(&mut self) { + rgctl_graph::set_sort_run_bytes_override(None); + } + } + let _clear_sort_run = ClearSortRunOverride; + let root = Path::new(path); // Source tree scanned for files; `.rgctl/` artifacts live under `store`. let store = artifact_root.unwrap_or(root); @@ -163,6 +189,11 @@ pub(crate) fn run_full_analysis( discovery, show_progress: human_output, materialize_fields, + thread_count: limits.as_ref().and_then(|l| l.threads), + stream_channel_capacity: limits + .as_ref() + .and_then(|l| l.stream_channel_capacity()) + .unwrap_or_else(|| PipelineConfig::default().stream_channel_capacity), ..PipelineConfig::default() }, ); @@ -344,6 +375,9 @@ pub(crate) fn run_full_analysis( profile.functions = functions.len(); // Seal ingest phase: absolute peak stays; analysis phase peak resets to current RSS. profile.ingest_peak_rss_mb = mem_monitor.seal_phase().unwrap_or(0.0); + if let Some(limit) = limits.as_ref().and_then(|l| l.max_mem_mb) { + check_memory_budget(profile.ingest_peak_rss_mb, limit)?; + } debug!( ingest_peak_mb = profile.ingest_peak_rss_mb, "{}", @@ -564,7 +598,11 @@ pub(crate) fn run_full_analysis( } let file_sources: Option = if with_ast_skeleton || with_cfg { - Some(preload_file_sources(&functions, root, None)) + Some(preload_file_sources( + &functions, + root, + limits.as_ref().and_then(|l| l.threads), + )) } else { None }; @@ -575,7 +613,7 @@ pub(crate) fn run_full_analysis( root, CfgAnalysisOptions { verbose, - thread_count: None, + thread_count: limits.as_ref().and_then(|l| l.threads), enable_taint: with_taint, dfg_loops: with_dfg_loops, }, @@ -1204,6 +1242,9 @@ pub(crate) fn run_full_analysis( let analysis_size = std::fs::metadata(&analysis_path)?.len() as f64 / (1024.0 * 1024.0); mem_monitor.stop_periodic_sampling(); profile.analysis_peak_rss_mb = mem_monitor.seal_phase().unwrap_or(0.0); + if let Some(limit) = limits.as_ref().and_then(|l| l.max_mem_mb) { + check_memory_budget(profile.analysis_peak_rss_mb.max(profile.ingest_peak_rss_mb), limit)?; + } let snapshot = mem_monitor.snapshot()?; profile.wall_total.secs = secs(run_start.elapsed()); profile.peak_rss_mb = snapshot.peak_mb; diff --git a/src/cli/discover_limits.rs b/src/cli/discover_limits.rs new file mode 100644 index 00000000..ddf7e599 --- /dev/null +++ b/src/cli/discover_limits.rs @@ -0,0 +1,169 @@ +//! Discover resource limits (`--with-limits` / `RGCTL_WITH_LIMITS`). + +use anyhow::{bail, Context, Result}; +use tracing::info; + +/// Opt-in discover constraints. Default discover leaves these unset. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct DiscoverLimits { + /// Soft RSS budget in mebibytes (`max-mem-mb`). + pub max_mem_mb: Option, + /// Cap Rayon / pipeline worker threads (`threads`). + pub threads: Option, +} + +impl DiscoverLimits { + /// Parse `max-mem-mb=4096,threads=1` (comma-separated `key=value`). + /// + /// Empty string ⇒ enabled with no overrides (placeholder for future cgroup defaults). + pub fn parse(spec: &str) -> Result { + let mut out = Self::default(); + let trimmed = spec.trim(); + if trimmed.is_empty() { + return Ok(out); + } + for part in trimmed.split(',') { + let part = part.trim(); + if part.is_empty() { + continue; + } + let (key, value) = part + .split_once('=') + .with_context(|| format!("invalid --with-limits entry `{part}` (want key=value)"))?; + let key = key.trim().to_ascii_lowercase().replace('_', "-"); + let value = value.trim(); + match key.as_str() { + "max-mem-mb" | "max-mem" | "memory-mb" => { + let mb: u64 = value + .parse() + .with_context(|| format!("invalid max-mem-mb `{value}`"))?; + if mb == 0 { + bail!("max-mem-mb must be > 0"); + } + out.max_mem_mb = Some(mb); + } + "threads" | "thread" | "max-thread" | "max-threads" => { + let n: usize = value + .parse() + .with_context(|| format!("invalid threads `{value}`"))?; + if n == 0 { + bail!("threads must be > 0"); + } + out.threads = Some(n); + } + other => bail!( + "unknown --with-limits key `{other}` (supported: max-mem-mb, threads)" + ), + } + } + Ok(out) + } + + /// Resolve CLI `--with-limits [SPEC]` or `RGCTL_WITH_LIMITS` when the flag is omitted. + pub fn from_cli(spec: Option<&str>) -> Result> { + let resolved = match spec { + Some(s) => Some(s.to_owned()), + None => std::env::var("RGCTL_WITH_LIMITS").ok(), + }; + match resolved.as_deref() { + None => Ok(None), + Some(s) => Ok(Some(Self::parse(s)?)), + } + } + + /// Whether any constraint was requested (flag present with values). + #[allow(dead_code)] + pub fn is_active(&self) -> bool { + self.max_mem_mb.is_some() || self.threads.is_some() + } + + /// Stream channel capacity when limits are active (else leave pipeline default). + pub fn stream_channel_capacity(&self) -> Option { + let budget = self.max_mem_mb? as usize; + // ~10 MB assumed per in-flight extraction; clamp to [32, 1024]. + Some((budget / 10).clamp(32, 1024)) + } + + /// Spill external-sort run size when limits are active. + pub fn sort_run_bytes(&self) -> Option { + match self.max_mem_mb? { + mb if mb <= 2048 => Some(32 * 1024 * 1024), + mb if mb <= 8192 => Some(64 * 1024 * 1024), + _ => None, // keep default 256 MiB on large budgets + } + } + + pub fn log_active(&self) { + info!( + max_mem_mb = ?self.max_mem_mb, + threads = ?self.threads, + stream_channel = ?self.stream_channel_capacity(), + sort_run_mb = self.sort_run_bytes().map(|b| b / (1024 * 1024)), + "discover --with-limits active" + ); + } +} + +/// Soft memory budget check (warn ≥90%, error ≥95%). +pub fn check_memory_budget(peak_mb: f64, limit_mb: u64) -> Result<()> { + let limit = limit_mb as f64; + if peak_mb >= limit * 0.95 { + bail!( + "Memory limit of {limit_mb} MB approached (peak RSS: {peak_mb:.0} MB). \ + Discovery aborted to avoid container OOM. \ + Raise max-mem-mb, set threads lower, or filter with -l." + ); + } + if peak_mb >= limit * 0.90 { + tracing::warn!( + peak_mb, + limit_mb, + "RSS within 90% of --with-limits max-mem-mb" + ); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_kv_pair() { + let l = DiscoverLimits::parse("max-mem-mb=4096,threads=1").unwrap(); + assert_eq!(l.max_mem_mb, Some(4096)); + assert_eq!(l.threads, Some(1)); + assert_eq!(l.stream_channel_capacity(), Some(409)); + assert_eq!(l.sort_run_bytes(), Some(64 * 1024 * 1024)); + } + + #[test] + fn parse_empty() { + let l = DiscoverLimits::parse("").unwrap(); + assert!(!l.is_active()); + } + + #[test] + fn reject_unknown_key() { + assert!(DiscoverLimits::parse("foo=1").is_err()); + } + + #[test] + fn budget_tripwire() { + // 90% of 4096 ≈ 3686; 95% ≈ 3891 + assert!(check_memory_budget(3700.0, 4096).is_ok()); + assert!(check_memory_budget(4000.0, 4096).is_err()); + } + + #[test] + fn from_cli_reads_env_when_flag_absent() { + // SAFETY: test process; key is test-scoped. + unsafe { std::env::set_var("RGCTL_WITH_LIMITS", "threads=2") }; + let l = DiscoverLimits::from_cli(None).unwrap().unwrap(); + assert_eq!(l.threads, Some(2)); + // Flag wins over env. + let l = DiscoverLimits::from_cli(Some("threads=1")).unwrap().unwrap(); + assert_eq!(l.threads, Some(1)); + unsafe { std::env::remove_var("RGCTL_WITH_LIMITS") }; + } +} diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 78918201..9fdc8d9d 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -14,6 +14,7 @@ mod diff; mod discover; mod discover_cfg; mod discover_impl; +mod discover_limits; mod kantra_discover; pub mod discover_output; mod export; @@ -177,6 +178,17 @@ pub enum Commands { #[arg(long = "full")] full: bool, + /// Constrain resource use for containers / small machines. + /// Spec: `max-mem-mb=4096,threads=1` (comma-separated). Flag alone enables limits + /// mode with no overrides. Env: `RGCTL_WITH_LIMITS`. + #[arg( + long = "with-limits", + value_name = "SPEC", + num_args = 0..=1, + default_missing_value = "" + )] + with_limits: Option, + /// Strategy preset for migration plan export. #[arg( long = "migration-preset", @@ -784,6 +796,7 @@ impl Cli { kantra_target, kantra_index_only, mut full, + with_limits, migration_preset, migration_order, files, @@ -814,6 +827,7 @@ impl Cli { kantra_target, kantra_index_only, full, + with_limits, migration_preset, migration_order, artifact_root: None, diff --git a/src/cli/pipeline_session.rs b/src/cli/pipeline_session.rs index 6cdd8a63..c0bf4e44 100644 --- a/src/cli/pipeline_session.rs +++ b/src/cli/pipeline_session.rs @@ -33,10 +33,14 @@ pub struct FullPipelineArgs { pub migration_preset: String, pub migration_order: String, pub artifact_root: Option, + pub limits: Option, } impl FullPipelineArgs { - pub fn from_discover(args: &DiscoverArgs) -> Self { + pub fn from_discover( + args: &DiscoverArgs, + limits: Option, + ) -> Self { Self { languages: args.languages.clone(), exclude: args.exclude.clone(), @@ -49,6 +53,7 @@ impl FullPipelineArgs { migration_preset: args.migration_preset.clone(), migration_order: args.migration_order.clone(), artifact_root: args.artifact_root.clone(), + limits, } } @@ -65,6 +70,7 @@ impl FullPipelineArgs { migration_preset: "hybrid_default".into(), migration_order: "scheduled".into(), artifact_root: None, + limits: None, } } } @@ -286,6 +292,7 @@ fn analysis_opts<'a>( force_reindex, emit_cli_summary: false, artifact_root: extras.artifact_root.as_deref(), + limits: extras.limits.clone(), } } diff --git a/tests/Containerfile b/tests/Containerfile new file mode 100644 index 00000000..21f7323c --- /dev/null +++ b/tests/Containerfile @@ -0,0 +1,50 @@ +# syntax=docker/dockerfile:1 +# Build release `rgctl` and smoke-test `--with-limits` against a mounted Linux tree. +# +# Host (from repo root): +# ./scripts/run-container-with-limits-smoke.sh +# +# Manual: +# podman build -f tests/Containerfile -t rgctl-with-limits-smoke . +# podman run --rm --memory=4g \ +# -v "$PWD/example/linux:/corpus:Z" \ +# -e SMOKE_SCOPE=scripts \ +# rgctl-with-limits-smoke + +# Deb bookworm + rustup so we can pin rustc >= workspace rust-version (1.99). +FROM docker.io/library/debian:bookworm-slim AS builder + +ARG RUST_VERSION=1.99.0 +ENV CARGO_HOME=/usr/local/cargo \ + RUSTUP_HOME=/usr/local/rustup \ + PATH=/usr/local/cargo/bin:$PATH + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + ca-certificates curl build-essential pkg-config libssl-dev git \ + && rm -rf /var/lib/apt/lists/* \ + && curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --default-toolchain "${RUST_VERSION}" --profile minimal \ + && rustc --version + +WORKDIR /src +# Context is filtered by repo-root `.dockerignore` (no example/, target/, …). +COPY . . +# Vocab-only binary: skip `semantic-onnx` (ort prebuilts need newer glibc than bookworm). +RUN cargo build --release --bin rgctl --no-default-features \ + && strip target/release/rgctl + +FROM docker.io/library/debian:bookworm-slim AS runtime + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ca-certificates \ + && rm -rf /var/lib/apt/lists/* + +COPY --from=builder /src/target/release/rgctl /usr/local/bin/rgctl +COPY --chmod=755 tests/container_with_limits_smoke.sh /usr/local/bin/container_with_limits_smoke.sh + +ENV RUST_LOG=info,profile=info +WORKDIR /corpus +VOLUME ["/corpus"] + +ENTRYPOINT ["/usr/local/bin/container_with_limits_smoke.sh"] diff --git a/tests/container_with_limits_smoke.sh b/tests/container_with_limits_smoke.sh new file mode 100755 index 00000000..dca16302 --- /dev/null +++ b/tests/container_with_limits_smoke.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +# Smoke-test `rgctl discover --with-limits` inside a memory-constrained container. +# Expects the Linux corpus mounted at /corpus (see tests/Containerfile). +set -euo pipefail + +CORPUS="${CORPUS:-/corpus}" +# Scoped path under the mounted tree for a fast smoke (override with SMOKE_SCOPE=. for full tree). +SMOKE_SCOPE="${SMOKE_SCOPE:-scripts}" +LIMITS_SPEC="${LIMITS_SPEC:-max-mem-mb=4096,threads=1}" +RGCTL_BIN="${RGCTL_BIN:-rgctl}" + +log() { printf '==> %s\n' "$*"; } +die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; } + +[[ -x "$(command -v "$RGCTL_BIN")" ]] || die "rgctl binary not found: $RGCTL_BIN" +[[ -d "$CORPUS" ]] || die "corpus mount missing at $CORPUS (pass -v example/linux:/corpus)" + +TARGET="$CORPUS" +if [[ "$SMOKE_SCOPE" != "." && -n "$SMOKE_SCOPE" ]]; then + TARGET="$CORPUS/$SMOKE_SCOPE" +fi +[[ -d "$TARGET" ]] || die "smoke scope not found: $TARGET" + +log "rgctl $($RGCTL_BIN --version 2>/dev/null || true)" +log "corpus=$CORPUS scope=$SMOKE_SCOPE limits=$LIMITS_SPEC" + +# --- CLI surface --- +HELP="$($RGCTL_BIN discover --help)" +echo "$HELP" | grep -q -- '--with-limits' \ + || die "discover --help missing --with-limits" + +# Invalid spec must fail fast (no discover). +if $RGCTL_BIN discover "$TARGET" --with-limits 'not-a-spec' >/tmp/bad-limits.out 2>/tmp/bad-limits.err; then + die "expected invalid --with-limits to fail" +fi +grep -qi 'with-limits\|invalid\|unknown\|want key' /tmp/bad-limits.err /tmp/bad-limits.out \ + || die "invalid --with-limits error message unclear" + +# --- Constrained discover --- +# Artifacts under the scanned tree (or CORPUS if we cd there with positional .). +cd "$TARGET" +rm -rf .rgctl +# Also clear parent corpus .rgctl if a prior full run left one (mount is shared). +rm -rf "$CORPUS/.rgctl" + +LOG=/tmp/rgctl-with-limits-smoke.log +set +e +NO_COLOR=1 RUST_LOG="${RUST_LOG:-info}" \ + "$RGCTL_BIN" discover . --with-limits "$LIMITS_SPEC" -l c -v \ + >"$LOG" 2>&1 +RC=$? +set -e +# Strip ANSI in case the TTY still colors tracing output. +CLEAN=$(sed 's/\x1b\[[0-9;]*m//g' "$LOG") +printf '%s\n' "$CLEAN" + +[[ $RC -eq 0 ]] || die "discover --with-limits failed (exit $RC)" + +printf '%s\n' "$CLEAN" | grep -q 'discover --with-limits active' \ + || die "missing 'discover --with-limits active' log line" +printf '%s\n' "$CLEAN" | grep -Eq 'max_mem_mb[= ].*Some\([0-9]+\)|max_mem_mb=Some\(' \ + || die "limits log missing max_mem_mb" +printf '%s\n' "$CLEAN" | grep -Eq 'threads[= ].*Some\([0-9]+\)|threads=Some\(' \ + || die "limits log missing threads" + +[[ -f .rgctl/graph.snapshot.bin ]] || die "missing .rgctl/graph.snapshot.bin after discover" + +# Env-form limits (no flag) should also activate. +rm -rf .rgctl +LOG2=/tmp/rgctl-with-limits-env.log +set +e +NO_COLOR=1 RUST_LOG=info RGCTL_WITH_LIMITS="$LIMITS_SPEC" \ + "$RGCTL_BIN" discover . -l c -v \ + >"$LOG2" 2>&1 +RC2=$? +set -e +CLEAN2=$(sed 's/\x1b\[[0-9;]*m//g' "$LOG2") +[[ $RC2 -eq 0 ]] || die "discover via RGCTL_WITH_LIMITS failed (exit $RC2)" +printf '%s\n' "$CLEAN2" | grep -q 'discover --with-limits active' \ + || die "RGCTL_WITH_LIMITS did not activate limits" + +log "PASS: --with-limits smoke on $TARGET" From ab9182ad3f7ce9796eaea9857a62991076d60933 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:05:48 +0200 Subject: [PATCH 07/18] commands over gql, save tokens, faster response times Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- .dockerignore | 20 + .gitignore | 1 + crates/rgctl-graph/src/columnar_snapshot.rs | 14 + crates/rgctl-graph/src/lib.rs | 8 + crates/rgctl-graph/src/structured_query.rs | 1405 +++++++++++++++++ .../groovy-ast-coverage.json | 4 +- crates/rgctl-lang-groovy/src/plugin.rs | 138 +- crates/rgctl-lang-typescript/src/plugin.rs | 165 ++ docs/agents/USER_AGENTS_TEMPLATE.md | 21 +- docs/internal/container-memory-management.md | 208 +++ docs/internal/gql-vs-graph-evaluation.md | 5 + ...ti-language-structured-query-evaluation.md | 5 + docs/internal/structured-query-design.md | 5 + scripts/run-structured-query-field-tests.py | 686 ++++++++ skills/rgctl/SKILL.md | 26 +- .../rgctl/references/command-encyclopedia.md | 32 +- src/cli/mod.rs | 535 ++++++- src/cli/structured_query.rs | 216 +++ 18 files changed, 3472 insertions(+), 22 deletions(-) create mode 100644 .dockerignore create mode 100644 crates/rgctl-graph/src/structured_query.rs create mode 100644 docs/internal/container-memory-management.md create mode 100644 docs/internal/gql-vs-graph-evaluation.md create mode 100644 docs/internal/multi-language-structured-query-evaluation.md create mode 100644 docs/internal/structured-query-design.md create mode 100755 scripts/run-structured-query-field-tests.py create mode 100644 src/cli/structured_query.rs diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 00000000..6d855d1c --- /dev/null +++ b/.dockerignore @@ -0,0 +1,20 @@ +# Keep the image build context lean. Runtime corpus is mounted, not copied. +.git +target +example +**/node_modules +dashboard/dist/node_modules +dashboard/node_modules +**/.rgctl +**/.rgctl-diff +docs +openspec +rgctl-tests +fuzz +.cursor +.claude +.idea +.vscode +*.log +*.profraw +*.profdata diff --git a/.gitignore b/.gitignore index f6a8fc4c..9cd55cac 100644 --- a/.gitignore +++ b/.gitignore @@ -3,6 +3,7 @@ .claude/ .opencode/ .agent +.reports # OpenSpec change proposals (local only — do not commit) openspec/ .scratch/ diff --git a/crates/rgctl-graph/src/columnar_snapshot.rs b/crates/rgctl-graph/src/columnar_snapshot.rs index 9e245a75..44b36355 100644 --- a/crates/rgctl-graph/src/columnar_snapshot.rs +++ b/crates/rgctl-graph/src/columnar_snapshot.rs @@ -263,12 +263,26 @@ impl ColumnarGraphMmap { &self.digest_hex } + /// Shared Arc to lazy-parsed name/type indexes (no HashMap clone). + /// + /// Prefer this for structured query hot paths. [`Self::name_index`] / [`Self::type_index`] + /// clone the maps and are intended for hydration / one-shot tooling only. + pub fn indexes_shared( + &self, + ) -> Result>, HashMap>)>> { + self.parsed_indexes() + } + /// Name → node id index (lazy-parsed from mmap on first access). + /// + /// Clones the map — prefer [`Self::indexes_shared`] on query hot paths. pub fn name_index(&self) -> Result>> { Ok(self.parsed_indexes()?.0.clone()) } /// Node type → node id index (lazy-parsed from mmap on first access). + /// + /// Clones the map — prefer [`Self::indexes_shared`] on query hot paths. pub fn type_index(&self) -> Result>> { Ok(self.parsed_indexes()?.1.clone()) } diff --git a/crates/rgctl-graph/src/lib.rs b/crates/rgctl-graph/src/lib.rs index 083fad79..854dbbce 100644 --- a/crates/rgctl-graph/src/lib.rs +++ b/crates/rgctl-graph/src/lib.rs @@ -35,6 +35,8 @@ pub mod segmented_spill; pub mod snapshot; /// Structural diff between two columnar snapshots. pub mod snapshot_diff; +/// Deterministic mmap structured query (`find` / `callers` / `relations` / `inventory`). +pub mod structured_query; /// Stable cross-snapshot node identity. pub mod stable_key; pub mod structural_sketch; @@ -67,6 +69,12 @@ pub use snapshot_diff::{ DiffSink, DiffStats, EdgeDeltaEvent, EdgeDeltaKind, NodeDeltaEvent, NodeDeltaKind, NoopDiffSink, SnapshotPair, VecDiffSink, diff_snapshots, }; +pub use structured_query::{ + ALL_EDGE_TYPES, ALL_NODE_TYPES, CallNeighborsResult, EntityRow, EdgeRow, FindResult, + InventoryBy, InventoryCount, InventoryResult, QueryFilters, RelationDirection, RelationsResult, + STRUCTURED_QUERY_SCHEMA_VERSION, ScopeMode, StructuredQuery, glob_match, parse_edge_type, + parse_node_type, +}; pub use stable_key::{ MmapNodeKey, NodeRowRef, StableNodeKey, NAMESPACE_RGCTL, deterministic_node_id, extension_digest, node_row_ref, node_scope_path_at, stable_key_from_facets, diff --git a/crates/rgctl-graph/src/structured_query.rs b/crates/rgctl-graph/src/structured_query.rs new file mode 100644 index 00000000..6242870d --- /dev/null +++ b/crates/rgctl-graph/src/structured_query.rs @@ -0,0 +1,1405 @@ +//! Structured query primitives over mmap snapshots (no `MemoryBackend` hydrate). +//! +//! # Index complexity (honest) +//! - Exact name: O(1) hash via shared Arc `name_index` ([`ColumnarGraphMmap::indexes_shared`]). +//! - Prefix / suffix / contains name patterns: O(|name keys|) scan of the HashMap keys. +//! - `--scope` on `qualified_name`: per-candidate node filter (no dedicated prefix index yet). +//! - Seedless edge scan: sequential typed-edge walk via [`SnapshotNodeStore::for_each_edge`]. +//! - Seeded multi-hop: builds a typed adjacency map once per invocation (O(E_type)). + +use crate::schema::{EdgeType, Node, NodeType}; +use crate::snapshot::SnapshotNodeStore; +use rgctl_error::{Error, Result}; +use serde::Serialize; +use std::collections::{HashMap, HashSet, VecDeque}; +use std::sync::Arc; +use uuid::Uuid; + +/// JSON schema version for structured-query envelopes. +pub const STRUCTURED_QUERY_SCHEMA_VERSION: u32 = 1; + +/// All [`NodeType`] variants for inventory zero-count emission. +pub const ALL_NODE_TYPES: &[NodeType] = &[ + NodeType::Function, + NodeType::Class, + NodeType::Struct, + NodeType::Enum, + NodeType::Interface, + NodeType::Annotation, + NodeType::Module, + NodeType::Variable, + NodeType::File, + NodeType::ConfigKey, + NodeType::TypeAlias, + NodeType::Macro, + NodeType::Import, + NodeType::Table, + NodeType::Dependency, + NodeType::Job, + NodeType::BuildStep, + NodeType::AnsiblePlaybook, + NodeType::AnsiblePlay, + NodeType::AnsibleTask, + NodeType::AnsibleRole, + NodeType::AnsibleHandler, + NodeType::AnsibleVariable, + NodeType::AnsibleTemplate, + NodeType::ChefCookbook, + NodeType::ChefRecipe, + NodeType::ChefResource, + NodeType::ChefAttribute, + NodeType::ChefTemplate, + NodeType::ChefCustomResource, + NodeType::PuppetModule, + NodeType::PuppetClass, + NodeType::PuppetDefinedType, + NodeType::PuppetResource, + NodeType::PuppetVariable, + NodeType::PuppetFact, + NodeType::PuppetNode, + NodeType::KantraRuleset, + NodeType::KantraRule, +]; + +/// All [`EdgeType`] variants except [`EdgeType::Unknown`] for inventory zeros. +pub const ALL_EDGE_TYPES: &[EdgeType] = &[ + EdgeType::Calls, + EdgeType::Contains, + EdgeType::Uses, + EdgeType::Implements, + EdgeType::Extends, + EdgeType::References, + EdgeType::Instantiates, + EdgeType::Modifies, + EdgeType::UsesConfig, + EdgeType::DefinedIn, + EdgeType::DependsOn, + EdgeType::IncludesRole, + EdgeType::DependsOnRole, + EdgeType::ExecutesTask, + EdgeType::NotifiesHandler, + EdgeType::IncludesPlaybook, + EdgeType::RendersTemplate, + EdgeType::DependsOnCookbook, + EdgeType::IncludesRecipe, + EdgeType::DeclaresResource, + EdgeType::UsesTemplate, + EdgeType::DefinesAttribute, + EdgeType::NotifiesResource, + EdgeType::DependsOnModule, + EdgeType::IncludesClass, + EdgeType::InheritsClass, + EdgeType::RequiresResource, + EdgeType::UsesFact, + EdgeType::AnnotatedWith, + EdgeType::Permits, + EdgeType::Violates, +]; + +/// How `--scope` filters `qualified_name` prefixes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum ScopeMode { + /// Keep nodes/endpoints whose FQN starts with the prefix. + #[default] + Inside, + /// Keep nodes/endpoints whose FQN does not start with the prefix. + Outside, + /// For edges: keep pairs that straddle the prefix boundary. + Crossing, +} + +impl ScopeMode { + /// Parse CLI token. + pub fn parse(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "inside" | "in" => Ok(Self::Inside), + "outside" | "out" => Ok(Self::Outside), + "crossing" | "cross" => Ok(Self::Crossing), + other => Err(Error::InvalidQuery(format!( + "unknown scope-mode '{other}' (expected inside|outside|crossing)" + ))), + } + } +} + +/// Lean entity row for JSON. +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +pub struct EntityRow { + /// Bare name + pub name: String, + /// Fully qualified name when present + #[serde(skip_serializing_if = "Option::is_none")] + pub qualified_name: Option, + /// Node type (lowercase CLI form) + #[serde(rename = "type")] + pub node_type: String, + /// Source file path + #[serde(skip_serializing_if = "Option::is_none")] + pub file: Option, + /// Definition line + #[serde(skip_serializing_if = "Option::is_none")] + pub line: Option, + /// Node UUID (string) + pub id: String, +} + +/// Keyed edge row (direction-stable; never positional). +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +pub struct EdgeRow { + /// Edge origin + pub source: EntityRow, + /// Edge type (lowercase CLI form) + pub edge: String, + /// Traversal direction for this emission + pub direction: String, + /// Edge destination + pub target: EntityRow, + /// Hop distance from seed (1 for seedless scans) + #[serde(skip_serializing_if = "Option::is_none")] + pub hops: Option, +} + +/// Find / list result envelope. +#[derive(Debug, Clone, Serialize)] +pub struct FindResult { + /// Schema version + pub schema_version: u32, + /// Emitted row count + pub returned: usize, + /// Matches before limit when known + pub total: usize, + /// Entity rows (empty when count_only) + #[serde(skip_serializing_if = "Vec::is_empty")] + pub entities: Vec, +} + +/// Callers / callees envelope. +#[derive(Debug, Clone, Serialize)] +pub struct CallNeighborsResult { + /// Schema version + pub schema_version: u32, + /// Resolved seed + pub target: EntityRow, + /// Neighbor rows + pub neighbors: Vec, + /// Hop labels aligned with neighbors + pub hops: Vec, + /// Emitted neighbor count + pub returned: usize, + /// Total neighbors before limit + pub total: usize, + /// Requested depth + pub depth: usize, + /// `callers` or `callees` + pub direction: String, +} + +/// Relations envelope. +#[derive(Debug, Clone, Serialize)] +pub struct RelationsResult { + /// Schema version + pub schema_version: u32, + /// Optional resolved seed + #[serde(skip_serializing_if = "Option::is_none")] + pub target: Option, + /// Keyed edges + pub edges: Vec, + /// Emitted count + pub returned: usize, + /// Total before limit + pub total: usize, +} + +/// Inventory envelope. +#[derive(Debug, Clone, Serialize)] +pub struct InventoryResult { + /// Schema version + pub schema_version: u32, + /// Aggregation dimension + pub by: String, + /// Counts (includes zeros for type/edge) + pub counts: Vec, +} + +/// One inventory bucket. +#[derive(Debug, Clone, Serialize)] +pub struct InventoryCount { + /// Bucket key (type/edge/lang/file/community) + pub key: String, + /// Count (may be zero) + pub count: usize, +} + +/// Filters shared by find / resolve. +#[derive(Debug, Clone, Default)] +pub struct QueryFilters { + /// Node type filter + pub node_type: Option, + /// Path glob (`*` wildcards; `?` single char) + pub file_glob: Option, + /// Language id (property `language` or file extension heuristic) + pub lang: Option, + /// qualified_name prefix + pub scope: Option, + /// Scope mode + pub scope_mode: ScopeMode, + /// Enclosing class/type name filter + pub class: Option, + /// Max rows (None = unbounded) + pub limit: Option, + /// Count only + pub count_only: bool, + /// Exact name match (no glob) + pub exact: bool, +} + +/// Session over an open snapshot store. +pub struct StructuredQuery<'a> { + store: &'a SnapshotNodeStore, +} + +impl<'a> StructuredQuery<'a> { + /// Borrow a snapshot store (must already be open; no hydrate). + pub fn new(store: &'a SnapshotNodeStore) -> Self { + Self { store } + } + + /// Documented serve offload contract: call heavy methods from `spawn_blocking`. + pub fn serve_offload_note() -> &'static str { + "When invoking StructuredQuery from rgctl serve, run find/callers/relations/inventory on spawn_blocking" + } + + fn project(node: &Node) -> EntityRow { + EntityRow { + name: node.name.to_string(), + qualified_name: node.qualified_name.as_ref().map(|s| s.to_string()), + node_type: node_type_cli(node.node_type), + file: node.file_path.as_ref().map(|s| s.to_string()), + line: node.start_line, + id: node.id.to_string(), + } + } + + fn matches_scope(node: &Node, scope: Option<&str>, mode: ScopeMode) -> bool { + let Some(prefix) = scope else { + return mode != ScopeMode::Crossing; + }; + let inside = node_in_scope(node, prefix); + match mode { + ScopeMode::Inside => inside, + ScopeMode::Outside => !inside, + // Node-level crossing is meaningless; treat as inside for node filters. + ScopeMode::Crossing => inside, + } + } + + fn edge_scope_ok(src: &Node, dst: &Node, scope: Option<&str>, mode: ScopeMode) -> bool { + let Some(prefix) = scope else { + return true; + }; + let s = node_in_scope(src, prefix); + let d = node_in_scope(dst, prefix); + match mode { + // Migration-friendly: keep edges whose *source* is in-scope (annotations / + // base types are often external). Use Crossing for true boundary edges. + ScopeMode::Inside => s, + ScopeMode::Outside => !s, + ScopeMode::Crossing => s != d, + } + } + + fn matches_file(node: &Node, glob: Option<&str>) -> bool { + let Some(raw) = glob else { + return true; + }; + let pat = normalize_file_glob(raw); + let path = node.file_path.as_deref().unwrap_or(""); + if glob_match(&pat, path) { + return true; + } + // Basename fallback when pattern has no path separator. + if !raw.contains('/') && !raw.contains('\\') { + let base = path.rsplit(['/', '\\']).next().unwrap_or(path); + return glob_match(&pat, base) || glob_match(raw, base); + } + false + } + + fn matches_lang(node: &Node, lang: Option<&str>) -> bool { + let Some(want) = lang else { + return true; + }; + let want = want.to_ascii_lowercase(); + if let Some(v) = node.get_property("language") { + return v.eq_ignore_ascii_case(&want); + } + let ext = node + .file_path + .as_deref() + .and_then(|p| p.rsplit('.').next()) + .unwrap_or(""); + lang_from_ext(ext).eq_ignore_ascii_case(&want) + } + + fn matches_class(node: &Node, class: Option<&str>) -> bool { + let Some(c) = class else { + return true; + }; + if let Some(qn) = node.qualified_name.as_deref() { + if qn.contains(c) { + return true; + } + } + if let Some(v) = node.get_property("class") { + return v == c; + } + if let Some(v) = node.get_property("enclosing_type") { + return v == c; + } + false + } + + fn node_passes(&self, node: &Node, f: &QueryFilters) -> bool { + if let Some(t) = f.node_type + && node.node_type != t + { + return false; + } + if !Self::matches_file(node, f.file_glob.as_deref()) { + return false; + } + if !Self::matches_lang(node, f.lang.as_deref()) { + return false; + } + if !Self::matches_scope(node, f.scope.as_deref(), f.scope_mode) { + return false; + } + if !Self::matches_class(node, f.class.as_deref()) { + return false; + } + true + } + + fn indexes( + &self, + ) -> Result>, HashMap>)>>> { + Ok(self + .store + .columnar() + .map(|c| c.indexes_shared()) + .transpose()?) + } + + /// Resolve a unique symbol or return ambiguity / not-found errors. + pub fn resolve_symbol( + &self, + symbol: &str, + filters: &QueryFilters, + ) -> Result { + let mut matches = self.lookup_name_candidates(symbol, filters.exact)?; + matches.retain(|n| self.node_passes(n, filters)); + match matches.len() { + 0 => Err(Error::NodeNotFound(symbol.to_string())), + 1 => Ok(matches.remove(0)), + n => Err(Error::AmbiguousSymbol { + name: symbol.to_string(), + count: n, + }), + } + } + + fn lookup_name_candidates(&self, symbol: &str, exact: bool) -> Result> { + if let Ok(uuid) = Uuid::parse_str(symbol) { + if let Some(n) = self.store.get_node(uuid)? { + return Ok(vec![n]); + } + } + if exact || !is_glob_pattern(symbol) { + if let Some(indexes) = self.indexes()? { + if let Some(ids) = indexes.0.get(symbol) { + let mut out = Vec::with_capacity(ids.len()); + for id in ids { + if let Some(n) = self.store.get_node(*id)? { + out.push(n); + } + } + if !out.is_empty() { + return Ok(out); + } + } + } else { + return self.store.find_nodes_by_name(symbol); + } + // Fall through: also try qualified_name exact via type scan is expensive; return empty. + return Ok(Vec::new()); + } + // Glob over name keys (O(|keys|)). + let Some(indexes) = self.indexes()? else { + return Err(Error::GraphError( + "name glob requires columnar snapshot indexes".into(), + )); + }; + let mut out = Vec::new(); + for (name, ids) in indexes.0.iter() { + if glob_match(symbol, name) { + for id in ids { + if let Some(n) = self.store.get_node(*id)? { + out.push(n); + } + } + } + } + Ok(out) + } + + /// Entity search (`rgctl find`). + pub fn find(&self, pattern: Option<&str>, filters: &QueryFilters) -> Result { + let mut candidates: Vec = Vec::new(); + if let Some(pat) = pattern { + candidates = self.lookup_name_candidates(pat, filters.exact)?; + } else if let Some(t) = filters.node_type { + if let Some(indexes) = self.indexes()? { + if let Some(ids) = indexes.1.get(&t) { + candidates.reserve(ids.len()); + for id in ids { + if let Some(n) = self.store.get_node(*id)? { + candidates.push(n); + } + } + } + } else { + for id in self.store.all_node_ids() { + if let Some(n) = self.store.get_node(id)? { + if n.node_type == t { + candidates.push(n); + } + } + } + } + } else { + for id in self.store.all_node_ids() { + if let Some(n) = self.store.get_node(id)? { + candidates.push(n); + } + } + } + + candidates.retain(|n| self.node_passes(n, filters)); + let total = candidates.len(); + let limit = filters.limit.unwrap_or(total); + let entities: Vec = if filters.count_only { + Vec::new() + } else { + candidates + .into_iter() + .take(limit) + .map(|n| Self::project(&n)) + .collect() + }; + let returned = if filters.count_only { + total.min(limit) + } else { + entities.len() + }; + Ok(FindResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + returned, + total, + entities, + }) + } + + /// Incoming (`callers`) or outgoing (`callees`) CALLS. + pub fn call_neighbors( + &self, + symbol: &str, + incoming: bool, + depth: usize, + filters: &QueryFilters, + ) -> Result { + let seed = self.resolve_symbol(symbol, filters)?; + let depth = depth.max(1); + let adj = self.build_typed_adjacency(EdgeType::Calls)?; + let mut seen = HashSet::from([seed.id]); + let mut q = VecDeque::from([(seed.id, 0usize)]); + let mut found: Vec<(Uuid, usize)> = Vec::new(); + while let Some((id, d)) = q.pop_front() { + if d >= depth { + continue; + } + let nexts = if incoming { + adj.incoming.get(&id) + } else { + adj.outgoing.get(&id) + }; + let Some(nexts) = nexts else { continue }; + for &nid in nexts { + if !seen.insert(nid) { + continue; + } + let hop = d + 1; + found.push((nid, hop)); + if hop < depth { + q.push_back((nid, hop)); + } + } + } + + let mut neighbors = Vec::new(); + let mut hops = Vec::new(); + for (id, hop) in &found { + let Some(n) = self.store.get_node(*id)? else { + continue; + }; + if !self.node_passes(&n, filters) && filters.scope.is_some() { + // For callers, scope typically filters the *neighbor* side. + if !Self::matches_scope(&n, filters.scope.as_deref(), filters.scope_mode) { + continue; + } + } + neighbors.push(Self::project(&n)); + hops.push(*hop); + } + let total = neighbors.len(); + let limit = filters.limit.unwrap_or(total); + if neighbors.len() > limit { + neighbors.truncate(limit); + hops.truncate(limit); + } + Ok(CallNeighborsResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + target: Self::project(&seed), + returned: neighbors.len(), + total, + neighbors, + hops, + depth, + direction: if incoming { + "callers".into() + } else { + "callees".into() + }, + }) + } + + /// Seeded or seedless relations. + pub fn relations( + &self, + symbol: Option<&str>, + edge: EdgeType, + direction: RelationDirection, + from_type: Option, + to_type: Option, + depth: usize, + filters: &QueryFilters, + ) -> Result { + if let Some(sym) = symbol { + return self.relations_seeded(sym, edge, direction, from_type, to_type, depth, filters); + } + self.relations_seedless(edge, direction, from_type, to_type, filters) + } + + fn relations_seedless( + &self, + edge: EdgeType, + direction: RelationDirection, + from_type: Option, + to_type: Option, + filters: &QueryFilters, + ) -> Result { + let mut edges = Vec::new(); + let mut total = 0usize; + let limit = filters.limit.unwrap_or(usize::MAX); + self.store.for_each_edge(|from, to, et| { + if et != edge { + return Ok(()); + } + let Some(src) = self.store.get_node(from)? else { + return Ok(()); + }; + let Some(dst) = self.store.get_node(to)? else { + return Ok(()); + }; + if let Some(ft) = from_type + && src.node_type != ft + { + return Ok(()); + } + if let Some(tt) = to_type + && dst.node_type != tt + { + return Ok(()); + } + if !Self::edge_scope_ok(&src, &dst, filters.scope.as_deref(), filters.scope_mode) { + return Ok(()); + } + // Emit according to direction (both = outbound orientation as stored). + let emit = match direction { + RelationDirection::Out | RelationDirection::Both => true, + RelationDirection::In => true, // still emit keyed as stored; direction field notes "in" view + }; + if !emit { + return Ok(()); + } + total += 1; + if edges.len() < limit { + let (source, target, dir_label) = match direction { + RelationDirection::In => (Self::project(&dst), Self::project(&src), "in"), + RelationDirection::Out | RelationDirection::Both => { + (Self::project(&src), Self::project(&dst), "out") + } + }; + edges.push(EdgeRow { + source, + edge: edge_type_cli(edge), + direction: dir_label.into(), + target, + hops: Some(1), + }); + } + Ok(()) + })?; + Ok(RelationsResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + target: None, + returned: edges.len(), + total, + edges, + }) + } + + fn relations_seeded( + &self, + symbol: &str, + edge: EdgeType, + direction: RelationDirection, + from_type: Option, + to_type: Option, + depth: usize, + filters: &QueryFilters, + ) -> Result { + let seed = self.resolve_symbol(symbol, filters)?; + let depth = depth.max(1); + let adj = self.build_typed_adjacency(edge)?; + let mut rows = Vec::new(); + let mut total = 0usize; + let limit = filters.limit.unwrap_or(usize::MAX); + + let walk = |incoming: bool, rows: &mut Vec, total: &mut usize| -> Result<()> { + let mut seen = HashSet::from([seed.id]); + let mut q = VecDeque::from([(seed.id, 0usize)]); + while let Some((id, d)) = q.pop_front() { + if d >= depth { + continue; + } + let nexts = if incoming { + adj.incoming.get(&id) + } else { + adj.outgoing.get(&id) + }; + let Some(nexts) = nexts else { continue }; + for &nid in nexts { + let hop = d + 1; + let (src_id, dst_id) = if incoming { (nid, id) } else { (id, nid) }; + let Some(src) = self.store.get_node(src_id)? else { + continue; + }; + let Some(dst) = self.store.get_node(dst_id)? else { + continue; + }; + if let Some(ft) = from_type + && src.node_type != ft + { + continue; + } + if let Some(tt) = to_type + && dst.node_type != tt + { + continue; + } + if !Self::edge_scope_ok(&src, &dst, filters.scope.as_deref(), filters.scope_mode) + { + continue; + } + *total += 1; + if rows.len() < limit { + rows.push(EdgeRow { + source: Self::project(&src), + edge: edge_type_cli(edge), + direction: if incoming { "in" } else { "out" }.into(), + target: Self::project(&dst), + hops: Some(hop), + }); + } + if seen.insert(nid) && hop < depth { + q.push_back((nid, hop)); + } + } + } + Ok(()) + }; + + match direction { + RelationDirection::Out => walk(false, &mut rows, &mut total)?, + RelationDirection::In => walk(true, &mut rows, &mut total)?, + RelationDirection::Both => { + walk(false, &mut rows, &mut total)?; + walk(true, &mut rows, &mut total)?; + } + } + + Ok(RelationsResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + target: Some(Self::project(&seed)), + returned: rows.len(), + total, + edges: rows, + }) + } + + /// Inventory with zero-count enums for type/edge. + pub fn inventory( + &self, + by: InventoryBy, + filters: &QueryFilters, + ) -> Result { + match by { + InventoryBy::Type => { + let mut map: HashMap = + ALL_NODE_TYPES.iter().map(|t| (*t, 0usize)).collect(); + for id in self.store.all_node_ids() { + let Some(n) = self.store.get_node(id)? else { + continue; + }; + if !self.node_passes(&n, filters) { + continue; + } + *map.entry(n.node_type).or_insert(0) += 1; + } + let mut counts: Vec = ALL_NODE_TYPES + .iter() + .map(|t| InventoryCount { + key: node_type_cli(*t), + count: *map.get(t).unwrap_or(&0), + }) + .collect(); + // Include any unexpected types not in ALL_NODE_TYPES. + for (t, c) in map { + if !ALL_NODE_TYPES.contains(&t) { + counts.push(InventoryCount { + key: node_type_cli(t), + count: c, + }); + } + } + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "type".into(), + counts, + }) + } + InventoryBy::Edge => { + let mut map: HashMap = + ALL_EDGE_TYPES.iter().map(|t| (*t, 0usize)).collect(); + self.store.for_each_edge(|from, to, et| { + if et == EdgeType::Unknown { + return Ok(()); + } + if filters.scope.is_some() { + let Some(src) = self.store.get_node(from)? else { + return Ok(()); + }; + let Some(dst) = self.store.get_node(to)? else { + return Ok(()); + }; + if !Self::edge_scope_ok( + &src, + &dst, + filters.scope.as_deref(), + filters.scope_mode, + ) { + return Ok(()); + } + } + *map.entry(et).or_insert(0) += 1; + Ok(()) + })?; + let counts = ALL_EDGE_TYPES + .iter() + .map(|t| InventoryCount { + key: edge_type_cli(*t), + count: *map.get(t).unwrap_or(&0), + }) + .collect(); + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "edge".into(), + counts, + }) + } + InventoryBy::Lang => { + let mut map: HashMap = HashMap::new(); + for id in self.store.all_node_ids() { + let Some(n) = self.store.get_node(id)? else { + continue; + }; + if !self.node_passes(&n, filters) { + continue; + } + let lang = n + .get_property("language") + .map(|s| s.to_string()) + .unwrap_or_else(|| { + let ext = n + .file_path + .as_deref() + .and_then(|p| p.rsplit('.').next()) + .unwrap_or(""); + lang_from_ext(ext) + }); + *map.entry(lang).or_insert(0) += 1; + } + let mut counts: Vec<_> = map + .into_iter() + .map(|(key, count)| InventoryCount { key, count }) + .collect(); + counts.sort_by(|a, b| a.key.cmp(&b.key)); + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "lang".into(), + counts, + }) + } + InventoryBy::File => { + let mut map: HashMap = HashMap::new(); + for id in self.store.all_node_ids() { + let Some(n) = self.store.get_node(id)? else { + continue; + }; + if !self.node_passes(&n, filters) { + continue; + } + let key = n + .file_path + .as_deref() + .unwrap_or("") + .to_string(); + *map.entry(key).or_insert(0) += 1; + } + let mut counts: Vec<_> = map + .into_iter() + .map(|(key, count)| InventoryCount { key, count }) + .collect(); + counts.sort_by(|a, b| a.key.cmp(&b.key)); + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "file".into(), + counts, + }) + } + InventoryBy::Community => { + let mut map: HashMap = HashMap::new(); + for id in self.store.all_node_ids() { + let Some(n) = self.store.get_node(id)? else { + continue; + }; + if !self.node_passes(&n, filters) { + continue; + } + let key = n + .get_property("community_id") + .unwrap_or("") + .to_string(); + *map.entry(key).or_insert(0) += 1; + } + let mut counts: Vec<_> = map + .into_iter() + .map(|(key, count)| InventoryCount { key, count }) + .collect(); + counts.sort_by(|a, b| a.key.cmp(&b.key)); + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "community".into(), + counts, + }) + } + } + } + + fn build_typed_adjacency(&self, edge: EdgeType) -> Result { + let mut outgoing: HashMap> = HashMap::new(); + let mut incoming: HashMap> = HashMap::new(); + self.store.for_each_edge(|from, to, et| { + if et != edge { + return Ok(()); + } + outgoing.entry(from).or_default().push(to); + incoming.entry(to).or_default().push(from); + Ok(()) + })?; + Ok(TypedAdj { outgoing, incoming }) + } +} + +struct TypedAdj { + outgoing: HashMap>, + incoming: HashMap>, +} + +/// Relation walk direction. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RelationDirection { + /// Follow edge as stored (from → to) + Out, + /// Reverse (to → from) + In, + /// Both + Both, +} + +impl RelationDirection { + /// Parse CLI token. + pub fn parse(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "out" | "outgoing" => Ok(Self::Out), + "in" | "incoming" => Ok(Self::In), + "both" => Ok(Self::Both), + other => Err(Error::InvalidQuery(format!( + "unknown direction '{other}' (expected in|out|both)" + ))), + } + } +} + +/// Inventory dimension. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum InventoryBy { + /// Node types + Type, + /// Edge types + Edge, + /// Language + Lang, + /// File path + File, + /// Community id property + Community, +} + +impl InventoryBy { + /// Parse CLI token. + pub fn parse(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "type" => Ok(Self::Type), + "edge" => Ok(Self::Edge), + "lang" | "language" => Ok(Self::Lang), + "file" => Ok(Self::File), + "community" => Ok(Self::Community), + other => Err(Error::InvalidQuery(format!( + "unknown inventory --by '{other}' (expected type|edge|lang|file|community)" + ))), + } + } +} + +/// Parse CLI node type string. +pub fn parse_node_type(s: &str) -> Result { + let key = s.to_ascii_lowercase().replace('-', "").replace('_', ""); + for t in ALL_NODE_TYPES { + if node_type_cli(*t).replace('_', "") == key { + return Ok(*t); + } + } + // Common aliases + match key.as_str() { + "config" => Ok(NodeType::ConfigKey), + "fn" | "method" => Ok(NodeType::Function), + "trait" => Ok(NodeType::Interface), + _ => Err(Error::InvalidQuery(format!( + "unknown node type '{s}' (e.g. function, class, import, annotation)" + ))), + } +} + +/// Parse CLI edge type string. +pub fn parse_edge_type(s: &str) -> Result { + let key = s.to_ascii_lowercase().replace('-', "").replace('_', ""); + for t in ALL_EDGE_TYPES { + if edge_type_cli(*t).replace('_', "") == key { + return Ok(*t); + } + } + Err(Error::InvalidQuery(format!( + "unknown edge type '{s}' (e.g. calls, uses, extends, implements, annotatedwith)" + ))) +} + +fn node_type_cli(t: NodeType) -> String { + format!("{t:?}").to_ascii_lowercase() +} + +fn edge_type_cli(t: EdgeType) -> String { + format!("{t:?}").to_ascii_lowercase() +} + +fn is_glob_pattern(s: &str) -> bool { + s.contains('*') || s.contains('?') +} + +/// Normalize scope keys so `/`, `\`, and `.` compare equivalently (PHP vs POSIX). +pub fn normalize_scope_key(s: &str) -> String { + s.chars() + .map(|c| match c { + '\\' | '/' | '.' => '/', + other => other, + }) + .collect() +} + +/// True when node's FQN or (if missing) file_path matches the scope prefix. +/// +/// Go/TS often leave `qualified_name` unset on types; package identity lives in `file_path`. +fn node_in_scope(node: &Node, scope_prefix: &str) -> bool { + let want = normalize_scope_key(scope_prefix); + if want.is_empty() { + return true; + } + if let Some(qn) = node.qualified_name.as_deref().filter(|s| !s.is_empty()) { + let hay = normalize_scope_key(qn); + return hay.starts_with(&want); + } + // Fallback when qualified_name is absent (common for Go/TS types). + let path = node.file_path.as_deref().unwrap_or(""); + if path.is_empty() { + return false; + } + let hay = normalize_scope_key(path); + path_contains_scope_segment(&hay, &want) +} + +/// Match `want` as a path segment / prefix inside a normalized file path. +fn path_contains_scope_segment(hay: &str, want: &str) -> bool { + if hay == want || hay.starts_with(&format!("{want}/")) || hay.ends_with(&format!("/{want}")) { + return true; + } + hay.contains(&format!("/{want}/")) +} + +/// If `--file` has no directory separator and no leading `*`, treat as basename glob. +pub fn normalize_file_glob(raw: &str) -> String { + if raw.is_empty() { + return raw.to_string(); + } + if raw.contains('/') || raw.contains('\\') || raw.starts_with('*') { + return raw.to_string(); + } + format!("*{raw}") +} + +/// Glob match with `*` and `?` (not full regex). +pub fn glob_match(pattern: &str, value: &str) -> bool { + glob_match_rec(pattern.as_bytes(), value.as_bytes()) +} + +fn glob_match_rec(pat: &[u8], val: &[u8]) -> bool { + let mut pi = 0usize; + let mut vi = 0usize; + let mut star_p = None; + let mut star_v = 0usize; + while vi < val.len() { + if pi < pat.len() && (pat[pi] == b'?' || pat[pi] == val[vi]) { + pi += 1; + vi += 1; + } else if pi < pat.len() && pat[pi] == b'*' { + star_p = Some(pi); + star_v = vi; + pi += 1; + } else if let Some(sp) = star_p { + pi = sp + 1; + star_v += 1; + vi = star_v; + } else { + return false; + } + } + while pi < pat.len() && pat[pi] == b'*' { + pi += 1; + } + pi == pat.len() +} + +fn lang_from_ext(ext: &str) -> String { + match ext.to_ascii_lowercase().as_str() { + "rs" => "rust".into(), + "java" => "java".into(), + "c" | "h" => "c".into(), + "cc" | "cpp" | "cxx" | "hpp" => "cpp".into(), + "py" => "python".into(), + "go" => "go".into(), + "js" | "mjs" | "cjs" => "javascript".into(), + "ts" | "tsx" => "typescript".into(), + "cs" => "csharp".into(), + "rb" => "ruby".into(), + "php" => "php".into(), + "kt" | "kts" => "kotlin".into(), + "groovy" => "groovy".into(), + "pp" => "puppet".into(), + "erb" => "erb".into(), + other => other.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::backend::{GraphBackend, MemoryBackend}; + use crate::columnar_snapshot::write_columnar_from_backend; + use crate::schema::Edge; + use crate::snapshot::SnapshotNodeStore; + use tempfile::tempdir; + + fn sample_store() -> (tempfile::TempDir, SnapshotNodeStore) { + let dir = tempdir().unwrap(); + let path = dir.path().join("graph.snapshot.bin"); + let mut backend = MemoryBackend::new(); + let f1 = Node::new(NodeType::Function, "getNextPrintPackage") + .with_qualified_name("de.metas.printing.esb.Svc.getNextPrintPackage") + .with_file_path("src/Svc.java") + .with_location(10, 20); + let a1 = Node::new(NodeType::Annotation, "Path") + .with_qualified_name("javax.ws.rs.Path") + .with_file_path(""); + let c1 = Node::new(NodeType::Class, "PRTRestServiceRoute") + .with_qualified_name("de.metas.printing.esb.camel.PRTRestServiceRoute") + .with_file_path("src/PRTRestServiceRoute.java") + .with_location(1, 50); + let base = Node::new(NodeType::Class, "RouteBuilder") + .with_qualified_name("org.apache.camel.builder.RouteBuilder") + .with_file_path(""); + let imp = Node::new(NodeType::Import, "import javax.ws.rs.Path;") + .with_file_path("src/Svc.java"); + let caller = Node::new(NodeType::Function, "handle") + .with_qualified_name("de.metas.printing.esb.Svc.handle") + .with_file_path("src/Svc.java") + .with_location(30, 40); + let f1_id = f1.id; + let a1_id = a1.id; + let c1_id = c1.id; + let base_id = base.id; + let caller_id = caller.id; + backend.insert_node(f1).unwrap(); + backend.insert_node(a1).unwrap(); + backend.insert_node(c1).unwrap(); + backend.insert_node(base).unwrap(); + backend.insert_node(imp).unwrap(); + backend.insert_node(caller).unwrap(); + backend + .insert_edge(Edge::new(f1_id, a1_id, EdgeType::AnnotatedWith)) + .unwrap(); + backend + .insert_edge(Edge::new(c1_id, base_id, EdgeType::Extends)) + .unwrap(); + backend + .insert_edge(Edge::new(caller_id, f1_id, EdgeType::Calls)) + .unwrap(); + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + (dir, store) + } + + #[test] + fn seedless_annotatedwith_keyed_and_typed() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .relations( + None, + EdgeType::AnnotatedWith, + RelationDirection::Out, + Some(NodeType::Function), + Some(NodeType::Annotation), + 1, + &QueryFilters { + scope: Some("de.metas.printing.esb".into()), + scope_mode: ScopeMode::Inside, + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 1); + assert_eq!(res.edges[0].source.node_type, "function"); + assert_eq!(res.edges[0].target.node_type, "annotation"); + assert_eq!(res.edges[0].source.name, "getNextPrintPackage"); + assert_eq!(res.edges[0].target.name, "Path"); + } + + #[test] + fn extends_keyed_source_is_subtype() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .relations( + None, + EdgeType::Extends, + RelationDirection::Out, + Some(NodeType::Class), + None, + 1, + &QueryFilters::default(), + ) + .unwrap(); + assert_eq!(res.total, 1); + assert_eq!(res.edges[0].source.name, "PRTRestServiceRoute"); + assert_eq!(res.edges[0].target.name, "RouteBuilder"); + } + + #[test] + fn inventory_type_includes_zeros() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .inventory(InventoryBy::Type, &QueryFilters::default()) + .unwrap(); + let config = res + .counts + .iter() + .find(|c| c.key == "configkey") + .expect("configkey present"); + assert_eq!(config.count, 0); + let functions = res.counts.iter().find(|c| c.key == "function").unwrap(); + assert!(functions.count >= 2); + } + + #[test] + fn find_import_prefix() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .find( + Some("import javax*"), + &QueryFilters { + node_type: Some(NodeType::Import), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 1); + assert!(res.entities[0].name.starts_with("import javax")); + } + + #[test] + fn callers_depth_one() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .call_neighbors("getNextPrintPackage", true, 1, &QueryFilters::default()) + .unwrap(); + assert_eq!(res.returned, 1); + assert_eq!(res.neighbors[0].name, "handle"); + } + + #[test] + fn glob_contains() { + assert!(glob_match("*Service*", "PRTRestServiceRoute")); + assert!(!glob_match("*Service*", "Path")); + } + + #[test] + fn normalize_scope_separators() { + assert_eq!(normalize_scope_key(r"App\Http"), "App/Http"); + assert_eq!(normalize_scope_key("App.Http"), "App/Http"); + assert_eq!(normalize_scope_key("App/Http"), "App/Http"); + } + + #[test] + fn normalize_file_basename_glob() { + assert_eq!(normalize_file_glob("PRTRestServiceRoute.java"), "*PRTRestServiceRoute.java"); + assert_eq!( + normalize_file_glob("*/printing/**/*.java"), + "*/printing/**/*.java" + ); + assert_eq!(normalize_file_glob("*Route.java"), "*Route.java"); + } + + #[test] + fn scope_falls_back_to_file_path_when_qn_absent() { + let dir = tempdir().unwrap(); + let path = dir.path().join("snap.bin"); + let mut backend = MemoryBackend::new(); + let n = Node::new(NodeType::Function, "ServeHTTP") + .with_file_path("/tmp/pkg/util/handler.go"); + // No qualified_name — Go/TS style. + backend.insert_node(n).unwrap(); + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + let q = StructuredQuery::new(&store); + let hit = q + .find( + Some("ServeHTTP"), + &QueryFilters { + scope: Some("pkg/util".into()), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(hit.total, 1); + let miss = q + .find( + Some("ServeHTTP"), + &QueryFilters { + scope: Some("pkg/other".into()), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(miss.total, 0); + } + + #[test] + fn scope_php_backslash_matches_slash_qn() { + let dir = tempdir().unwrap(); + let path = dir.path().join("snap.bin"); + let mut backend = MemoryBackend::new(); + let n = Node::new(NodeType::Class, "Controller") + .with_qualified_name(r"App\Http\Controllers\Controller"); + backend.insert_node(n).unwrap(); + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + let q = StructuredQuery::new(&store); + let hit = q + .find( + Some("Controller"), + &QueryFilters { + scope: Some("App/Http".into()), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(hit.total, 1); + } + + #[test] + fn file_filter_accepts_basename() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .find( + Some("*Service*"), + &QueryFilters { + file_glob: Some("PRTRestServiceRoute.java".into()), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 1); + assert_eq!(res.entities[0].name, "PRTRestServiceRoute"); + } + + #[test] + fn reject_unknown_edge_and_type() { + assert!(parse_edge_type("not_a_real_edge").is_err()); + assert!(parse_node_type("not_a_type").is_err()); + assert!(parse_edge_type("annotatedwith").is_ok()); + assert!(parse_node_type("import").is_ok()); + } +} diff --git a/crates/rgctl-lang-groovy/groovy-ast-coverage.json b/crates/rgctl-lang-groovy/groovy-ast-coverage.json index 2108d193..d91fdab8 100644 --- a/crates/rgctl-lang-groovy/groovy-ast-coverage.json +++ b/crates/rgctl-lang-groovy/groovy-ast-coverage.json @@ -2,7 +2,7 @@ "grammar": "tree-sitter-groovy@0.1.2", "handlers": { "annotated_type": "Skip", - "annotation": "AstSkeleton", + "annotation": "Relation", "annotation_argument_list": "Skip", "annotation_type_body": "Skip", "annotation_type_declaration": "Symbol", @@ -81,7 +81,7 @@ "local_variable_declaration": "Skip", "map_item": "Skip", "map_literal": "Literal", - "marker_annotation": "AstSkeleton", + "marker_annotation": "Relation", "method_declaration": "Symbol", "method_invocation": "Relation", "method_reference": "Relation", diff --git a/crates/rgctl-lang-groovy/src/plugin.rs b/crates/rgctl-lang-groovy/src/plugin.rs index ce6a6a90..aa142a41 100644 --- a/crates/rgctl-lang-groovy/src/plugin.rs +++ b/crates/rgctl-lang-groovy/src/plugin.rs @@ -164,7 +164,8 @@ impl GroovyPlugin { while let Some(node) = stack.pop() { match node.kind() { - "class_declaration" | "interface_declaration" | "enum_declaration" => { + "class_declaration" | "interface_declaration" | "enum_declaration" + | "annotation_type_declaration" => { let Some(simple) = Self::type_name(node, source) else { let mut cursor = node.walk(); for child in node.children(&mut cursor).collect::>().into_iter().rev() @@ -176,6 +177,7 @@ impl GroovyPlugin { let symbol_type = match node.kind() { "interface_declaration" => SymbolType::Interface, "enum_declaration" => SymbolType::Enum, + "annotation_type_declaration" => SymbolType::Annotation, _ => SymbolType::Class, }; let qn = Self::qualify(package.as_deref(), &simple); @@ -394,6 +396,20 @@ impl GroovyPlugin { } } } + // AnnotatedWith on types / methods / constructors (Java-shaped modifiers). + if matches!( + node.kind(), + "class_declaration" + | "interface_declaration" + | "enum_declaration" + | "annotation_type_declaration" + | "method_declaration" + | "function_definition" + | "constructor_declaration" + | "compact_constructor_declaration" + ) { + Self::push_annotated_with_for(node, source, file_path, root, &mut relations); + } let mut cursor = node.walk(); for child in node.children(&mut cursor).collect::>().into_iter().rev() { stack.push(child); @@ -402,6 +418,101 @@ impl GroovyPlugin { Ok(relations) } + /// Emit `AnnotatedWith` for `@Foo` / `@Foo(...)` on a declaration. + fn push_annotated_with_for( + node: Node, + source: &[u8], + file_path: &Path, + root: Node, + relations: &mut Vec, + ) { + let package = Self::package_name(root, source); + let from = match node.kind() { + "class_declaration" + | "interface_declaration" + | "enum_declaration" + | "annotation_type_declaration" => Self::type_name(node, source) + .map(|n| Self::qualify(package.as_deref(), &n)), + "method_declaration" | "function_definition" => { + let name = node + .child_by_field_name("name") + .and_then(|n| n.utf8_text(source).ok()) + .map(|s| s.trim().to_string()); + name.map(|name| { + if let Some(cls) = Self::enclosing_class(node, source) { + Self::qualify(package.as_deref(), &format!("{cls}.{name}")) + } else { + Self::qualify(package.as_deref(), &name) + } + }) + } + "constructor_declaration" | "compact_constructor_declaration" => { + let cls = Self::enclosing_class(node, source) + .or_else(|| Self::type_name(node, source)) + .unwrap_or_else(|| "Unknown".into()); + Some(Self::qualify(package.as_deref(), &format!("{cls}."))) + } + _ => None, + }; + let Some(from) = from else { + return; + }; + + let mut cursor = node.walk(); + let modifiers = node.children(&mut cursor).find(|c| c.kind() == "modifiers"); + let Some(modifiers) = modifiers else { + return; + }; + let mut mc = modifiers.walk(); + for child in modifiers.children(&mut mc) { + if !matches!(child.kind(), "annotation" | "marker_annotation") { + continue; + } + let raw_name = child + .child_by_field_name("name") + .and_then(|n| n.utf8_text(source).ok()) + .map(str::to_string) + .or_else(|| { + // Fallback: strip leading `@` from annotation text. + child.utf8_text(source).ok().map(|t| { + t.trim() + .trim_start_matches('@') + .split(['(', ' ', '\n']) + .next() + .unwrap_or("") + .to_string() + }) + }) + .unwrap_or_default(); + let simple = raw_name + .rsplit('.') + .next() + .unwrap_or(&raw_name) + .trim() + .to_string(); + if simple.is_empty() { + continue; + } + let args = child + .child_by_field_name("arguments") + .and_then(|n| n.utf8_text(source).ok()) + .map(str::to_string); + let mut metadata = serde_json::json!({ "language": "groovy" }); + if let Some(args) = args { + metadata["arguments"] = serde_json::Value::String(args); + } + relations.push(Relation { + from: from.clone(), + to: simple, + relation_type: RelationType::AnnotatedWith, + location: Self::loc(&file_path.to_string_lossy(), child), + metadata, + to_qualified_hint: None, + to_type_hint: None, + }); + } + } + fn calculate_cyclomatic(&self, node: Node) -> usize { let mut complexity = 1usize; let mut stack = vec![node]; @@ -545,4 +656,29 @@ mod tests { all.relations ); } + + #[test] + fn extracts_annotated_with() { + let src = b"@CompileStatic\nclass OrderService {\n @Override\n String toString() { \"x\" }\n}\n"; + let plugin = GroovyPlugin::new().unwrap(); + let all = plugin + .extract_all(Path::new("OrderService.groovy"), src) + .unwrap(); + assert!( + all.relations.iter().any(|r| { + r.relation_type == RelationType::AnnotatedWith + && r.to == "CompileStatic" + && r.from.contains("OrderService") + }), + "expected AnnotatedWith CompileStatic: {:?}", + all.relations + ); + assert!( + all.relations.iter().any(|r| { + r.relation_type == RelationType::AnnotatedWith && r.to == "Override" + }), + "expected AnnotatedWith Override: {:?}", + all.relations + ); + } } diff --git a/crates/rgctl-lang-typescript/src/plugin.rs b/crates/rgctl-lang-typescript/src/plugin.rs index 10593daa..23da5752 100644 --- a/crates/rgctl-lang-typescript/src/plugin.rs +++ b/crates/rgctl-lang-typescript/src/plugin.rs @@ -435,6 +435,132 @@ impl TypeScriptPlugin { }) } + fn extract_enum(&self, node: Node, source: &[u8], file_path: &str) -> Result { + let name = node + .child_by_field_name("name") + .and_then(|n| n.utf8_text(source).ok()) + .map(str::to_string) + .or_else(|| { + let mut cursor = node.walk(); + node.children(&mut cursor).find_map(|c| { + if matches!(c.kind(), "type_identifier" | "identifier") { + c.utf8_text(source).ok().map(str::to_string) + } else { + None + } + }) + }) + .ok_or_else(|| Error::ParseError { + file: file_path.into(), + line: node.start_position().row + 1, + message: "Enum missing name".to_string(), + })?; + + let mut fields = Vec::new(); + if let Some(body) = node.child_by_field_name("body") { + let mut cursor = body.walk(); + for child in body.children(&mut cursor) { + if child.kind() == "enum_assignment" + || child.kind() == "property_identifier" + || child.kind() == "identifier" + { + let member = if child.kind() == "enum_assignment" { + child + .child_by_field_name("name") + .and_then(|n| n.utf8_text(source).ok()) + .map(str::to_string) + .or_else(|| { + let mut c = child.walk(); + child.children(&mut c).find_map(|n| { + if matches!(n.kind(), "property_identifier" | "identifier") { + n.utf8_text(source).ok().map(str::to_string) + } else { + None + } + }) + }) + } else { + child.utf8_text(source).ok().map(str::to_string) + }; + if let Some(member) = member.filter(|s| !s.is_empty() && s != ",") { + fields.push(Field { + name: member, + field_type: None, + visibility: None, + }); + } + } + } + } + + Ok(Symbol { + name, + symbol_type: SymbolType::Enum, + qualified_name: None, + location: SourceLocation { + file: file_path.to_string(), + start_line: node.start_position().row + 1, + end_line: node.end_position().row + 1, + start_column: node.start_position().column, + end_column: node.end_position().column, + }, + signature: None, + return_type: None, + parameters: vec![], + fields, + modifiers: vec![], + documentation: None, + metadata: serde_json::json!({ "language": "typescript" }), + }) + } + + fn extract_type_alias(&self, node: Node, source: &[u8], file_path: &str) -> Result { + let name = node + .child_by_field_name("name") + .and_then(|n| n.utf8_text(source).ok()) + .map(str::to_string) + .or_else(|| { + let mut cursor = node.walk(); + node.children(&mut cursor).find_map(|c| { + if matches!(c.kind(), "type_identifier" | "identifier") { + c.utf8_text(source).ok().map(str::to_string) + } else { + None + } + }) + }) + .ok_or_else(|| Error::ParseError { + file: file_path.into(), + line: node.start_position().row + 1, + message: "Type alias missing name".to_string(), + })?; + + let value = node + .child_by_field_name("value") + .and_then(|n| n.utf8_text(source).ok()) + .map(|s| s.trim().to_string()); + + Ok(Symbol { + name, + symbol_type: SymbolType::TypeAlias, + qualified_name: None, + location: SourceLocation { + file: file_path.to_string(), + start_line: node.start_position().row + 1, + end_line: node.end_position().row + 1, + start_column: node.start_position().column, + end_column: node.end_position().column, + }, + signature: value.clone(), + return_type: value, + parameters: vec![], + fields: vec![], + modifiers: vec![], + documentation: None, + metadata: serde_json::json!({ "language": "typescript" }), + }) + } + fn extract_interface_properties(&self, object_type: Node, source: &[u8]) -> Result> { let mut fields = Vec::new(); let mut cursor = object_type.walk(); @@ -626,6 +752,12 @@ impl TypeScriptPlugin { } } } + "enum_declaration" => { + symbols.push(plugin.extract_enum(node, source, file_path)?); + } + "type_alias_declaration" => { + symbols.push(plugin.extract_type_alias(node, source, file_path)?); + } "import_statement" => { symbols.extend(extract_import_symbols( node, @@ -1383,6 +1515,39 @@ export function gamma(n: number): number { // The important thing is we found the interface } + #[test] + fn test_extract_enum_and_type_alias() { + let plugin = TypeScriptPlugin::new().unwrap(); + let source = br#" +enum Status { Pending, Done } +type Id = string | number; +"#; + let symbols = plugin + .extract_symbols(Path::new("types.ts"), source) + .unwrap(); + let en = symbols + .iter() + .find(|s| s.name == "Status" && s.symbol_type == SymbolType::Enum) + .expect("Status enum"); + assert!( + en.fields.iter().any(|f| f.name == "Pending"), + "enum members: {:?}", + en.fields + ); + let alias = symbols + .iter() + .find(|s| s.name == "Id" && s.symbol_type == SymbolType::TypeAlias) + .expect("Id type alias"); + assert!( + alias + .signature + .as_deref() + .is_some_and(|s| s.contains("string")), + "alias signature: {:?}", + alias.signature + ); + } + #[test] fn test_extract_relations_calls() { let source = br#" diff --git a/docs/agents/USER_AGENTS_TEMPLATE.md b/docs/agents/USER_AGENTS_TEMPLATE.md index 97d4a644..a25d1e68 100644 --- a/docs/agents/USER_AGENTS_TEMPLATE.md +++ b/docs/agents/USER_AGENTS_TEMPLATE.md @@ -45,7 +45,7 @@ Artifacts live at **`{repo}/.rgctl/`**. Set `REPO` to the repository root: ```bash export REPO=/path/to/repo -rgctl -r "$REPO" -f json gql 'MATCH (n:Function) RETURN n LIMIT 20' +rgctl -r "$REPO" -f json find --type function --limit 20 ``` Upgrading from an old daemon install: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/` into the repo (see [installation.md](../installation.md)). @@ -58,16 +58,18 @@ Upgrading from an old daemon install: `rgctl migrate-cache` copies `~/.rgctl/cac |--------|---------| | Full session (graph + CFG + dashboard + semantic) | `rgctl discover PATH --full` (queryable after stage 1; status in `.rgctl/pipeline_status.json`) | | HTTP session (auto-pipeline) | `rgctl serve` — `GET /api/status`; `--no-pipeline` restores fail-fast | -| Inventory functions | `rgctl -f json gql --macro-name all_functions unused` | -| List communities | `rgctl -f json gql --macro-name all_communities unused` | -| Find symbol by pattern | `rgctl -f json gql "MATCH (n:Function) WHERE n.name LIKE '*Service*' RETURN n LIMIT 20"` | -| Find by FQN (not `n.name`) | `rgctl -f json gql "MATCH (n:Class) WHERE n.qualified_name = 'com.example.Foo' RETURN n"` | -| Community members | `rgctl -f json gql "MATCH (f:Function) WHERE f.community_id = '12' RETURN f LIMIT 20"` | +| Schema / counts (zeros included) | `rgctl -f json inventory --by type` or `--by edge` | +| Inventory / count functions | `rgctl -f json find --type function --count-only` | +| Find symbol by pattern | `rgctl -f json find "*Service*" --type function --limit 20` | +| Find by package scope | `rgctl -f json find --type class --scope com.example` | +| Callers / callees | `rgctl -f json callers --depth 2` / `callees ` | +| Typed edges (seedless OK) | `rgctl -f json relations --edge annotatedwith --from-type function --to-type annotation` | +| List communities | `rgctl -f json communities list` | | Natural-language function search | `rgctl semantic index` then `rgctl -f json semantic query "checkout flow" --limit 10` | | Community semantic search | `rgctl -f json semantic query "checkout" --scope community --limit 10` | | Impact before editing | `rgctl -f json blast-radius [--depth N]` | | Architectural hotspots | `rgctl -f json metrics --pagerank` | -| Call neighborhood | `rgctl -f json gql "MATCH (a:Function)-[:CALLS*1..3]->(b:Function) RETURN a,b LIMIT 50"` | +| Experimental Cypher | `rgctl -f json gql 'MATCH …'` — prefer find/callers/relations | | Doc headings / cross-links | `discover` indexes `.md` / `.mdx` by default; GQL on `:Module` with `kind=heading` and `REFERENCES` — see [markdown-context.md](../markdown-context.md) | | Obsidian vault from docs | `rgctl -r "$REPO" discover -l markdown` then `export --export-format obsidian --export-output "$REPO/vault" --query all` — see [markdown-context.md](../markdown-context.md#obsidian-vault-export) | | Doc section semantic search | `rgctl semantic index --scope docs --embedder hash` then `rgctl -f json semantic query "checkout flow" --scope docs --limit 10` (query scope does not filter — index must be doc-scoped) | @@ -92,7 +94,8 @@ Upgrading from an old daemon install: `rgctl migrate-cache` copies `~/.rgctl/cac ```bash export REPO=/path/to/repo -rgctl -r "$REPO" -f json gql 'MATCH (n:Function) RETURN n LIMIT 5' +rgctl -r "$REPO" -f json find --type function --limit 5 +rgctl -r "$REPO" -f json callers ShoppingCartService --depth 2 rgctl -r "$REPO" -f json blast-radius ShoppingCartService ``` @@ -110,7 +113,7 @@ See [http-api.md](../http-api.md). ## Rules of thumb 0. **Artifacts** — always `{repo}/.rgctl/` after `discover`. Add `.rgctl/` to `.gitignore`. -1. **Index first** — `gql`, `blast-radius`, `metrics` fail without `discover`. +1. **Index first** — `find`/`callers`/`relations`/`inventory`, `blast-radius`, `metrics` fail without `discover`. 2. **Discover target** — `cd repo && rgctl discover .` or `rgctl -r PATH discover` (no trailing `.` with `-r`). 3. **Use `-f json`** — stable `schema_version` fields; see [json-api.md](../json-api.md). 4. **`inspect` takes a symbol only** — no `--class` (use `blast-radius` for disambiguation). diff --git a/docs/internal/container-memory-management.md b/docs/internal/container-memory-management.md new file mode 100644 index 00000000..4a1651f5 --- /dev/null +++ b/docs/internal/container-memory-management.md @@ -0,0 +1,208 @@ +# Containerized Memory Management & RSS Threshold Enforcement for `rgctl` + +## Executive Summary + +When running `rgctl` on massive codebases (e.g., the Linux kernel with ~71k files, 2.65M nodes, and 1.86M functions), peak Resident Set Size (RSS) currently reaches **12.7 GB to 14.7 GB**. + +In containerized environments (Kubernetes pods, Docker containers, AWS ECS / Fargate, CI/CD runners), memory limits are strictly enforced by the Linux kernel using cgroups. If a process exceeds its assigned memory ceiling, the Linux kernel **immediately terminates it via the Out-Of-Memory (OOM) killer** with exit code `137` (`SIGKILL`). + +This report details: +1. **Anatomy of Memory Consumption**: Where the 12–15 GB is spent during Ingest, Spill, and Analysis. +2. **Container-Specific Failure Modes**: Why standard container environments amplify memory usage (e.g., thread explosion from host core detection). +3. **Immediate Operational Mitigations**: How to configure `rgctl` today to stay within reasonable boundaries. +4. **Architectural Roadmap**: A 5-pillar blueprint to introduce hard and soft memory caps, adaptive streaming, streaming columnar assembly, and graceful degradation. + +--- + +## 1. Anatomy of Memory Hotspots in `rgctl` + +Memory consumption in `rgctl` occurs in two distinct, non-overlapping macro phases: **Ingest (Discover & Spill)** and **Analysis (Topology & Graph Metrics)**. + +```mermaid +flowchart TD + subgraph Ingest["Phase 1: Ingest Hotspots (~12-14 GB Peak)"] + TS["Tree-Sitter Parsing
(Rayon worker pool)"] --> InFlight["In-Flight Channel Buffers
(DEFAULT_STREAM_CHANNEL_CAPACITY = 1024)"] + InFlight --> GB["GraphBuilder In-Memory Maps
(symbol_index, suffixes, symbol_files)"] + GB --> SpillSort["SegmentedSpill Sort Runs
(DEFAULT_SORT_RUN_BYTES = 256 MiB x 2)"] + SpillSort --> ColumnarAsm["Columnar Assembly
(node_rows, edge_rows, name_index in RAM)"] + end + + subgraph Analysis["Phase 2: Analysis Hotspots (~6-8 GB Peak)"] + CSR["PetGraphView & StructuralTopology
(Duplicate uuid_to_index HashMaps)"] --> Comm["CommunityDetector
(Vec>: 2.65M heap allocs)"] + CSR --> Centrality["Centrality Engine
(Dense f64 PageRank, Brandes scratch, HyperBall)"] + CSR --> CFG["Control Flow / PDG Archives
(--with-cfg / --full)"] + end +``` + +### 1.1 Ingest Phase Hotspots + +| Component | Source Location | Mechanism | Impact on Large Repos (Linux Kernel) | +|---|---|---|---| +| **In-Flight Queue** | `crates/rgctl-pipeline/src/stream.rs:16` | `DEFAULT_STREAM_CHANNEL_CAPACITY = 1024` buffers `FileExtraction` payloads between worker threads and merge. | Up to 1,024 parsed AST results with symbol/relation vectors held in RAM concurrently. | +| **Pass-1 Name Maps** | `crates/rgctl-extraction/src/graph_builder.rs:28-69` | `HashMap` and `HashMap>` for `symbol_index`, `symbols_by_qualified`, `symbols_by_suffix`, and `symbol_files`. | 2.65M heap strings + hash table bucket overhead = **several gigabytes** of heap allocations. | +| **Spill External Sort** | `crates/rgctl-graph/src/segmented_spill.rs:29` | `DEFAULT_SORT_RUN_BYTES = 256 * 1024 * 1024`. Sort runs for nodes and edges execute concurrently in Rayon. | 2 × 256 MiB raw record buffers + deserialized records + I/O buffers = **~1.0 to 1.5 GB**. | +| **Columnar Assembly** | `crates/rgctl-graph/src/segmented_spill.rs:265-332` | `write_columnar_from_spill` loads all sorted nodes and edges into `node_rows`, `edge_rows`, `name_index`, `type_index`, and `StringPool`. | Entire node and edge index table materialized in memory simultaneously prior to serialization. | + +### 1.2 Analysis Phase Hotspots + +| Component | Source Location | Mechanism | Impact on Large Repos (Linux Kernel) | +|---|---|---|---| +| **Dual Topology Maps** | `crates/rgctl-analysis/src/graph_utils.rs:68-80` | `PetGraphView::from_topo` clones `index_to_uuid` and rebuilds a duplicate `uuid_to_index` map over `StructuralTopology`. | Two full 2.65M-entry `HashMap` structures in memory. | +| **Community Adjacency** | `crates/rgctl-analysis/src/community.rs:200-260` | Community detection uses `neighbors: Vec>` instead of flat CSR slices. | **2.65 million individual heap allocations**, leading to allocator metadata fragmentation. | +| **Centrality Calculations** | `crates/rgctl-analysis/src/centrality.rs` | Dense `Vec` arrays for PageRank, Brandes scratchpads for sampled betweenness. | Hundreds of MBs in dense arrays; exact harmonic BFS / HyperBall takes multi-GB if enabled. | + +--- + +## 2. The Containerization Trap: Why Containers OOM Faster + +When running inside Docker or Kubernetes, two specific system interactions cause memory consumption to spike even higher than on bare-metal machines: + +``` +HOST MACHINE (e.g., 64-128 Physical/Logical Cores, 256 GB RAM) +┌────────────────────────────────────────────────────────────────────────┐ +│ Kubernetes Pod / Container (CPU Limit: 2 Cores, Memory Limit: 4 GB) │ +│ │ +│ ❌ CPU Detection Trap: │ +│ std::thread::available_parallelism() returns 128 (Host cores)! │ +│ Rayon spawns 128 worker threads inside a 2-core container. │ +│ │ +│ ❌ Memory Amplification: │ +│ 128 threads x (Tree-Sitter scratch + stack + in-flight queue) │ +│ Process allocates 6-8 GB in seconds -> Exceeds 4 GB limit. │ +│ │ +│ 💥 Linux Kernel OOM Killer: │ +│ cgroup memory.max exceeded -> SIGKILL (Exit code 137). │ +└────────────────────────────────────────────────────────────────────────┘ +``` + +1. **Host CPU Count vs CFS Quotas**: + - `rayon::ThreadPoolBuilder` and standard Rust concurrency libraries read `/sys/devices/system/cpu` or `sched_getaffinity`. On a 64- or 128-core host node, Rayon will spawn 64 to 128 threads even if the pod is limited to `resources.limits.cpu: 2`. + - 128 active workers parsing files concurrently will saturate the 1024-element channel buffer with large files, triggering massive allocation spikes. +2. **Missing Cgroup Memory Awareness**: + - Currently, `rgctl` does not inspect `/sys/fs/cgroup/memory.max` (cgroups v2) or `/sys/fs/cgroup/memory/memory.limit_in_bytes` (cgroups v1). + - The tool proceeds with desktop/server assumptions (`DEFAULT_SORT_RUN_BYTES = 256 MiB`, unconstrained queues). + +--- + +## 3. Operational Playbook: Immediate Mitigations for Containers + +If running `rgctl` in containerized CI/CD or production environments today, use the following operational flags and environment variables to constrain memory: + +### 3.1 Docker / Kubernetes Recommended Configuration + +```bash +# 1. Cap Rayon worker threads to match container CPU limits (Crucial!) +export RAYON_NUM_THREADS=4 + +# 2. Avoid deep passes on massive corpora inside small containers +# DO NOT pass --with-harmonic or --full on Linux-scale repos +rgctl discover . -v \ + --repo /workspace \ + -l c # Filter to target language if applicable +``` + +### 3.2 Feature Profile vs Memory Footprint + +| Command / Flag | Target Corpus | Peak RSS | Minimum Container RAM | +|---|---|---|---| +| `rgctl discover .` (Default) | Linux kernel (~71k files) | **~12.7 GB** | **16 GB** (or 14 GB + swap) | +| `rgctl discover .` (Default) | Medium Repo (~10k files, Roslyn/VS Code) | **~1.5 – 3.9 GB** | **4 GB – 6 GB** | +| `rgctl discover .` (Default) | Standard Repo (<2k files) | **~300 – 800 MB** | **1 GB – 2 GB** | +| `rgctl discover . --full` | Medium Repo (metasfresh) | **~6.4 GB** | **8 GB** | +| `rgctl discover . --with-harmonic` | Any large repo (>500k nodes) | **+3 – 5 GB** | **+6 GB** overhead | + +> [!WARNING] +> Never use `--with-harmonic` or `--full` in containers with less than 16 GB of memory on kernel-scale repositories. + +--- + +## 4. Architectural Solution: 5-Pillar Memory Guard Design + +To make `rgctl` bulletproof inside containerized environments with strict thresholds (e.g., 2 GB, 4 GB, or 8 GB), we recommend implementing the following 5-pillar design. + +```mermaid +flowchart LR + A["Pillar 1: Cgroup Auto-Sensing"] --> B["Pillar 2: Dynamic Budget Allocation"] + B --> C["Pillar 3: Adaptive Ingest Throttling"] + B --> D["Pillar 4: Zero-Residency CSR & Spill"] + B --> E["Pillar 5: Proactive OOM Circuit Breaker"] +``` + +### Pillar 1: Cgroup & Container Environment Sensing +Automatically detect container constraints at startup: +- Read cgroups v2 (`/sys/fs/cgroup/memory.max` and `/sys/fs/cgroup/cpu.max`) or cgroups v1 (`memory.limit_in_bytes` and `cpu.cfs_quota_us / cpu.cfs_period_us`). +- Provide an explicit CLI flag `--memory-limit ` and environment variable `RGCTL_MEMORY_LIMIT_MB`. +- Auto-calculate: + $$\text{Effective Budget} = \min(\text{CLI Flag}, \text{Cgroup Limit} \times 0.85, \text{Host Free RAM} \times 0.85)$$ + +### Pillar 2: Dynamic Budget Allocation +Partition the memory budget across pipeline phases: + +```text +Total Container Budget: 4,096 MB (4 GB) +├── Operating Headroom (15%): 614 MB (Kernel buffers, binary, thread stacks) +├── Phase 1 (Ingest Budget): 3,482 MB +│ ├── Rayon Worker Buffers: 512 MB (Workers × parser scratch) +│ ├── In-Flight Stream Channel: 256 MB (Dynamically scaled queue) +│ ├── GraphBuilder HashMaps: 2,000 MB (String pool & lookup tables) +│ └── External Sort Runs: 714 MB (Run buffers scaled down to 64 MB) +└── Phase 2 (Analysis Budget): 3,482 MB (Ingest freed, mmap opened) + ├── PetGraphView CSR: 500 MB (Single CSR topology) + ├── Analysis Results Columns: 1,200 MB (Flat f32/u32 arrays) + └── Working Scratch: 1,782 MB (Community & centrality buffers) +``` + +### Pillar 3: Adaptive Ingest & Streaming Throttling +1. **Thread Pool Clamping**: + Clamp `thread_count` to $\min(\text{cgroup\_quota}, \text{threads})$. +2. **Channel Capacity Adaptation**: + Instead of a static `DEFAULT_STREAM_CHANNEL_CAPACITY = 1024`, scale dynamically: + $$\text{Capacity} = \text{clamp}\left(\frac{\text{Budget MB}}{10}, 32, 1024\right)$$ + In a 2 GB container, capacity drops to 128 items, preventing worker extraction from outpacing the merge thread and ballooning heap memory. +3. **Sort Run Scaling**: + Scale `DEFAULT_SORT_RUN_BYTES` down from 256 MiB to 32 MiB or 64 MiB when budget is $\le 4\text{ GB}$. External merge-sort will take a few more I/O passes, but peak RSS will drop by hundreds of megabytes. + +### Pillar 4: Ingest & Analysis Compaction +1. **String Interning for Symbol Maps**: + Replace full `String` keys in `GraphBuilder` (`symbol_index`, `symbols_by_suffix`, `symbols_by_qualified`) with `CompactString` (inline up to 24 bytes) or an arena-backed `StringPool` integer index (`SymbolId`). +2. **Flatten Community Detection (`Vec>` $\to$ CSR)**: + In `crates/rgctl-analysis/src/community.rs`, replace `Vec>` with flat CSR index slices: + ```rust + pub struct FlatAdjacency { + pub offsets: Vec, // node_count + 1 + pub targets: Vec, // edge_count + } + ``` + Eliminates 2.65 million small heap allocations, reducing allocator overhead and fragmentation by over 1 GB on Linux-scale graphs. +3. **Deduplicate `uuid_to_index`**: + Remove the redundant `uuid_to_index` map in `PetGraphView`, borrowing directly from `StructuralTopology`. + +### Pillar 5: Proactive OOM Circuit Breaker (Soft-Landing) +A process killed by `SIGKILL` (137) produces zero artifacts and leaves a corrupt cache. Instead, `rgctl` should feature an active soft-landing monitor: + +1. **Background High-Frequency Monitor**: + The existing `MemoryMonitor` in `rgctl-core` samples every 50ms. +2. **Tripwire at 90% Budget**: + - If RSS exceeds $0.90 \times \text{Budget}$: + 1. **Pause Workers**: Signal the extraction channel to pause worker file reading. + 2. **Spill Flush**: Force `GraphBuilder` to immediately flush in-memory structures to disk spill. + 3. **Stage Shedding**: If memory remains critical during analysis, skip secondary metrics (e.g. skip betweenness and circular dependency detection, keep PageRank and basic topology). + 4. **Safe Exit**: If memory reaches 95%, write a consistent partial snapshot to `.rgctl/` and exit with an informative error message: + ```text + [ERROR] Memory limit of 4096 MB approached (current RSS: 3892 MB). + Discovery gracefully aborted to prevent container OOMKill. + Partial graph snapshot saved. Increase container memory or filter by language (-l). + ``` + +--- + +## 5. Implementation Summary & Recommendations + +| Priority | Action Item | Target Crate | Complexity | Expected RSS Reduction | +|---|---|---|---|---| +| **P0 (Immediate)** | Add `--memory-limit-mb` CLI & cgroup limit detection | `rgctl`, `rgctl-core` | Low | Prevents OOM kills via auto-scaling | +| **P0 (Immediate)** | Clamp Rayon worker pool to cgroup CPU quota | `rgctl-pipeline` | Low | Cuts multi-GB spikes on large host nodes | +| **P1 (Near-term)** | Dynamic `stream_channel_capacity` and `SORT_RUN_BYTES` | `rgctl-pipeline`, `rgctl-graph` | Medium | Saves 500 MB – 1.5 GB in ingest | +| **P1 (Near-term)** | Flatten `CommunityDetector` adjacency from `Vec>` to CSR | `rgctl-analysis` | Medium | Saves 1.0 – 1.5 GB in analysis | +| **P2 (Long-term)** | Replace `String` keys in `GraphBuilder` with interned IDs | `rgctl-extraction` | High | Saves 2.0 – 3.0 GB on 2M+ node graphs | +| **P2 (Long-term)** | Streaming columnar snapshot assembly directly from spill | `rgctl-graph` | High | Eliminates assembly-time RAM spike | diff --git a/docs/internal/gql-vs-graph-evaluation.md b/docs/internal/gql-vs-graph-evaluation.md new file mode 100644 index 00000000..136ecb61 --- /dev/null +++ b/docs/internal/gql-vs-graph-evaluation.md @@ -0,0 +1,5 @@ +# GQL vs graph evaluation + +Canonical copy (local reports, gitignored): + +[`.reports/gql-vs-graph-evaluation.md`](../../.reports/gql-vs-graph-evaluation.md) diff --git a/docs/internal/multi-language-structured-query-evaluation.md b/docs/internal/multi-language-structured-query-evaluation.md new file mode 100644 index 00000000..0ff10e80 --- /dev/null +++ b/docs/internal/multi-language-structured-query-evaluation.md @@ -0,0 +1,5 @@ +# Multi-language structured query evaluation + +Canonical copy (local reports, gitignored): + +[`.reports/multi-language-structured-query-evaluation.md`](../../.reports/multi-language-structured-query-evaluation.md) diff --git a/docs/internal/structured-query-design.md b/docs/internal/structured-query-design.md new file mode 100644 index 00000000..ca3678ea --- /dev/null +++ b/docs/internal/structured-query-design.md @@ -0,0 +1,5 @@ +# Structured query design + +Canonical copy (local reports, gitignored): + +[`.reports/structured-query-design.md`](../../.reports/structured-query-design.md) diff --git a/scripts/run-structured-query-field-tests.py b/scripts/run-structured-query-field-tests.py new file mode 100755 index 00000000..59cb03fc --- /dev/null +++ b/scripts/run-structured-query-field-tests.py @@ -0,0 +1,686 @@ +#!/usr/bin/env python3 +"""Exhaustive structured-query field runner (OpenSpec test-plan-multi-language). + +Usage: + python3 scripts/run-structured-query-field-tests.py [--phase warm|cold|all] [--lang go,java,...] + +Writes: + .reports/sq-field-results.jsonl + .reports/multi-language-structured-query-field-report.md +""" +from __future__ import annotations + +import argparse +import json +import os +import re +import subprocess +import sys +import time +from dataclasses import dataclass, asdict +from pathlib import Path +from typing import Any, Optional + +ROOT = Path(__file__).resolve().parents[1] +RGCTL = ROOT / "target" / "release" / "rgctl" +REPORTS = ROOT / ".reports" +JSONL = REPORTS / "sq-field-results.jsonl" +MD = REPORTS / "multi-language-structured-query-field-report.md" + +# Gate B corpora (see AGENTS.md / OpenSpec test plan) +CORPORA: dict[str, dict[str, Any]] = { + "go": { + "path": ROOT / "example" / "kubernetes", + "discover": ["discover", "-l", "go", "pkg/", "cmd/"], + "scope": "pkg/util", + "find_pat": "*Controller*", + "find_type": "struct", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "warm", + }, + "java": { + "path": ROOT / "example" / "metasfresh-4.9.8b", + "discover": ["discover", "--full", "."], + "scope": "de.metas", + "find_pat": "*Service*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": True, + "phase": "warm", + "latency": True, + }, + "php": { + "path": ROOT / "example" / "magento2", + "discover": ["discover", "-l", "php", "."], + "scope": "Magento/Customer", + "scope_alt": r"Magento\Customer", + "find_pat": "*Customer*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "warm", + }, + "python": { + "path": ROOT / "example" / "home-assistant", + "discover": ["discover", "-l", "python", "."], + "scope": "homeassistant", + "find_pat": "*Sensor*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "warm", + }, + "ruby": { + "path": ROOT / "example" / "discourse", + "discover": ["discover", "-l", "ruby", "."], + "scope": "app/models", + "find_pat": "*User*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "warm", + }, + "rust": { + "path": ROOT / "example" / "rust", + "discover": ["discover", "-l", "rust", "."], + "scope": "compiler", + "find_pat": "*Resolver*", + "find_type": "struct", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "warm", + }, + "groovy": { + "path": ROOT / "example" / "groovy", + "discover": ["discover", "-l", "groovy", "."], + "scope": "org.gradle", + "find_pat": "*Service*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": True, # after re-discover + "phase": "warm", + "rediscover": True, + }, + "kotlin": { + "path": ROOT / "example" / "kotlin", + "discover": ["discover", "-l", "kotlin", "."], + "scope": "libraries", + "find_pat": "*Factory*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": False, # corpus currently 0; track as extraction gap + "phase": "warm", + }, + "typescript": { + "path": ROOT / "example" / "vscode" / "src", + "discover": ["discover", "-l", "typescript", "."], + "scope": "vs/workbench", + "find_pat": "*Editor*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": True, + "phase": "cold", + "rediscover": True, + }, + "javascript": { + "path": ROOT / "example" / "node" / "test", + "discover": ["discover", "-l", "javascript", "."], + "scope": "parallel", + "find_pat": "*test*", + "find_type": "function", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "cold", + }, + "csharp": { + "path": ROOT / "example" / "roslyn" / "src", + "discover": ["discover", "-l", "csharp", "."], + "scope": "Compilers", + "find_pat": "*Syntax*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": True, + "phase": "cold", + }, + "cpp": { + "path": ROOT / "example" / "llvm-project" / "clang", + "discover": ["discover", "-l", "cpp", "."], + "scope": "lib", + "find_pat": "*Parser*", + "find_type": "function", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "cold", + }, + "c": { + "path": ROOT / "example" / "linux", + "discover": ["discover", "."], + "scope": "kernel", + "find_pat": "*sched*", + "find_type": "function", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "cold", + "optional": True, + }, + "puppet": { + "path": ROOT / "example" / "theforeman", + "discover": ["discover", "-l", "puppet,erb,ruby", "-e", "spec,vendor", "."], + "scope": "modules", + "find_pat": "*", + "find_type": "class", + "file_basename": None, + "expect_annotatedwith": False, + "phase": "cold", + "optional": True, + }, +} + + +def extract_json(text: str) -> Optional[Any]: + text = text.strip() + if not text: + return None + # Prefer first top-level JSON value (banners may precede; never rfind into nested objects). + for i, ch in enumerate(text): + if ch not in "{[": + continue + opener, closer = ("{", "}") if ch == "{" else ("[", "]") + depth = 0 + in_str = False + esc = False + for j in range(i, len(text)): + c = text[j] + if in_str: + if esc: + esc = False + elif c == "\\": + esc = True + elif c == '"': + in_str = False + continue + if c == '"': + in_str = True + elif c == opener: + depth += 1 + elif c == closer: + depth -= 1 + if depth == 0: + chunk = text[i : j + 1] + try: + return json.loads(chunk) + except json.JSONDecodeError: + break + # Plain count-only integer + if re.fullmatch(r"-?\d+", text.splitlines()[-1].strip() if text else ""): + return {"returned": int(text.splitlines()[-1].strip()), "total": int(text.splitlines()[-1].strip()), "schema_version": 1, "count_only": True} + return None + + +def probe_ok_envelope(data: Any, check_schema: bool) -> tuple[bool, str]: + if data is None: + return False, "no JSON" + if not isinstance(data, dict): + return False, f"non-object JSON: {type(data)}" + if not check_schema: + return True, "" + if "schema_version" in data: + return True, "" + if "by" in data and "counts" in data: + return True, "" # inventory envelope + if "count_only" in data: + return True, "" + return False, "missing schema_version" + + +@dataclass +class ProbeResult: + lang: str + probe_id: str + cmd: list[str] + ok: bool + exit_code: int + wall_s: float + detail: str + returned: Optional[int] = None + total: Optional[int] = None + extra: Optional[dict] = None + + +def run_rgctl(cwd: Path, args: list[str], timeout: int = 600) -> tuple[int, str, str, float]: + cmd = [str(RGCTL), "-f", "json", *args] + t0 = time.perf_counter() + try: + p = subprocess.run( + cmd, + cwd=str(cwd), + capture_output=True, + text=True, + timeout=timeout, + ) + wall = time.perf_counter() - t0 + return p.returncode, p.stdout, p.stderr, wall + except subprocess.TimeoutExpired as e: + wall = time.perf_counter() - t0 + out = (e.stdout or b"").decode() if isinstance(e.stdout, bytes) else (e.stdout or "") + err = (e.stderr or b"").decode() if isinstance(e.stderr, bytes) else (e.stderr or "") + return 124, out, err + "\nTIMEOUT", wall + + +def counts_map(data: Any) -> dict[str, int]: + if not isinstance(data, dict): + return {} + out = {} + for row in data.get("counts") or []: + if isinstance(row, dict) and "key" in row: + out[str(row["key"])] = int(row.get("count") or 0) + return out + + +def first_entity_name(data: Any) -> Optional[str]: + if not isinstance(data, dict): + return None + ents = data.get("entities") or data.get("neighbors") or [] + if ents and isinstance(ents[0], dict): + return ents[0].get("name") + return None + + +def probe( + lang: str, + probe_id: str, + cwd: Path, + args: list[str], + expect_ok: bool = True, + expect_total_gt: Optional[int] = None, + check_schema: bool = True, +) -> ProbeResult: + code, out, err, wall = run_rgctl(cwd, args) + data = extract_json(out) + detail_parts = [] + ok = (code == 0) if expect_ok else (code != 0) + returned = total = None + extra: dict[str, Any] = {} + if expect_ok: + env_ok, env_msg = probe_ok_envelope(data, check_schema) + if not env_ok: + ok = False + detail_parts.append(f"{env_msg}; stderr={err[:200]!r}") + elif isinstance(data, dict): + returned = data.get("returned") + total = data.get("total") + if returned is None and "counts" in data: + extra["nonzero"] = sum(1 for c in data["counts"] if c.get("count", 0) > 0) + extra["keys"] = len(data["counts"]) + if expect_total_gt is not None: + t = total if total is not None else 0 + if t <= expect_total_gt: + ok = False + detail_parts.append(f"total={t} not > {expect_total_gt}") + if "counts" in data: + cm = counts_map(data) + extra["sample"] = {k: cm[k] for k in list(cm)[:8]} + extra["top"] = sorted(cm.items(), key=lambda x: -x[1])[:6] + else: + detail_parts.append(f"exit={code} err={err[:160]!r}") + if data is not None and returned is None and isinstance(data, dict): + returned = data.get("returned") + total = data.get("total") + detail = "; ".join(detail_parts) if detail_parts else ("pass" if ok else "fail") + if returned is not None or total is not None: + detail = f"returned={returned} total={total}; {detail}" + return ProbeResult(lang, probe_id, [str(RGCTL), "-f", "json", *args], ok, code, wall, detail, returned, total, extra or None) + + +def has_snapshot(path: Path) -> bool: + return (path / ".rgctl" / "graph.snapshot.bin").is_file() + + +def ensure_discover(lang: str, cfg: dict[str, Any], force: bool = False) -> ProbeResult: + path: Path = cfg["path"] + if not path.is_dir(): + return ProbeResult(lang, "DISCOVER", [], False, 2, 0.0, f"corpus missing: {path}") + if has_snapshot(path) and not force and not cfg.get("rediscover"): + return ProbeResult(lang, "DISCOVER", [], True, 0, 0.0, "reuse warm snapshot") + if has_snapshot(path) and cfg.get("rediscover") and not force: + # still allow warm pass; rediscover flagged separately + return ProbeResult(lang, "DISCOVER", [], True, 0, 0.0, "warm snapshot present (rediscover deferred)") + args = cfg["discover"] + print(f"[{lang}] discovering: {' '.join(args)}", flush=True) + t0 = time.perf_counter() + p = subprocess.run( + [str(RGCTL), *args], + cwd=str(path), + capture_output=True, + text=True, + timeout=7200, + ) + wall = time.perf_counter() - t0 + ok = p.returncode == 0 and has_snapshot(path) + return ProbeResult( + lang, + "DISCOVER", + [str(RGCTL), *args], + ok, + p.returncode, + wall, + "ok" if ok else f"fail stderr={p.stderr[-400:]}", + ) + + +def run_lang(lang: str, cfg: dict[str, Any], do_discover: bool, force_rediscover: bool) -> list[ProbeResult]: + path: Path = cfg["path"] + results: list[ProbeResult] = [] + if not path.is_dir(): + results.append(ProbeResult(lang, "SKIP", [], False, 0, 0.0, f"corpus absent: {path}")) + return results + + if do_discover or (force_rediscover and cfg.get("rediscover")): + results.append(ensure_discover(lang, cfg, force=force_rediscover and bool(cfg.get("rediscover")))) + elif not has_snapshot(path): + results.append(ProbeResult(lang, "SKIP", [], False, 0, 0.0, "no snapshot; cold discover not requested")) + return results + else: + results.append(ProbeResult(lang, "DISCOVER", [], True, 0, 0.0, "reuse warm snapshot")) + + if not has_snapshot(path): + return results + + # A inventory + for by in ("type", "edge", "lang", "file", "community"): + results.append(probe(lang, f"A-{by}", path, ["inventory", "--by", by])) + + # B find + results.append(probe(lang, "B1", path, ["find", "--type", "function", "--count-only"])) + results.append(probe(lang, "B2", path, ["find", "--type", cfg["find_type"], "--count-only"])) + results.append(probe(lang, "B4", path, ["find", cfg["find_pat"], "--type", cfg["find_type"], "--limit", "20"])) + results.append(probe(lang, "B5", path, ["find", cfg["find_pat"], "--limit", "20"])) + results.append(probe(lang, "B6", path, ["find", "--type", "import", "--count-only"])) + results.append(probe(lang, "B10", path, ["find", "--type", "not_a_real_type"], expect_ok=False, check_schema=False)) + results.append(probe(lang, "B11", path, ["query", "find", cfg["find_pat"], "--limit", "5"])) + + # C scope + scope = cfg.get("scope") + if scope: + expect = 0 # just complete; mark soft + r = probe(lang, "C-scope", path, ["find", cfg["find_pat"], "--type", cfg["find_type"], "--scope", scope, "--limit", "10"]) + # Soft pass: command ok; note if total==0 + if r.ok and (r.total or 0) == 0: + r.detail += "; WARN total=0 (scope miss?)" + results.append(r) + if cfg.get("scope_alt"): + r = probe( + lang, + "C-scope-alt", + path, + ["find", cfg["find_pat"], "--type", cfg["find_type"], "--scope", cfg["scope_alt"], "--limit", "10"], + ) + if r.ok and (r.total or 0) == 0: + r.detail += "; WARN total=0" + results.append(r) + + # Seed symbol for callers from B4 — prefer concrete file_path (skip ) + code, out, _, _ = run_rgctl( + path, ["find", cfg["find_pat"], "--type", cfg["find_type"], "--limit", "50"] + ) + data = extract_json(out) + # Prefer a unique name (exact total==1) so callers/callees aren't blocked on ambiguity. + seed = None + seed_file = None + if isinstance(data, dict): + candidates = [] + for ent in data.get("entities") or []: + fp = ent.get("file") or ent.get("file_path") or "" + name = ent.get("name") + if name and fp and fp != "": + candidates.append((name, fp)) + for name, fp in candidates: + _, out_u, _, _ = run_rgctl(path, ["find", name, "--exact", "--count-only"]) + du = extract_json(out_u) + tot = (du or {}).get("total") if isinstance(du, dict) else None + if tot == 1: + seed, seed_file = name, fp + break + if seed is None and candidates: + seed, seed_file = candidates[0] + + if seed_file: + base = Path(seed_file).name + results.append(probe(lang, "C5-file-basename", path, ["find", "*", "--file", base, "--limit", "5"])) + # Path-segment scope from file path (Go/TS style) + parts = Path(seed_file).parts + if len(parts) >= 2: + seg = "/".join(parts[-3:-1]) if len(parts) >= 3 else parts[-2] + r = probe( + lang, + "C-scope-from-file", + path, + ["find", "*", "--type", cfg["find_type"], "--scope", seg, "--limit", "10"], + ) + if r.ok and (r.total or 0) == 0: + r.detail += "; WARN total=0" + results.append(r) + + if seed: + results.append(probe(lang, "B3-exact", path, ["find", seed, "--exact", "--limit", "5"])) + # Prefer basename --file for disambiguation (full paths often still collide on name). + file_flag = Path(seed_file).name if seed_file else None + call_args = ["callers", seed, "--depth", "1", "--limit", "20"] + callee_args = ["callees", seed, "--depth", "1", "--limit", "20"] + call2_args = ["callers", seed, "--depth", "2", "--limit", "20"] + rel_args = ["relations", seed, "--edge", "calls", "--direction", "out", "--depth", "1", "--limit", "20"] + if file_flag: + call_args += ["--file", file_flag] + callee_args += ["--file", file_flag] + call2_args += ["--file", file_flag] + rel_args += ["--file", file_flag] + results.append(probe(lang, "D1-callers", path, call_args)) + results.append(probe(lang, "D2-callees", path, callee_args)) + results.append(probe(lang, "D3-callers-d2", path, call2_args)) + results.append(probe(lang, "E7-seeded", path, rel_args)) + # D4: without --file — ambiguous error OR unique resolve both acceptable + code_d4, out_d4, err_d4, wall_d4 = run_rgctl(path, ["callers", seed, "--depth", "1", "--limit", "5"]) + d4_ok = code_d4 != 0 and "Ambiguous" in (err_d4 + out_d4) or code_d4 == 0 + results.append( + ProbeResult( + lang, + "D4-ambiguous", + [str(RGCTL), "-f", "json", "callers", seed], + d4_ok, + code_d4, + wall_d4, + f"exit={code_d4}; ambiguous_or_unique ok", + ) + ) + else: + results.append(ProbeResult(lang, "SEED", [], False, 0, 0.0, "no seed symbol from find")) + + results.append(probe(lang, "D5-missing", path, ["callers", "__no_such_symbol_zz__"], expect_ok=False, check_schema=False)) + + # E relations seedless + for edge, pid in ( + ("calls", "E1"), + ("extends", "E2"), + ("implements", "E3"), + ("annotatedwith", "E4"), + ("instantiates", "E6"), + ): + results.append(probe(lang, pid, path, ["relations", "--edge", edge, "--limit", "20"])) + results.append( + probe( + lang, + "E5", + path, + [ + "relations", + "--edge", + "annotatedwith", + "--from-type", + "function", + "--to-type", + "annotation", + "--limit", + "20", + ], + ) + ) + results.append(probe(lang, "E8", path, ["relations", "--edge", "not_a_real_edge"], expect_ok=False, check_schema=False)) + if scope: + results.append( + probe( + lang, + "E9", + path, + ["relations", "--edge", "extends", "--scope", scope, "--scope-mode", "inside", "--limit", "20"], + ) + ) + results.append( + probe( + lang, + "C6-crossing", + path, + ["relations", "--edge", "calls", "--scope", scope, "--scope-mode", "crossing", "--limit", "20"], + ) + ) + + # F extraction checks via inventory type/edge + code, out, _, wall = run_rgctl(path, ["inventory", "--by", "type"]) + tdata = extract_json(out) + cm = counts_map(tdata) + code2, out2, _, wall2 = run_rgctl(path, ["inventory", "--by", "edge"]) + em = counts_map(extract_json(out2)) + notes = [] + if cfg.get("expect_annotatedwith") and em.get("annotatedwith", 0) == 0: + notes.append("FAIL expect annotatedwith>0") + if lang == "typescript": + if cm.get("enum", 0) == 0: + notes.append("WARN enum=0 (need re-discover?)") + if cm.get("typealias", 0) == 0 and cm.get("type_alias", 0) == 0: + notes.append("WARN typealias=0 (need re-discover?)") + if lang == "groovy" and cm.get("annotation", 0) == 0: + notes.append("WARN annotation nodes=0 (usage edges may still exist)") + detail = f"types_top={sorted(cm.items(), key=lambda x:-x[1])[:5]}; edges_top={sorted(em.items(), key=lambda x:-x[1])[:5]}; " + "; ".join(notes) + ok_f = "FAIL" not in detail + results.append(ProbeResult(lang, "F-extract", ["inventory"], ok_f, 0, wall + wall2, detail, extra={"types": cm, "edges": em})) + + # G latency on java + if cfg.get("latency") and seed: + for pid, args in ( + ("G1-find-scope", ["find", cfg["find_pat"], "--type", cfg["find_type"], "--scope", scope, "--limit", "50"]), + ( + "G2-anno", + [ + "relations", + "--edge", + "annotatedwith", + "--from-type", + "function", + "--to-type", + "annotation", + "--limit", + "50", + ], + ), + ("G3-callers", ["callers", seed, "--depth", "1", "--limit", "50"]), + ): + # warm: run twice, keep second + run_rgctl(path, args) + r = probe(lang, pid, path, args) + if r.wall_s >= 3.0: + r.ok = False + r.detail += "; FAIL latency>=3s (GQL floor)" + results.append(r) + + return results + + +def write_report(all_results: list[ProbeResult]) -> None: + REPORTS.mkdir(parents=True, exist_ok=True) + with JSONL.open("w") as f: + for r in all_results: + f.write(json.dumps(asdict(r), default=str) + "\n") + + by_lang: dict[str, list[ProbeResult]] = {} + for r in all_results: + by_lang.setdefault(r.lang, []).append(r) + + lines = [ + "# Multi-language structured-query field report", + "", + f"Generated from OpenSpec [test-plan-multi-language.md](../openspec/changes/add-structured-query-cli/test-plan-multi-language.md).", + f"Binary: `{RGCTL}` (`rgctl` release).", + f"Probes: {len(all_results)}; languages: {len(by_lang)}.", + "", + "## Summary", + "", + "| Lang | Pass | Fail | Skip/notes | Max wall (s) |", + "|------|------|------|------------|--------------|", + ] + for lang, rows in sorted(by_lang.items()): + passes = sum(1 for r in rows if r.ok and r.probe_id != "SKIP") + fails = sum(1 for r in rows if not r.ok and r.probe_id != "SKIP") + skips = [r.detail for r in rows if r.probe_id == "SKIP"] + mx = max((r.wall_s for r in rows), default=0) + note = skips[0] if skips else "" + lines.append(f"| {lang} | {passes} | {fails} | {note[:60]} | {mx:.3f} |") + + lines += ["", "## Per-language probes", ""] + for lang, rows in sorted(by_lang.items()): + lines += [f"### {lang}", "", "| ID | OK | wall_s | detail |", "|----|----|--------|--------|"] + for r in rows: + mark = "✅" if r.ok else "❌" + det = r.detail.replace("|", "\\|")[:180] + lines.append(f"| {r.probe_id} | {mark} | {r.wall_s:.3f} | {det} |") + lines.append("") + + fails = [r for r in all_results if not r.ok and r.probe_id not in ("SKIP",)] + lines += ["## Failures", ""] + if not fails: + lines.append("None.") + else: + for r in fails: + lines.append(f"- **{r.lang}/{r.probe_id}**: {r.detail} — `{' '.join(r.cmd[-6:])}`") + lines.append("") + MD.write_text("\n".join(lines)) + print(f"Wrote {JSONL} and {MD}", flush=True) + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--phase", choices=("warm", "cold", "all"), default="warm") + ap.add_argument("--lang", default="", help="comma-separated language filter") + ap.add_argument("--discover", action="store_true", help="run discover when snapshot missing") + ap.add_argument("--rediscover", action="store_true", help="force re-discover for flagged langs") + args = ap.parse_args() + if not RGCTL.is_file(): + print(f"missing release binary: {RGCTL}", file=sys.stderr) + return 2 + + langs = [x.strip() for x in args.lang.split(",") if x.strip()] or list(CORPORA.keys()) + all_results: list[ProbeResult] = [] + for lang in langs: + cfg = CORPORA.get(lang) + if not cfg: + print(f"unknown lang {lang}", file=sys.stderr) + continue + phase = cfg.get("phase", "warm") + if args.phase == "warm" and phase != "warm": + all_results.append(ProbeResult(lang, "SKIP", [], True, 0, 0.0, f"phase={phase}; skipped in warm")) + continue + if args.phase == "cold" and phase != "cold": + continue + if cfg.get("optional") and args.phase != "all": + all_results.append(ProbeResult(lang, "SKIP", [], True, 0, 0.0, "optional corpus; use --phase all")) + continue + print(f"=== {lang} ===", flush=True) + need_discover = args.discover or not has_snapshot(cfg["path"]) + all_results.extend(run_lang(lang, cfg, do_discover=need_discover, force_rediscover=args.rediscover)) + + write_report(all_results) + hard_fails = [r for r in all_results if not r.ok and r.probe_id not in ("SKIP",)] + return 1 if hard_fails else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/rgctl/SKILL.md b/skills/rgctl/SKILL.md index 23d0f3ee..af1766ad 100644 --- a/skills/rgctl/SKILL.md +++ b/skills/rgctl/SKILL.md @@ -101,19 +101,27 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ ### 2. Query & Search +**Prefer structured verbs** (mmap; no Cypher). Parse `-f json` from **stdout** (`schema_version`); never `2>/dev/null`. + | User Intent | CLI Command | |-------------|-------------| -| Inventory functions | `gql --macro-name all_functions unused` | -| Find callers/callees | `gql "MATCH (a)-[:CALLS]->(b) WHERE ..."` | +| Schema / counts (incl. zeros) | `inventory --by type` or `inventory --by edge` | +| Count functions | `find --type function --count-only` | +| Find by name/type | `find "User*" --type class --limit 50` | +| javax import worklist | `find "import javax*" --type import --scope ` | +| Annotation pairs (seedless) | `relations --edge annotatedwith --from-type function --to-type annotation --scope ` | +| Find callers/callees | `callers --depth 1` / `callees ` | +| Outside callers of a module | `callers --scope --scope-mode outside` | +| EXTENDS / IMPLEMENTS inventory | `relations --edge extends --from-type class` (omit SYMBOL) | | Natural-language search | `semantic query "checkout flow"` | | List communities | `communities list` | -| Community members | `gql "MATCH (f) WHERE f.community_id='12'"` | | Subsystem ownership | `semantic query "X" --scope community` | | Refresh community labels | `communities label --write` | +| Ad-hoc Cypher (experimental) | `gql "MATCH …"` — uncanny valley; prefer verbs above | -**GQL limitations:** no `COUNT`/`ORDER BY`; LIKE prefix/suffix only; CALLS misses dynamic dispatch; Konveyor labels need backticks in `WHERE`. +**Complexity honesty:** exact name = hash index; prefix/`*mid*`/`--scope` may scan keys/columns until better indexes land. Module re-index is still a strong speed lever. Annotation **arguments** (e.g. `@Path("/x")`) are not indexed yet. -**See:** [GQL Reference](references/gql-reference.md), [Semantic Search Guide](../../docs/guides/semantic-search.md) +**See:** [Command Encyclopedia](references/command-encyclopedia.md) (find/callers/relations/inventory), [GQL Reference](references/gql-reference.md) (legacy), [Semantic Search Guide](../../docs/guides/semantic-search.md) ### 3. Impact & Safety @@ -164,7 +172,8 @@ Needs `discover --with-cfg`. `--function` is method name, not class. | "Where is checkout flow?" | `semantic query "checkout flow" --limit 10` | | "Impact if I change X" | `blast-radius X --depth 2` | | "Validate against policy" | `check --policy-file policy.json` | -| "Who calls X" | `gql "MATCH (a)-[:CALLS*1..3]->(b) WHERE a.name='X' RETURN a,b"` | +| "Who calls X" | `callers X --depth 2` (impact → `blast-radius X`) | +| "javax imports / annotations" | `find "import javax*" --type import`; `relations --edge annotatedwith --from-type function --to-type annotation` | | "Where is X mutated?" | `cpg mutations --type X --exclude-ctors` | ## Failure Playbook @@ -174,9 +183,10 @@ Needs `discover --with-cfg`. `--function` is method name, not class. | No `.rgctl/` in repo | Run `cd repo && rgctl discover .`; or `rgctl migrate-cache` from legacy daemon cache | | slice/inspect/cpg fails | Re-discover with `--with-cfg` | | semantic query fails | `semantic index` | -| Ambiguous symbol | Add `--class` or `--file`; disambiguate via GQL | +| Ambiguous symbol | Add `--class` or `--file` on callers/find | | `check` exit 1 | Report violations (JSON still on stdout) | -| GQL LIKE returns 0 | Try `communities list`, `semantic query`, or broader type patterns | +| find/relations empty | Run `inventory --by type` / `--by edge` (zeros mean unpopulated schema); check `--scope` | +| GQL LIKE returns 0 | Prefer structured verbs; or `semantic query` / `communities list` | ## Artifacts diff --git a/skills/rgctl/references/command-encyclopedia.md b/skills/rgctl/references/command-encyclopedia.md index 5e48b95c..540edd01 100644 --- a/skills/rgctl/references/command-encyclopedia.md +++ b/skills/rgctl/references/command-encyclopedia.md @@ -100,11 +100,41 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json --- +## find / callers / callees / relations / inventory + +**Commands:** + +```bash +rgctl -f json find [PATTERN] --type function --scope pkg --limit 50 +rgctl -f json find --type function --count-only +rgctl -f json callers --depth 1 --file PATH --class NAME +rgctl -f json callees --depth 1 +rgctl -f json relations [SYMBOL] --edge annotatedwith --from-type function --to-type annotation --scope pkg +rgctl -f json relations --edge extends --from-type class # seedless +rgctl -f json inventory --by type # includes zero-count kinds +rgctl -f json inventory --by edge +rgctl -f json query find … # alias namespace +``` + +**Purpose:** Deterministic mmap structured query (no Cypher, no `MemoryBackend` hydrate). Prefer these over `gql` for agent work. + +**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). + +**Flags:** `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. + +**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. + +**Agent should report:** counts, lean names/files, keyed edge pairs — not full node dumps. + +**See:** OpenSpec `add-structured-query-cli`; demote GQL only after latency + probe-catalog gates. + +--- + ## gql **Command:** `rgctl -f json gql ''` or `rgctl -f json gql --macro-name unused` -**Purpose:** Inventory, callers/callees, communities, path/relationship queries. +**Purpose:** Experimental Cypher subset. Prefer `find` / `callers` / `relations` / `inventory` for agent workflows. **Prerequisites:** `discover` done. Virtual `:Community` needs analysis overlay from discover. diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 9fdc8d9d..608a1799 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -39,6 +39,7 @@ pub mod semantic_output; mod slice; pub mod slice_output; mod stage_profile; +mod structured_query; pub use args::OutputFormat; @@ -218,7 +219,7 @@ pub enum Commands { extra: Vec, }, - /// Execute graph query language + /// Execute graph query language (experimental; prefer find/callers/relations/inventory) Gql { query: String, @@ -229,6 +230,172 @@ pub enum Commands { macro_name: Option, }, + /// Find symbols by name/type over the mmap graph index (no GQL) + Find { + /// Name or glob (`Foo`, `User*`, `*Service*`). Omit with `--type` to list by type. + #[arg(value_name = "PATTERN")] + pattern: Option, + + /// Filter by node type (function, class, import, annotation, …) + #[arg(short = 't', long = "type", value_name = "TYPE")] + type_name: Option, + + /// Path glob filter + #[arg(long = "file", value_name = "GLOB")] + file: Option, + + /// Language filter (java, rust, …) + #[arg(short = 'l', long = "lang", value_name = "LANG")] + lang: Option, + + /// qualified_name prefix scope + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + + /// Scope mode: inside (default), outside, crossing + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + + /// Alias for `--scope-mode outside` + #[arg(long = "exclude-scope")] + exclude_scope: bool, + + /// Exact name match (disable glob) + #[arg(long = "exact")] + exact: bool, + + /// Max results [default: 50] + #[arg(long = "limit", value_name = "N")] + limit: Option, + + /// Return counts only + #[arg(long = "count-only")] + count_only: bool, + }, + + /// Incoming CALLS neighbors for a symbol + Callers { + #[arg(value_name = "SYMBOL")] + symbol: String, + + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + + #[arg(long = "file", value_name = "PATH")] + file: Option, + + #[arg(long = "class", value_name = "NAME")] + class: Option, + + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + + #[arg(long = "exclude-scope")] + exclude_scope: bool, + + #[arg(long = "limit", value_name = "N")] + limit: Option, + }, + + /// Outgoing CALLS neighbors for a symbol + Callees { + #[arg(value_name = "SYMBOL")] + symbol: String, + + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + + #[arg(long = "file", value_name = "PATH")] + file: Option, + + #[arg(long = "class", value_name = "NAME")] + class: Option, + + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + + #[arg(long = "exclude-scope")] + exclude_scope: bool, + + #[arg(long = "limit", value_name = "N")] + limit: Option, + }, + + /// Typed edge traversal; omit SYMBOL for seedless set-wide scan + Relations { + /// Optional seed symbol (omit for seedless typed-edge scan) + #[arg(value_name = "SYMBOL")] + symbol: Option, + + /// Edge type (calls, uses, extends, implements, annotatedwith, …) + #[arg(short = 'e', long = "edge", value_name = "TYPE")] + edge: String, + + /// in | out | both [default: out] + #[arg(long = "direction", default_value = "out")] + direction: String, + + /// Filter edge source node type (seedless / seeded) + #[arg(long = "from-type", value_name = "TYPE")] + from_type: Option, + + /// Filter edge target node type + #[arg(long = "to-type", value_name = "TYPE")] + to_type: Option, + + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + + #[arg(long = "file", value_name = "PATH")] + file: Option, + + #[arg(long = "class", value_name = "NAME")] + class: Option, + + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + + #[arg(long = "exclude-scope")] + exclude_scope: bool, + + #[arg(long = "limit", value_name = "N")] + limit: Option, + }, + + /// Aggregate symbol/edge counts (includes zero-count schema kinds for type/edge) + Inventory { + /// Aggregation dimension: type | edge | lang | file | community + #[arg(long = "by", default_value = "type")] + by: String, + + #[arg(long = "file", value_name = "GLOB")] + file: Option, + + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + + #[arg(long = "exclude-scope")] + exclude_scope: bool, + }, + + /// Alias namespace for structured query verbs (`query find`, `query callers`, …) + Query { + #[command(subcommand)] + action: QueryCommands, + }, + /// Line-level program slice or taint trace Slice { file: String, @@ -502,6 +669,108 @@ pub enum Commands { }, } +#[derive(Subcommand)] +pub enum QueryCommands { + /// Alias for `rgctl find` + Find { + #[arg(value_name = "PATTERN")] + pattern: Option, + #[arg(short = 't', long = "type", value_name = "TYPE")] + type_name: Option, + #[arg(long = "file", value_name = "GLOB")] + file: Option, + #[arg(short = 'l', long = "lang", value_name = "LANG")] + lang: Option, + #[arg(long = "scope", value_name = "PREFIX")] + scope: Option, + #[arg(long = "scope-mode", value_name = "MODE")] + scope_mode: Option, + #[arg(long = "exclude-scope")] + exclude_scope: bool, + #[arg(long = "exact")] + exact: bool, + #[arg(long = "limit", value_name = "N")] + limit: Option, + #[arg(long = "count-only")] + count_only: bool, + }, + /// Alias for `rgctl callers` + Callers { + symbol: String, + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + #[arg(long = "file")] + file: Option, + #[arg(long = "class")] + class: Option, + #[arg(long = "scope")] + scope: Option, + #[arg(long = "scope-mode")] + scope_mode: Option, + #[arg(long = "exclude-scope")] + exclude_scope: bool, + #[arg(long = "limit")] + limit: Option, + }, + /// Alias for `rgctl callees` + Callees { + symbol: String, + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + #[arg(long = "file")] + file: Option, + #[arg(long = "class")] + class: Option, + #[arg(long = "scope")] + scope: Option, + #[arg(long = "scope-mode")] + scope_mode: Option, + #[arg(long = "exclude-scope")] + exclude_scope: bool, + #[arg(long = "limit")] + limit: Option, + }, + /// Alias for `rgctl relations` + Relations { + symbol: Option, + #[arg(short = 'e', long = "edge")] + edge: String, + #[arg(long = "direction", default_value = "out")] + direction: String, + #[arg(long = "from-type")] + from_type: Option, + #[arg(long = "to-type")] + to_type: Option, + #[arg(long = "depth", default_value_t = 1)] + depth: usize, + #[arg(long = "file")] + file: Option, + #[arg(long = "class")] + class: Option, + #[arg(long = "scope")] + scope: Option, + #[arg(long = "scope-mode")] + scope_mode: Option, + #[arg(long = "exclude-scope")] + exclude_scope: bool, + #[arg(long = "limit")] + limit: Option, + }, + /// Alias for `rgctl inventory` + Inventory { + #[arg(long = "by", default_value = "type")] + by: String, + #[arg(long = "file")] + file: Option, + #[arg(long = "scope")] + scope: Option, + #[arg(long = "scope-mode")] + scope_mode: Option, + #[arg(long = "exclude-scope")] + exclude_scope: bool, + }, +} + #[derive(Subcommand)] pub enum SemanticCommands { /// Build `.rgctl/semantic_index.bin` from function symbols (not run by default discover) @@ -890,6 +1159,258 @@ impl Cli { file, }, ), + Commands::Find { + pattern, + type_name, + file, + lang, + scope, + scope_mode, + exclude_scope, + exact, + limit, + count_only, + } => structured_query::run_find( + &ctx, + pattern, + type_name, + structured_query::SharedQueryArgs { + file, + class: None, + scope, + scope_mode, + exclude_scope, + lang, + limit, + }, + exact, + count_only, + ), + Commands::Callers { + symbol, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_call_neighbors( + &ctx, + symbol, + true, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + Commands::Callees { + symbol, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_call_neighbors( + &ctx, + symbol, + false, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + Commands::Relations { + symbol, + edge, + direction, + from_type, + to_type, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_relations( + &ctx, + symbol, + edge, + direction, + from_type, + to_type, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + Commands::Inventory { + by, + file, + scope, + scope_mode, + exclude_scope, + } => structured_query::run_inventory( + &ctx, + by, + structured_query::SharedQueryArgs { + file, + class: None, + scope, + scope_mode, + exclude_scope, + lang: None, + limit: None, + }, + ), + Commands::Query { action } => match action { + QueryCommands::Find { + pattern, + type_name, + file, + lang, + scope, + scope_mode, + exclude_scope, + exact, + limit, + count_only, + } => structured_query::run_find( + &ctx, + pattern, + type_name, + structured_query::SharedQueryArgs { + file, + class: None, + scope, + scope_mode, + exclude_scope, + lang, + limit, + }, + exact, + count_only, + ), + QueryCommands::Callers { + symbol, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_call_neighbors( + &ctx, + symbol, + true, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + QueryCommands::Callees { + symbol, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_call_neighbors( + &ctx, + symbol, + false, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + QueryCommands::Relations { + symbol, + edge, + direction, + from_type, + to_type, + depth, + file, + class, + scope, + scope_mode, + exclude_scope, + limit, + } => structured_query::run_relations( + &ctx, + symbol, + edge, + direction, + from_type, + to_type, + depth, + structured_query::SharedQueryArgs { + file, + class, + scope, + scope_mode, + exclude_scope, + lang: None, + limit, + }, + ), + QueryCommands::Inventory { + by, + file, + scope, + scope_mode, + exclude_scope, + } => structured_query::run_inventory( + &ctx, + by, + structured_query::SharedQueryArgs { + file, + class: None, + scope, + scope_mode, + exclude_scope, + lang: None, + limit: None, + }, + ), + }, Commands::Inspect { symbol, layer } => { inspect::run(&ctx, inspect::InspectArgs { symbol, layer }) } @@ -1192,6 +1713,18 @@ fn command_label_for(command: &Commands) -> &'static str { match command { Commands::Discover { .. } => "discover", Commands::Gql { .. } => "gql", + Commands::Find { .. } => "find", + Commands::Callers { .. } => "callers", + Commands::Callees { .. } => "callees", + Commands::Relations { .. } => "relations", + Commands::Inventory { .. } => "inventory", + Commands::Query { action } => match action { + QueryCommands::Find { .. } => "query find", + QueryCommands::Callers { .. } => "query callers", + QueryCommands::Callees { .. } => "query callees", + QueryCommands::Relations { .. } => "query relations", + QueryCommands::Inventory { .. } => "query inventory", + }, Commands::Slice { .. } => "slice", Commands::BlastRadius { .. } => "blast-radius", Commands::Inspect { .. } => "inspect", diff --git a/src/cli/structured_query.rs b/src/cli/structured_query.rs new file mode 100644 index 00000000..1a9cc69d --- /dev/null +++ b/src/cli/structured_query.rs @@ -0,0 +1,216 @@ +//! Structured query CLI: `find`, `callers`, `callees`, `relations`, `inventory`. + +use super::context::CliContext; +use super::OutputFormat; +use anyhow::{Context, Result}; +use rgctl_graph::{ + parse_edge_type, parse_node_type, InventoryBy, QueryFilters, RelationDirection, ScopeMode, + StructuredQuery, +}; + +/// Shared filter flags for structured query verbs. +#[derive(Debug, Clone, Default)] +pub struct SharedQueryArgs { + pub file: Option, + pub class: Option, + pub scope: Option, + pub scope_mode: Option, + pub exclude_scope: bool, + pub lang: Option, + pub limit: Option, +} + +impl SharedQueryArgs { + fn into_filters(self, node_type: Option) -> Result { + let scope_mode = if self.exclude_scope { + ScopeMode::Outside + } else if let Some(m) = self.scope_mode.as_deref() { + ScopeMode::parse(m).map_err(|e| anyhow::anyhow!("{e}"))? + } else { + ScopeMode::Inside + }; + Ok(QueryFilters { + node_type, + file_glob: self.file, + lang: self.lang, + scope: self.scope, + scope_mode, + class: self.class, + limit: self.limit, + count_only: false, + exact: false, + }) + } +} + +fn open_store(ctx: &CliContext) -> Result> { + ctx.open_snapshot_store()? + .context("Graph snapshot not found (run `rgctl discover` first)") +} + +fn emit_json(ctx: &CliContext, value: &T) -> Result<()> { + let v = serde_json::to_value(value)?; + ctx.emit_json_value(&v)?; + Ok(()) +} + +fn emit_text_entities(ctx: &CliContext, names: impl IntoIterator) -> Result<()> { + for name in names { + ctx.stdout_line(&name)?; + } + Ok(()) +} + +/// `rgctl find` +pub fn run_find( + ctx: &CliContext, + pattern: Option, + type_name: Option, + shared: SharedQueryArgs, + exact: bool, + count_only: bool, +) -> Result<()> { + let store = open_store(ctx)?; + let node_type = type_name + .as_deref() + .map(parse_node_type) + .transpose() + .map_err(|e| anyhow::anyhow!("{e}"))?; + let mut filters = shared.into_filters(node_type)?; + filters.exact = exact; + filters.count_only = count_only; + // Default limit for find when not counting + if filters.limit.is_none() && !count_only { + filters.limit = Some(50); + } + let q = StructuredQuery::new(store.as_ref()); + let result = q + .find(pattern.as_deref(), &filters) + .map_err(|e| anyhow::anyhow!("{e}"))?; + if ctx.format == OutputFormat::Json { + return emit_json(ctx, &result); + } + if count_only { + ctx.stdout_line(&format!("{}", result.total))?; + } else { + emit_text_entities(ctx, result.entities.into_iter().map(|e| e.name))?; + } + Ok(()) +} + +/// `rgctl callers` / `callees` +pub fn run_call_neighbors( + ctx: &CliContext, + symbol: String, + incoming: bool, + depth: usize, + shared: SharedQueryArgs, +) -> Result<()> { + let store = open_store(ctx)?; + let mut filters = shared.into_filters(None)?; + if filters.limit.is_none() { + filters.limit = Some(50); + } + let q = StructuredQuery::new(store.as_ref()); + let result = q + .call_neighbors(&symbol, incoming, depth, &filters) + .map_err(|e| anyhow::anyhow!("{e}"))?; + if ctx.format == OutputFormat::Json { + // Shape as callers/callees field name for agents + let mut v = serde_json::to_value(&result)?; + if let Some(obj) = v.as_object_mut() { + let key = if incoming { "callers" } else { "callees" }; + if let Some(n) = obj.remove("neighbors") { + obj.insert(key.into(), n); + } + } + return ctx.emit_json_value(&v); + } + emit_text_entities(ctx, result.neighbors.into_iter().map(|e| e.name))?; + Ok(()) +} + +/// `rgctl relations` +pub fn run_relations( + ctx: &CliContext, + symbol: Option, + edge: String, + direction: String, + from_type: Option, + to_type: Option, + depth: usize, + shared: SharedQueryArgs, +) -> Result<()> { + let store = open_store(ctx)?; + let edge_ty = parse_edge_type(&edge).map_err(|e| anyhow::anyhow!("{e}"))?; + let dir = RelationDirection::parse(&direction).map_err(|e| anyhow::anyhow!("{e}"))?; + let from_ty = from_type + .as_deref() + .map(parse_node_type) + .transpose() + .map_err(|e| anyhow::anyhow!("{e}"))?; + let to_ty = to_type + .as_deref() + .map(parse_node_type) + .transpose() + .map_err(|e| anyhow::anyhow!("{e}"))?; + let mut filters = shared.into_filters(None)?; + if filters.limit.is_none() { + filters.limit = Some(50); + } + if symbol.is_none() && from_ty.is_none() && to_ty.is_none() && filters.scope.is_none() { + // Allow but warn via stderr — still run (seedless full scan). + eprintln!("note: seedless relations without --from-type/--to-type/--scope may return a large edge set"); + } + let q = StructuredQuery::new(store.as_ref()); + let result = q + .relations( + symbol.as_deref(), + edge_ty, + dir, + from_ty, + to_ty, + depth, + &filters, + ) + .map_err(|e| anyhow::anyhow!("{e}"))?; + if ctx.format == OutputFormat::Json { + return emit_json(ctx, &result); + } + for e in result.edges { + ctx.stdout_line(&format!( + "{} -[{}]-> {}", + e.source.name, e.edge, e.target.name + ))?; + } + Ok(()) +} + +/// `rgctl inventory` +pub fn run_inventory( + ctx: &CliContext, + by: String, + shared: SharedQueryArgs, +) -> Result<()> { + let store = open_store(ctx)?; + let dim = InventoryBy::parse(&by).map_err(|e| anyhow::anyhow!("{e}"))?; + let filters = shared.into_filters(None)?; + let q = StructuredQuery::new(store.as_ref()); + let result = q + .inventory(dim, &filters) + .map_err(|e| anyhow::anyhow!("{e}"))?; + if ctx.format == OutputFormat::Json { + return emit_json(ctx, &result); + } + for c in result.counts { + ctx.stdout_line(&format!("{}\t{}", c.key, c.count))?; + } + Ok(()) +} + +/// `rgctl query …` alias — argv after `query` is re-parsed by clap parent; this helper +/// is unused when parent dispatches directly. Kept for documentation of serve offload. +#[allow(dead_code)] +pub fn serve_offload_contract() -> &'static str { + StructuredQuery::serve_offload_note() +} From 7782008b490bcf795308d7a2cacff46ead45366b Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:36:35 +0200 Subject: [PATCH 08/18] fix counts and occurances Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- crates/rgctl-analysis/src/cpg.rs | 1 + crates/rgctl-analysis/src/macro_call_index.rs | 1 + .../rgctl-analysis/src/macro_call_lookup.rs | 2 + crates/rgctl-error/Cargo.toml | 1 + crates/rgctl-error/src/lib.rs | 33 +- crates/rgctl-graph/src/structured_query.rs | 313 +++++++++++++++--- .../rgctl/references/command-encyclopedia.md | 10 +- src/cli/discover_impl.rs | 4 +- src/cli/inspect.rs | 6 +- src/cli/mod.rs | 34 ++ src/cli/structured_query.rs | 67 +++- 11 files changed, 410 insertions(+), 62 deletions(-) diff --git a/crates/rgctl-analysis/src/cpg.rs b/crates/rgctl-analysis/src/cpg.rs index e44f0c98..3bd77a41 100644 --- a/crates/rgctl-analysis/src/cpg.rs +++ b/crates/rgctl-analysis/src/cpg.rs @@ -196,6 +196,7 @@ fn require_unique_function(backend: &MemoryBackend, symbol: &str) -> Result Err(Error::AmbiguousSymbol { name: symbol.to_string(), count: many.len(), + candidates: vec![], }), } } diff --git a/crates/rgctl-analysis/src/macro_call_index.rs b/crates/rgctl-analysis/src/macro_call_index.rs index 1bade713..a9c840bc 100644 --- a/crates/rgctl-analysis/src/macro_call_index.rs +++ b/crates/rgctl-analysis/src/macro_call_index.rs @@ -381,6 +381,7 @@ impl MacroCallIndex { count => Err(rgctl_error::Error::AmbiguousSymbol { name: symbol.to_string(), count, + candidates: vec![], }), } } diff --git a/crates/rgctl-analysis/src/macro_call_lookup.rs b/crates/rgctl-analysis/src/macro_call_lookup.rs index 27d0890c..ae0bf9be 100644 --- a/crates/rgctl-analysis/src/macro_call_lookup.rs +++ b/crates/rgctl-analysis/src/macro_call_lookup.rs @@ -251,6 +251,7 @@ pub fn resolve_symbol_uuid(candidates: &[MacroIndexEntry], parsed: &ParsedSymbol Err(Error::AmbiguousSymbol { name: parsed.target_name.clone(), count, + candidates: vec![], }) } } @@ -806,6 +807,7 @@ impl MacroCallLookupDb { count => Err(Error::AmbiguousSymbol { name: symbol.to_string(), count, + candidates: vec![], }), } } diff --git a/crates/rgctl-error/Cargo.toml b/crates/rgctl-error/Cargo.toml index 6a1cc6eb..9c982823 100644 --- a/crates/rgctl-error/Cargo.toml +++ b/crates/rgctl-error/Cargo.toml @@ -8,6 +8,7 @@ license = "MIT OR Apache-2.0" [dependencies] rgctl-plugin-api = { path = "../rgctl-plugin-api" } +serde = { version = "1", features = ["derive"] } serde_json = "1" serde_yaml = "0.9" thiserror = "1" diff --git a/crates/rgctl-error/src/lib.rs b/crates/rgctl-error/src/lib.rs index 985b4e0a..2fed1fea 100644 --- a/crates/rgctl-error/src/lib.rs +++ b/crates/rgctl-error/src/lib.rs @@ -4,8 +4,30 @@ //! We use `thiserror` for ergonomic error definitions with automatic trait implementations. use std::path::PathBuf; +use serde::Serialize; use thiserror::Error; +/// Lean candidate row when a symbol name matches multiple nodes. +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +pub struct SymbolCandidate { + /// Node UUID + pub id: String, + /// Bare name + pub name: String, + /// Fully qualified name when present + #[serde(skip_serializing_if = "Option::is_none")] + pub qualified_name: Option, + /// Node type (lowercase CLI form) + #[serde(rename = "type")] + pub node_type: String, + /// Source file + #[serde(skip_serializing_if = "Option::is_none")] + pub file: Option, + /// Definition line + #[serde(skip_serializing_if = "Option::is_none")] + pub line: Option, +} + /// Main error type for rgctl operations #[derive(Error, Debug)] pub enum Error { @@ -73,9 +95,16 @@ pub enum Error { #[error("Resource not found: {0}")] NotFound(String), - /// Symbol name matched multiple graph nodes + /// Symbol name matched multiple graph nodes. + /// + /// `candidates` carries lean rows for agent recovery (may be empty for legacy call sites). + /// Prefer emitting them under `-f json` as an `ambiguous_symbol` envelope. #[error("Ambiguous symbol '{name}': {count} matches")] - AmbiguousSymbol { name: String, count: usize }, + AmbiguousSymbol { + name: String, + count: usize, + candidates: Vec, + }, /// Generic error with context #[error("{0}")] diff --git a/crates/rgctl-graph/src/structured_query.rs b/crates/rgctl-graph/src/structured_query.rs index 6242870d..63f8029a 100644 --- a/crates/rgctl-graph/src/structured_query.rs +++ b/crates/rgctl-graph/src/structured_query.rs @@ -16,7 +16,9 @@ use std::sync::Arc; use uuid::Uuid; /// JSON schema version for structured-query envelopes. -pub const STRUCTURED_QUERY_SCHEMA_VERSION: u32 = 1; +/// +/// v2: relations rows carry `occurrences`; `total` is distinct `(source,target,edge)` count. +pub const STRUCTURED_QUERY_SCHEMA_VERSION: u32 = 2; /// All [`NodeType`] variants for inventory zero-count emission. pub const ALL_NODE_TYPES: &[NodeType] = &[ @@ -157,6 +159,8 @@ pub struct EdgeRow { /// Hop distance from seed (1 for seedless scans) #[serde(skip_serializing_if = "Option::is_none")] pub hops: Option, + /// How many stored edges collapsed into this distinct relationship + pub occurrences: usize, } /// Find / list result envelope. @@ -226,8 +230,13 @@ pub struct InventoryResult { pub struct InventoryCount { /// Bucket key (type/edge/lang/file/community) pub key: String, - /// Count (may be zero) + /// Primary count. For `--by edge`: distinct `(source,target)` relationships + /// (aligned with `relations.total`). For other dimensions: entity count. pub count: usize, + /// Raw stored edge instances when `count` is distinct (`--by edge` only). + /// Omitted for non-edge inventories. + #[serde(skip_serializing_if = "Option::is_none")] + pub occurrences: Option, } /// Filters shared by find / resolve. @@ -245,6 +254,8 @@ pub struct QueryFilters { pub scope_mode: ScopeMode, /// Enclosing class/type name filter pub class: Option, + /// Definition start line (disambiguates same-class overloads) + pub line: Option, /// Max rows (None = unbounded) pub limit: Option, /// Count only @@ -377,6 +388,11 @@ impl<'a> StructuredQuery<'a> { if !Self::matches_class(node, f.class.as_deref()) { return false; } + if let Some(want_line) = f.line + && node.start_line != Some(want_line) + { + return false; + } true } @@ -401,10 +417,25 @@ impl<'a> StructuredQuery<'a> { match matches.len() { 0 => Err(Error::NodeNotFound(symbol.to_string())), 1 => Ok(matches.remove(0)), - n => Err(Error::AmbiguousSymbol { - name: symbol.to_string(), - count: n, - }), + n => { + let candidates = matches + .into_iter() + .take(50) + .map(|node| rgctl_error::SymbolCandidate { + id: node.id.to_string(), + name: node.name.to_string(), + qualified_name: node.qualified_name.as_ref().map(|s| s.to_string()), + node_type: node_type_cli(node.node_type), + file: node.file_path.as_ref().map(|s| s.to_string()), + line: node.start_line, + }) + .collect(); + Err(Error::AmbiguousSymbol { + name: symbol.to_string(), + count: n, + candidates, + }) + } } } @@ -607,9 +638,8 @@ impl<'a> StructuredQuery<'a> { to_type: Option, filters: &QueryFilters, ) -> Result { - let mut edges = Vec::new(); - let mut total = 0usize; - let limit = filters.limit.unwrap_or(usize::MAX); + // Aggregate duplicate stored edges into distinct (source,target) with occurrences. + let mut agg: HashMap<(Uuid, Uuid), EdgeRow> = HashMap::new(); self.store.for_each_edge(|from, to, et| { if et != edge { return Ok(()); @@ -633,32 +663,45 @@ impl<'a> StructuredQuery<'a> { if !Self::edge_scope_ok(&src, &dst, filters.scope.as_deref(), filters.scope_mode) { return Ok(()); } - // Emit according to direction (both = outbound orientation as stored). - let emit = match direction { - RelationDirection::Out | RelationDirection::Both => true, - RelationDirection::In => true, // still emit keyed as stored; direction field notes "in" view + let (source, target, dir_label, key) = match direction { + RelationDirection::In => ( + Self::project(&dst), + Self::project(&src), + "in", + (to, from), + ), + RelationDirection::Out | RelationDirection::Both => ( + Self::project(&src), + Self::project(&dst), + "out", + (from, to), + ), }; - if !emit { - return Ok(()); - } - total += 1; - if edges.len() < limit { - let (source, target, dir_label) = match direction { - RelationDirection::In => (Self::project(&dst), Self::project(&src), "in"), - RelationDirection::Out | RelationDirection::Both => { - (Self::project(&src), Self::project(&dst), "out") - } - }; - edges.push(EdgeRow { + agg.entry(key) + .and_modify(|row| row.occurrences = row.occurrences.saturating_add(1)) + .or_insert(EdgeRow { source, edge: edge_type_cli(edge), direction: dir_label.into(), target, hops: Some(1), + occurrences: 1, }); - } Ok(()) })?; + let total = agg.len(); + let limit = filters.limit.unwrap_or(usize::MAX); + let mut edges: Vec = agg.into_values().collect(); + // Stable order for agents / goldens: source name, then target name. + edges.sort_by(|a, b| { + (&a.source.name, &a.target.name, &a.source.id, &a.target.id).cmp(&( + &b.source.name, + &b.target.name, + &b.source.id, + &b.target.id, + )) + }); + edges.truncate(limit); Ok(RelationsResult { schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, target: None, @@ -681,11 +724,9 @@ impl<'a> StructuredQuery<'a> { let seed = self.resolve_symbol(symbol, filters)?; let depth = depth.max(1); let adj = self.build_typed_adjacency(edge)?; - let mut rows = Vec::new(); - let mut total = 0usize; - let limit = filters.limit.unwrap_or(usize::MAX); + let mut agg: HashMap<(Uuid, Uuid, bool), EdgeRow> = HashMap::new(); - let walk = |incoming: bool, rows: &mut Vec, total: &mut usize| -> Result<()> { + let walk = |incoming: bool, agg: &mut HashMap<(Uuid, Uuid, bool), EdgeRow>| -> Result<()> { let mut seen = HashSet::from([seed.id]); let mut q = VecDeque::from([(seed.id, 0usize)]); while let Some((id, d)) = q.pop_front() { @@ -721,16 +762,25 @@ impl<'a> StructuredQuery<'a> { { continue; } - *total += 1; - if rows.len() < limit { - rows.push(EdgeRow { + let key = (src_id, dst_id, incoming); + agg.entry(key) + .and_modify(|row| { + row.occurrences = row.occurrences.saturating_add(1); + // Keep the shortest hop when collapsing duplicates. + if let (Some(h), Some(prev)) = (Some(hop), row.hops) { + if h < prev { + row.hops = Some(h); + } + } + }) + .or_insert(EdgeRow { source: Self::project(&src), edge: edge_type_cli(edge), direction: if incoming { "in" } else { "out" }.into(), target: Self::project(&dst), hops: Some(hop), + occurrences: 1, }); - } if seen.insert(nid) && hop < depth { q.push_back((nid, hop)); } @@ -740,14 +790,27 @@ impl<'a> StructuredQuery<'a> { }; match direction { - RelationDirection::Out => walk(false, &mut rows, &mut total)?, - RelationDirection::In => walk(true, &mut rows, &mut total)?, + RelationDirection::Out => walk(false, &mut agg)?, + RelationDirection::In => walk(true, &mut agg)?, RelationDirection::Both => { - walk(false, &mut rows, &mut total)?; - walk(true, &mut rows, &mut total)?; + walk(false, &mut agg)?; + walk(true, &mut agg)?; } } + let total = agg.len(); + let limit = filters.limit.unwrap_or(usize::MAX); + let mut rows: Vec = agg.into_values().collect(); + rows.sort_by(|a, b| { + (&a.source.name, &a.target.name, &a.source.id, &a.target.id).cmp(&( + &b.source.name, + &b.target.name, + &b.source.id, + &b.target.id, + )) + }); + rows.truncate(limit); + Ok(RelationsResult { schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, target: Some(Self::project(&seed)), @@ -781,6 +844,7 @@ impl<'a> StructuredQuery<'a> { .map(|t| InventoryCount { key: node_type_cli(*t), count: *map.get(t).unwrap_or(&0), + occurrences: None, }) .collect(); // Include any unexpected types not in ALL_NODE_TYPES. @@ -789,6 +853,7 @@ impl<'a> StructuredQuery<'a> { counts.push(InventoryCount { key: node_type_cli(t), count: c, + occurrences: None, }); } } @@ -799,7 +864,12 @@ impl<'a> StructuredQuery<'a> { }) } InventoryBy::Edge => { - let mut map: HashMap = + // `count` = distinct (from,to) pairs; `occurrences` = raw stored edges. + let mut distinct: HashMap> = ALL_EDGE_TYPES + .iter() + .map(|t| (*t, HashSet::new())) + .collect(); + let mut occurrences: HashMap = ALL_EDGE_TYPES.iter().map(|t| (*t, 0usize)).collect(); self.store.for_each_edge(|from, to, et| { if et == EdgeType::Unknown { @@ -821,14 +891,20 @@ impl<'a> StructuredQuery<'a> { return Ok(()); } } - *map.entry(et).or_insert(0) += 1; + *occurrences.entry(et).or_insert(0) += 1; + distinct.entry(et).or_default().insert((from, to)); Ok(()) })?; let counts = ALL_EDGE_TYPES .iter() - .map(|t| InventoryCount { - key: edge_type_cli(*t), - count: *map.get(t).unwrap_or(&0), + .map(|t| { + let occ = *occurrences.get(t).unwrap_or(&0); + let dist = distinct.get(t).map(|s| s.len()).unwrap_or(0); + InventoryCount { + key: edge_type_cli(*t), + count: dist, + occurrences: Some(occ), + } }) .collect(); Ok(InventoryResult { @@ -861,7 +937,11 @@ impl<'a> StructuredQuery<'a> { } let mut counts: Vec<_> = map .into_iter() - .map(|(key, count)| InventoryCount { key, count }) + .map(|(key, count)| InventoryCount { + key, + count, + occurrences: None, + }) .collect(); counts.sort_by(|a, b| a.key.cmp(&b.key)); Ok(InventoryResult { @@ -888,7 +968,11 @@ impl<'a> StructuredQuery<'a> { } let mut counts: Vec<_> = map .into_iter() - .map(|(key, count)| InventoryCount { key, count }) + .map(|(key, count)| InventoryCount { + key, + count, + occurrences: None, + }) .collect(); counts.sort_by(|a, b| a.key.cmp(&b.key)); Ok(InventoryResult { @@ -914,7 +998,11 @@ impl<'a> StructuredQuery<'a> { } let mut counts: Vec<_> = map .into_iter() - .map(|(key, count)| InventoryCount { key, count }) + .map(|(key, count)| InventoryCount { + key, + count, + occurrences: None, + }) .collect(); counts.sort_by(|a, b| a.key.cmp(&b.key)); Ok(InventoryResult { @@ -1298,6 +1386,139 @@ mod tests { assert_eq!(res.neighbors[0].name, "handle"); } + #[test] + fn seedless_calls_dedupes_with_occurrences() { + let dir = tempdir().unwrap(); + let path = dir.path().join("graph.snapshot.bin"); + let mut backend = MemoryBackend::new(); + let a = Node::new(NodeType::Function, "assumeNotEmpty") + .with_file_path("Check.java") + .with_location(290, 300); + let b = Node::new(NodeType::Function, "assume") + .with_file_path("Check.java") + .with_location(138, 150); + let a_id = a.id; + let b_id = b.id; + backend.insert_node(a).unwrap(); + backend.insert_node(b).unwrap(); + for _ in 0..4 { + backend + .insert_edge(Edge::new(a_id, b_id, EdgeType::Calls)) + .unwrap(); + } + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + let q = StructuredQuery::new(&store); + let res = q + .relations( + None, + EdgeType::Calls, + RelationDirection::Out, + None, + None, + 1, + &QueryFilters::default(), + ) + .unwrap(); + assert_eq!(res.total, 1, "distinct relationships"); + assert_eq!(res.returned, 1); + assert_eq!(res.edges[0].occurrences, 4); + assert_eq!(res.schema_version, STRUCTURED_QUERY_SCHEMA_VERSION); + assert_eq!(res.edges[0].source.name, "assumeNotEmpty"); + assert_eq!(res.edges[0].target.name, "assume"); + } + + #[test] + fn ambiguous_symbol_includes_candidates() { + let dir = tempdir().unwrap(); + let path = dir.path().join("graph.snapshot.bin"); + let mut backend = MemoryBackend::new(); + let a = Node::new(NodeType::Function, "assume") + .with_file_path("A.java") + .with_location(10, 20); + let b = Node::new(NodeType::Function, "assume") + .with_file_path("B.java") + .with_location(30, 40); + backend.insert_node(a).unwrap(); + backend.insert_node(b).unwrap(); + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + let q = StructuredQuery::new(&store); + let err = q + .resolve_symbol("assume", &QueryFilters::default()) + .unwrap_err(); + match err { + Error::AmbiguousSymbol { + count, + candidates, + .. + } => { + assert_eq!(count, 2); + assert_eq!(candidates.len(), 2); + assert!(candidates.iter().any(|c| c.file.as_deref() == Some("A.java"))); + } + other => panic!("expected AmbiguousSymbol, got {other:?}"), + } + let ok = q + .resolve_symbol( + "assume", + &QueryFilters { + line: Some(30), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(ok.start_line, Some(30)); + } + + #[test] + fn inventory_edge_distinct_matches_relations_total() { + let dir = tempdir().unwrap(); + let path = dir.path().join("graph.snapshot.bin"); + let mut backend = MemoryBackend::new(); + let a = Node::new(NodeType::Function, "assumeNotEmpty") + .with_file_path("Check.java") + .with_location(290, 300); + let b = Node::new(NodeType::Function, "assume") + .with_file_path("Check.java") + .with_location(138, 150); + let a_id = a.id; + let b_id = b.id; + backend.insert_node(a).unwrap(); + backend.insert_node(b).unwrap(); + for _ in 0..4 { + backend + .insert_edge(Edge::new(a_id, b_id, EdgeType::Calls)) + .unwrap(); + } + write_columnar_from_backend(&backend, &path).unwrap(); + let store = SnapshotNodeStore::open(&path).unwrap(); + let q = StructuredQuery::new(&store); + let inv = q + .inventory(InventoryBy::Edge, &QueryFilters::default()) + .unwrap(); + let calls = inv + .counts + .iter() + .find(|c| c.key == "calls") + .expect("calls bucket"); + assert_eq!(calls.count, 1, "distinct relationships"); + assert_eq!(calls.occurrences, Some(4)); + let rel = q + .relations( + None, + EdgeType::Calls, + RelationDirection::Out, + None, + None, + 1, + &QueryFilters::default(), + ) + .unwrap(); + assert_eq!(rel.total, calls.count); + assert_eq!(rel.edges[0].occurrences, calls.occurrences.unwrap()); + } + #[test] fn glob_contains() { assert!(glob_match("*Service*", "PRTRestServiceRoute")); diff --git a/skills/rgctl/references/command-encyclopedia.md b/skills/rgctl/references/command-encyclopedia.md index 540edd01..53c1b840 100644 --- a/skills/rgctl/references/command-encyclopedia.md +++ b/skills/rgctl/references/command-encyclopedia.md @@ -107,7 +107,7 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json ```bash rgctl -f json find [PATTERN] --type function --scope pkg --limit 50 rgctl -f json find --type function --count-only -rgctl -f json callers --depth 1 --file PATH --class NAME +rgctl -f json callers --depth 1 --file PATH --class NAME --line N rgctl -f json callees --depth 1 rgctl -f json relations [SYMBOL] --edge annotatedwith --from-type function --to-type annotation --scope pkg rgctl -f json relations --edge extends --from-type class # seedless @@ -116,17 +116,17 @@ rgctl -f json inventory --by edge rgctl -f json query find … # alias namespace ``` -**Purpose:** Deterministic mmap structured query (no Cypher, no `MemoryBackend` hydrate). Prefer these over `gql` for agent work. +**Purpose:** Deterministic mmap structured query (no Cypher, no `MemoryBackend` hydrate). Prefer these over `gql` for agent work. Relations `total` is distinct `(source,target,edge)`; duplicates collapse with `occurrences` (`schema_version` ≥ 2). `inventory --by edge` uses the same rule: `count` = distinct, `occurrences` = raw stored edges. **Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). -**Flags:** `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. +**Flags:** `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). `--file` / `--class` / `--line` disambiguate. Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. -**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. +**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. **Agent should report:** counts, lean names/files, keyed edge pairs — not full node dumps. -**See:** OpenSpec `add-structured-query-cli`; demote GQL only after latency + probe-catalog gates. +**See:** OpenSpec `add-structured-query-cli`; GQL demotion is a follow-up change (latency + cardinality gates cleared). --- diff --git a/src/cli/discover_impl.rs b/src/cli/discover_impl.rs index 87f5bcf3..977f853f 100644 --- a/src/cli/discover_impl.rs +++ b/src/cli/discover_impl.rs @@ -1280,7 +1280,9 @@ pub(crate) fn run_full_analysis( info!(""); info!("[i] Next steps:"); - info!(" rgctl gql \"MATCH (n:Function) RETURN n\" # Query the graph"); + info!(" rgctl -f json find --type function --limit 20"); + info!(" rgctl -f json inventory --by type"); + info!(" rgctl -f json relations --edge calls --limit 20"); info!(" rgctl slice --line --variable "); if dashboard_dir.join("manifest.json").is_file() { info!(" rgctl serve --open # Dashboard + query API at http://127.0.0.1:8080"); diff --git a/src/cli/inspect.rs b/src/cli/inspect.rs index cdac2cd1..bf9f5c70 100644 --- a/src/cli/inspect.rs +++ b/src/cli/inspect.rs @@ -147,7 +147,11 @@ fn resolve_symbol_function( let source = fs::read_to_string(&file)?; return Ok((node, source)); } - Err(rgctl_error::Error::AmbiguousSymbol { name, count }) => { + Err(rgctl_error::Error::AmbiguousSymbol { + name, + count, + candidates: _, + }) => { anyhow::bail!( "Symbol '{name}' is ambiguous. Found {count} matches. \ Refine with path syntax: rgctl inspect \"path/to/file.ts::{name}\" cfg" diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 608a1799..af801f33 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -287,6 +287,10 @@ pub enum Commands { #[arg(long = "class", value_name = "NAME")] class: Option, + /// Definition line (disambiguates same-name overloads) + #[arg(long = "line", value_name = "N")] + line: Option, + #[arg(long = "scope", value_name = "PREFIX")] scope: Option, @@ -314,6 +318,10 @@ pub enum Commands { #[arg(long = "class", value_name = "NAME")] class: Option, + /// Definition line (disambiguates same-name overloads) + #[arg(long = "line", value_name = "N")] + line: Option, + #[arg(long = "scope", value_name = "PREFIX")] scope: Option, @@ -358,6 +366,10 @@ pub enum Commands { #[arg(long = "class", value_name = "NAME")] class: Option, + /// Definition line (disambiguates same-name overloads) + #[arg(long = "line", value_name = "N")] + line: Option, + #[arg(long = "scope", value_name = "PREFIX")] scope: Option, @@ -703,6 +715,8 @@ pub enum QueryCommands { file: Option, #[arg(long = "class")] class: Option, + #[arg(long = "line", value_name = "N")] + line: Option, #[arg(long = "scope")] scope: Option, #[arg(long = "scope-mode")] @@ -721,6 +735,8 @@ pub enum QueryCommands { file: Option, #[arg(long = "class")] class: Option, + #[arg(long = "line", value_name = "N")] + line: Option, #[arg(long = "scope")] scope: Option, #[arg(long = "scope-mode")] @@ -747,6 +763,8 @@ pub enum QueryCommands { file: Option, #[arg(long = "class")] class: Option, + #[arg(long = "line", value_name = "N")] + line: Option, #[arg(long = "scope")] scope: Option, #[arg(long = "scope-mode")] @@ -1177,6 +1195,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class: None, + line: None, scope, scope_mode, exclude_scope, @@ -1191,6 +1210,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1203,6 +1223,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1215,6 +1236,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1227,6 +1249,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1243,6 +1266,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1258,6 +1282,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1277,6 +1302,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class: None, + line: None, scope, scope_mode, exclude_scope, @@ -1303,6 +1329,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class: None, + line: None, scope, scope_mode, exclude_scope, @@ -1317,6 +1344,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1329,6 +1357,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1341,6 +1370,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1353,6 +1383,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1369,6 +1400,7 @@ impl Cli { depth, file, class, + line, scope, scope_mode, exclude_scope, @@ -1384,6 +1416,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class, + line, scope, scope_mode, exclude_scope, @@ -1403,6 +1436,7 @@ impl Cli { structured_query::SharedQueryArgs { file, class: None, + line: None, scope, scope_mode, exclude_scope, diff --git a/src/cli/structured_query.rs b/src/cli/structured_query.rs index 1a9cc69d..d49fbfa0 100644 --- a/src/cli/structured_query.rs +++ b/src/cli/structured_query.rs @@ -3,9 +3,10 @@ use super::context::CliContext; use super::OutputFormat; use anyhow::{Context, Result}; +use rgctl_error::Error as GraphError; use rgctl_graph::{ parse_edge_type, parse_node_type, InventoryBy, QueryFilters, RelationDirection, ScopeMode, - StructuredQuery, + STRUCTURED_QUERY_SCHEMA_VERSION, StructuredQuery, }; /// Shared filter flags for structured query verbs. @@ -13,6 +14,7 @@ use rgctl_graph::{ pub struct SharedQueryArgs { pub file: Option, pub class: Option, + pub line: Option, pub scope: Option, pub scope_mode: Option, pub exclude_scope: bool, @@ -36,6 +38,7 @@ impl SharedQueryArgs { scope: self.scope, scope_mode, class: self.class, + line: self.line, limit: self.limit, count_only: false, exact: false, @@ -61,6 +64,43 @@ fn emit_text_entities(ctx: &CliContext, names: impl IntoIterator) Ok(()) } +/// Map graph errors; under `-f json`, emit an `ambiguous_symbol` envelope on stdout. +fn map_sq_err(ctx: &CliContext, err: GraphError) -> anyhow::Error { + if let GraphError::AmbiguousSymbol { + name, + count, + candidates, + } = &err + { + if ctx.format == OutputFormat::Json { + let envelope = serde_json::json!({ + "schema_version": STRUCTURED_QUERY_SCHEMA_VERSION, + "error": "ambiguous_symbol", + "name": name, + "count": count, + "candidates": candidates, + }); + let _ = ctx.emit_json_value(&envelope); + } else if !candidates.is_empty() { + eprintln!("Ambiguous symbol '{name}': {count} matches. Candidates:"); + for c in candidates.iter().take(20) { + let file = c.file.as_deref().unwrap_or("?"); + let line = c + .line + .map(|n| n.to_string()) + .unwrap_or_else(|| "?".into()); + let qn = c.qualified_name.as_deref().unwrap_or(""); + eprintln!( + " - id={} type={} file={file}:{line} name={} {qn}", + c.id, c.node_type, c.name + ); + } + eprintln!("Disambiguate with --file, --class, and/or --line."); + } + } + anyhow::anyhow!("{err}") +} + /// `rgctl find` pub fn run_find( ctx: &CliContext, @@ -86,7 +126,7 @@ pub fn run_find( let q = StructuredQuery::new(store.as_ref()); let result = q .find(pattern.as_deref(), &filters) - .map_err(|e| anyhow::anyhow!("{e}"))?; + .map_err(|e| map_sq_err(ctx, e))?; if ctx.format == OutputFormat::Json { return emit_json(ctx, &result); } @@ -114,7 +154,7 @@ pub fn run_call_neighbors( let q = StructuredQuery::new(store.as_ref()); let result = q .call_neighbors(&symbol, incoming, depth, &filters) - .map_err(|e| anyhow::anyhow!("{e}"))?; + .map_err(|e| map_sq_err(ctx, e))?; if ctx.format == OutputFormat::Json { // Shape as callers/callees field name for agents let mut v = serde_json::to_value(&result)?; @@ -173,13 +213,18 @@ pub fn run_relations( depth, &filters, ) - .map_err(|e| anyhow::anyhow!("{e}"))?; + .map_err(|e| map_sq_err(ctx, e))?; if ctx.format == OutputFormat::Json { return emit_json(ctx, &result); } for e in result.edges { + let occ = if e.occurrences > 1 { + format!(" x{}", e.occurrences) + } else { + String::new() + }; ctx.stdout_line(&format!( - "{} -[{}]-> {}", + "{} -[{}]-> {}{occ}", e.source.name, e.edge, e.target.name ))?; } @@ -198,12 +243,20 @@ pub fn run_inventory( let q = StructuredQuery::new(store.as_ref()); let result = q .inventory(dim, &filters) - .map_err(|e| anyhow::anyhow!("{e}"))?; + .map_err(|e| map_sq_err(ctx, e))?; if ctx.format == OutputFormat::Json { return emit_json(ctx, &result); } for c in result.counts { - ctx.stdout_line(&format!("{}\t{}", c.key, c.count))?; + if let Some(occ) = c.occurrences { + if occ != c.count { + ctx.stdout_line(&format!("{}\t{}\toccurrences={}", c.key, c.count, occ))?; + } else { + ctx.stdout_line(&format!("{}\t{}", c.key, c.count))?; + } + } else { + ctx.stdout_line(&format!("{}\t{}", c.key, c.count))?; + } } Ok(()) } From c27ece4d108d7153baad14eb978ac08fb489c8cc Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Fri, 2 Oct 2026 14:42:31 +0200 Subject: [PATCH 09/18] =?UTF-8?q?=20=20=E2=80=A2=20find=20--annotation=20@?= =?UTF-8?q?Foo[,@Bar=E2=80=A6]=20(+=20honest=20--show-attributes=20error)?= =?UTF-8?q?=20=20=20=E2=80=A2=20inventory=20--by=20import-prefix=20=20=20?= =?UTF-8?q?=E2=80=A2=20rgctl=20status=20(digest,=20nodes/edges,=20kantra?= =?UTF-8?q?=20findings=20hint)=20=20=20=E2=80=A2=20rgctl=20resources=20(pe?= =?UTF-8?q?rsistence=20/=20weblogic=20/=20jboss=20/=20web=20/=20beans)=20?= =?UTF-8?q?=20=20=E2=80=A2=20rgctl=20rules=20run=20=20(--target,=20--?= =?UTF-8?q?index-only,=20--catalog)=20=20=20=E2=80=A2=20Skills=20+=20encyc?= =?UTF-8?q?lopedia=20migration=20probe=20recipe?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- crates/rgctl-graph/src/lib.rs | 4 +- crates/rgctl-graph/src/structured_query.rs | 344 +++++++++++++++- skills/rgctl/SKILL.md | 15 +- .../rgctl/references/command-encyclopedia.md | 24 +- skills/rgctl/workflows/kantra.md | 3 + src/cli/mod.rs | 76 +++- src/cli/resources.rs | 367 ++++++++++++++++++ src/cli/rules.rs | 192 +++++++++ src/cli/session_status.rs | 108 ++++++ src/cli/structured_query.rs | 12 + 10 files changed, 1132 insertions(+), 13 deletions(-) create mode 100644 src/cli/resources.rs create mode 100644 src/cli/rules.rs create mode 100644 src/cli/session_status.rs diff --git a/crates/rgctl-graph/src/lib.rs b/crates/rgctl-graph/src/lib.rs index 854dbbce..dcf8eaea 100644 --- a/crates/rgctl-graph/src/lib.rs +++ b/crates/rgctl-graph/src/lib.rs @@ -72,8 +72,8 @@ pub use snapshot_diff::{ pub use structured_query::{ ALL_EDGE_TYPES, ALL_NODE_TYPES, CallNeighborsResult, EntityRow, EdgeRow, FindResult, InventoryBy, InventoryCount, InventoryResult, QueryFilters, RelationDirection, RelationsResult, - STRUCTURED_QUERY_SCHEMA_VERSION, ScopeMode, StructuredQuery, glob_match, parse_edge_type, - parse_node_type, + STRUCTURED_QUERY_SCHEMA_VERSION, ScopeMode, StructuredQuery, glob_match, import_package_prefix, + parse_annotation_list, parse_edge_type, parse_node_type, }; pub use stable_key::{ MmapNodeKey, NodeRowRef, StableNodeKey, NAMESPACE_RGCTL, deterministic_node_id, diff --git a/crates/rgctl-graph/src/structured_query.rs b/crates/rgctl-graph/src/structured_query.rs index 63f8029a..c271b942 100644 --- a/crates/rgctl-graph/src/structured_query.rs +++ b/crates/rgctl-graph/src/structured_query.rs @@ -262,6 +262,11 @@ pub struct QueryFilters { pub count_only: bool, /// Exact name match (no glob) pub exact: bool, + /// Annotation invert: simple names / `@Name` / FQNs (OR). When set, `find` returns + /// AnnotatedWith **sources** matching any listed annotation. + pub annotation_names: Option>, + /// Request annotation argument payloads when indexed (`--show-attributes`). + pub show_attributes: bool, } /// Session over an open snapshot store. @@ -485,6 +490,30 @@ impl<'a> StructuredQuery<'a> { /// Entity search (`rgctl find`). pub fn find(&self, pattern: Option<&str>, filters: &QueryFilters) -> Result { + if filters.show_attributes { + // Argument indexing is Phase C; refuse inventing empty lookup= fields. + if filters + .annotation_names + .as_ref() + .map(|v| !v.is_empty()) + .unwrap_or(false) + { + return Err(Error::InvalidQuery( + "annotation attributes are not indexed yet; omit --show-attributes or re-discover after arg indexing lands" + .into(), + )); + } + return Err(Error::InvalidQuery( + "--show-attributes requires --annotation and indexed annotation arguments".into(), + )); + } + + if let Some(annots) = filters.annotation_names.as_ref() { + if !annots.is_empty() { + return self.find_by_annotation(annots, pattern, filters); + } + } + let mut candidates: Vec = Vec::new(); if let Some(pat) = pattern { candidates = self.lookup_name_candidates(pat, filters.exact)?; @@ -516,6 +545,65 @@ impl<'a> StructuredQuery<'a> { } candidates.retain(|n| self.node_passes(n, filters)); + Self::finish_find(candidates, filters) + } + + /// Invert AnnotatedWith: return sources carrying any of the listed annotations (OR). + fn find_by_annotation( + &self, + annots: &[String], + pattern: Option<&str>, + filters: &QueryFilters, + ) -> Result { + let normalized: Vec = annots + .iter() + .map(|a| normalize_annotation_name(a)) + .filter(|a| !a.is_empty()) + .collect(); + if normalized.is_empty() { + return Err(Error::InvalidQuery( + "--annotation requires at least one name (e.g. @MessageDriven)".into(), + )); + } + + let mut seen = HashSet::new(); + let mut candidates: Vec = Vec::new(); + self.store.for_each_edge(|from, to, et| { + if et != EdgeType::AnnotatedWith { + return Ok(()); + } + let Some(ann) = self.store.get_node(to)? else { + return Ok(()); + }; + if !normalized.iter().any(|want| annotation_name_matches(&ann, want)) { + return Ok(()); + } + if !seen.insert(from) { + return Ok(()); + } + let Some(src) = self.store.get_node(from)? else { + return Ok(()); + }; + if let Some(pat) = pattern { + let ok = if filters.exact || !is_glob_pattern(pat) { + src.name == pat + } else { + glob_match(pat, &src.name) + }; + if !ok { + return Ok(()); + } + } + if self.node_passes(&src, filters) { + candidates.push(src); + } + Ok(()) + })?; + + Self::finish_find(candidates, filters) + } + + fn finish_find(candidates: Vec, filters: &QueryFilters) -> Result { let total = candidates.len(); let limit = filters.limit.unwrap_or(total); let entities: Vec = if filters.count_only { @@ -1011,6 +1099,38 @@ impl<'a> StructuredQuery<'a> { counts, }) } + InventoryBy::ImportPrefix => { + // Observed prefixes only (no zero-fill). Default depth = 2 dotted segments. + const DEPTH: usize = 2; + let mut map: HashMap = HashMap::new(); + for id in self.store.all_node_ids() { + let Some(n) = self.store.get_node(id)? else { + continue; + }; + if n.node_type != NodeType::Import { + continue; + } + if !self.node_passes(&n, filters) { + continue; + } + let key = import_package_prefix(&n.name, DEPTH); + *map.entry(key).or_insert(0) += 1; + } + let mut counts: Vec<_> = map + .into_iter() + .map(|(key, count)| InventoryCount { + key, + count, + occurrences: None, + }) + .collect(); + counts.sort_by(|a, b| b.count.cmp(&a.count).then_with(|| a.key.cmp(&b.key))); + Ok(InventoryResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + by: "import-prefix".into(), + counts, + }) + } } } @@ -1072,6 +1192,8 @@ pub enum InventoryBy { File, /// Community id property Community, + /// Import package prefix (first N dotted segments; observed only) + ImportPrefix, } impl InventoryBy { @@ -1083,8 +1205,9 @@ impl InventoryBy { "lang" | "language" => Ok(Self::Lang), "file" => Ok(Self::File), "community" => Ok(Self::Community), + "import-prefix" | "importprefix" | "import_prefix" => Ok(Self::ImportPrefix), other => Err(Error::InvalidQuery(format!( - "unknown inventory --by '{other}' (expected type|edge|lang|file|community)" + "unknown inventory --by '{other}' (expected type|edge|lang|file|community|import-prefix)" ))), } } @@ -1237,6 +1360,66 @@ fn lang_from_ext(ext: &str) -> String { } } +/// Strip `@` and whitespace from an annotation CLI token. +pub fn normalize_annotation_name(raw: &str) -> String { + raw.trim().trim_start_matches('@').trim().to_string() +} + +/// Parse `--annotation @A,@B` into normalized names (empty tokens dropped). +pub fn parse_annotation_list(raw: &str) -> Vec { + raw.split(',') + .map(normalize_annotation_name) + .filter(|s| !s.is_empty()) + .collect() +} + +/// True when an Annotation node matches a normalized want (simple name or FQN). +fn annotation_name_matches(node: &Node, want: &str) -> bool { + if want.is_empty() { + return false; + } + if node.name == want { + return true; + } + if let Some(simple) = node.name.rsplit('.').next() { + if simple == want { + return true; + } + } + if let Some(qn) = node.qualified_name.as_deref() { + if qn == want { + return true; + } + if let Some(simple) = qn.rsplit('.').next() { + if simple == want { + return true; + } + } + } + false +} + +/// Derive import package prefix: strip `import` / `static` / `.*` / `;`, take first `depth` segments. +pub fn import_package_prefix(import_name: &str, depth: usize) -> String { + let mut s = import_name.trim(); + if let Some(rest) = s.strip_prefix("import ") { + s = rest.trim(); + } + if let Some(rest) = s.strip_prefix("static ") { + s = rest.trim(); + } + s = s.trim_end_matches(';').trim(); + if let Some(rest) = s.strip_suffix(".*") { + s = rest.trim(); + } + let parts: Vec<&str> = s.split('.').filter(|p| !p.is_empty()).collect(); + if parts.is_empty() { + return "".into(); + } + let take = depth.max(1).min(parts.len()); + parts[..take].join(".") +} + #[cfg(test)] mod tests { use super::*; @@ -1266,6 +1449,27 @@ mod tests { .with_file_path(""); let imp = Node::new(NodeType::Import, "import javax.ws.rs.Path;") .with_file_path("src/Svc.java"); + let imp_ejb = Node::new(NodeType::Import, "import javax.ejb.MessageDriven;") + .with_file_path("src/OrderMDB.java"); + let imp_jms = Node::new(NodeType::Import, "import javax.jms.Topic;") + .with_file_path("src/OrderMDB.java"); + let imp_eclipselink = + Node::new(NodeType::Import, "import org.eclipse.persistence.sessions.Session;") + .with_file_path("src/Jpa.java"); + let mdb_ann = Node::new(NodeType::Annotation, "MessageDriven") + .with_qualified_name("javax.ejb.MessageDriven") + .with_file_path(""); + let scoped_ann = Node::new(NodeType::Annotation, "SessionScoped") + .with_qualified_name("javax.enterprise.context.SessionScoped") + .with_file_path(""); + let mdb = Node::new(NodeType::Class, "OrderMDB") + .with_qualified_name("com.coolstore.OrderMDB") + .with_file_path("src/OrderMDB.java") + .with_location(1, 80); + let cart = Node::new(NodeType::Class, "CartResource") + .with_qualified_name("com.coolstore.CartResource") + .with_file_path("src/CartResource.java") + .with_location(1, 40); let caller = Node::new(NodeType::Function, "handle") .with_qualified_name("de.metas.printing.esb.Svc.handle") .with_file_path("src/Svc.java") @@ -1275,11 +1479,22 @@ mod tests { let c1_id = c1.id; let base_id = base.id; let caller_id = caller.id; + let mdb_ann_id = mdb_ann.id; + let scoped_ann_id = scoped_ann.id; + let mdb_id = mdb.id; + let cart_id = cart.id; backend.insert_node(f1).unwrap(); backend.insert_node(a1).unwrap(); backend.insert_node(c1).unwrap(); backend.insert_node(base).unwrap(); backend.insert_node(imp).unwrap(); + backend.insert_node(imp_ejb).unwrap(); + backend.insert_node(imp_jms).unwrap(); + backend.insert_node(imp_eclipselink).unwrap(); + backend.insert_node(mdb_ann).unwrap(); + backend.insert_node(scoped_ann).unwrap(); + backend.insert_node(mdb).unwrap(); + backend.insert_node(cart).unwrap(); backend.insert_node(caller).unwrap(); backend .insert_edge(Edge::new(f1_id, a1_id, EdgeType::AnnotatedWith)) @@ -1290,6 +1505,12 @@ mod tests { backend .insert_edge(Edge::new(caller_id, f1_id, EdgeType::Calls)) .unwrap(); + backend + .insert_edge(Edge::new(mdb_id, mdb_ann_id, EdgeType::AnnotatedWith)) + .unwrap(); + backend + .insert_edge(Edge::new(cart_id, scoped_ann_id, EdgeType::AnnotatedWith)) + .unwrap(); write_columnar_from_backend(&backend, &path).unwrap(); let store = SnapshotNodeStore::open(&path).unwrap(); (dir, store) @@ -1371,8 +1592,127 @@ mod tests { }, ) .unwrap(); + assert!(res.total >= 1); + assert!(res.entities.iter().any(|e| e.name.starts_with("import javax"))); + } + + #[test] + fn find_by_annotation_message_driven() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .find( + None, + &QueryFilters { + annotation_names: Some(vec!["MessageDriven".into()]), + node_type: Some(NodeType::Class), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 1); + assert_eq!(res.entities[0].name, "OrderMDB"); + } + + #[test] + fn find_by_annotation_or_list() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .find( + None, + &QueryFilters { + annotation_names: Some(vec!["MessageDriven".into(), "SessionScoped".into()]), + node_type: Some(NodeType::Class), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 2); + let names: HashSet<_> = res.entities.iter().map(|e| e.name.as_str()).collect(); + assert!(names.contains("OrderMDB")); + assert!(names.contains("CartResource")); + } + + #[test] + fn find_by_annotation_at_prefix_and_fqn() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let list = parse_annotation_list("@SessionScoped,javax.ejb.MessageDriven"); + let res = q + .find( + None, + &QueryFilters { + annotation_names: Some(list), + node_type: Some(NodeType::Class), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(res.total, 2); + } + + #[test] + fn show_attributes_errors_when_not_indexed() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let err = q + .find( + None, + &QueryFilters { + annotation_names: Some(vec!["Resource".into()]), + show_attributes: true, + ..Default::default() + }, + ) + .unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("not indexed") || msg.contains("attributes")); + } + + #[test] + fn inventory_import_prefix_census() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .inventory(InventoryBy::ImportPrefix, &QueryFilters::default()) + .unwrap(); + assert_eq!(res.by, "import-prefix"); + let keys: HashMap<_, _> = res.counts.iter().map(|c| (c.key.as_str(), c.count)).collect(); + assert!(keys.get("javax.ws").copied().unwrap_or(0) >= 1); + assert!(keys.get("javax.ejb").copied().unwrap_or(0) >= 1); + assert!(keys.get("javax.jms").copied().unwrap_or(0) >= 1); + assert!(keys.get("org.eclipse").copied().unwrap_or(0) >= 1); + } + + #[test] + fn import_package_prefix_heuristic() { + assert_eq!( + import_package_prefix("import javax.ejb.MessageDriven;", 2), + "javax.ejb" + ); + assert_eq!( + import_package_prefix("import static org.junit.Assert.*;", 2), + "org.junit" + ); + assert_eq!(import_package_prefix("com.fasterxml.jackson.databind.ObjectMapper", 2), "com.fasterxml"); + } + + #[test] + fn find_mdb_suffix_glob() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + let res = q + .find( + Some("*MDB*"), + &QueryFilters { + node_type: Some(NodeType::Class), + ..Default::default() + }, + ) + .unwrap(); assert_eq!(res.total, 1); - assert!(res.entities[0].name.starts_with("import javax")); + assert_eq!(res.entities[0].name, "OrderMDB"); } #[test] diff --git a/skills/rgctl/SKILL.md b/skills/rgctl/SKILL.md index af1766ad..75743771 100644 --- a/skills/rgctl/SKILL.md +++ b/skills/rgctl/SKILL.md @@ -90,9 +90,10 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | User Intent | CLI Command | |-------------|-------------| -| Evaluate migration rules | `discover . --with-kantra` | -| Filter by migration target | `discover . --with-kantra --kantra-target quarkus` | -| CI / custom ruleset | `discover . --with-kantra --kantra-rules PATH` | +| Evaluate migration rules | `discover . --with-kantra` or `rules run ./rules/` | +| Filter by migration target | `discover . --with-kantra --kantra-target quarkus` / `rules run ./rules/ --target quarkus` | +| CI / custom ruleset | `discover . --with-kantra --kantra-rules PATH` / `rules run PATH` | +| Index rules only | `discover . --with-kantra --kantra-index-only` / `rules run PATH --index-only` | | List indexed rules (GQL) | `gql "MATCH (r:KantraRule) RETURN r LIMIT 20"` | | Rules for one target label | `gql` with `` r.`konveyor.io/target` `` property (backticks) | | Read violations artifact | `.rgctl/kantra_findings.json` | @@ -105,11 +106,16 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | User Intent | CLI Command | |-------------|-------------| +| Session / index freshness | `status` | | Schema / counts (incl. zeros) | `inventory --by type` or `inventory --by edge` | +| Import prefix census | `inventory --by import-prefix` | | Count functions | `find --type function --count-only` | | Find by name/type | `find "User*" --type class --limit 50` | +| Suffix scan (MDB / Remote) | `find '*MDB*' --type class` | +| Classes with annotation | `find --annotation @MessageDriven --type class` | | javax import worklist | `find "import javax*" --type import --scope ` | | Annotation pairs (seedless) | `relations --edge annotatedwith --from-type function --to-type annotation --scope ` | +| Deployment / persistence config | `resources` (persistence.xml, weblogic/jboss/web/beans) | | Find callers/callees | `callers --depth 1` / `callees ` | | Outside callers of a module | `callers --scope --scope-mode outside` | | EXTENDS / IMPLEMENTS inventory | `relations --edge extends --from-type class` (omit SYMBOL) | @@ -119,8 +125,9 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | Refresh community labels | `communities label --write` | | Ad-hoc Cypher (experimental) | `gql "MATCH …"` — uncanny valley; prefer verbs above | -**Complexity honesty:** exact name = hash index; prefix/`*mid*`/`--scope` may scan keys/columns until better indexes land. Module re-index is still a strong speed lever. Annotation **arguments** (e.g. `@Path("/x")`) are not indexed yet. +**Migration probe order:** `status` → `inventory --by import-prefix` → `find --annotation …` / suffix globs → `resources` → `rules run` / `--with-kantra` → `callers InitialContext`. +**Complexity honesty:** exact name = hash index; prefix/`*mid*`/`--scope` may scan keys/columns until better indexes land. Module re-index is still a strong speed lever. Annotation **arguments** (e.g. `@Path("/x")`) are not indexed yet (`--show-attributes` errors until they are). **See:** [Command Encyclopedia](references/command-encyclopedia.md) (find/callers/relations/inventory), [GQL Reference](references/gql-reference.md) (legacy), [Semantic Search Guide](../../docs/guides/semantic-search.md) ### 3. Impact & Safety diff --git a/skills/rgctl/references/command-encyclopedia.md b/skills/rgctl/references/command-encyclopedia.md index 53c1b840..19afe33b 100644 --- a/skills/rgctl/references/command-encyclopedia.md +++ b/skills/rgctl/references/command-encyclopedia.md @@ -107,26 +107,42 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json ```bash rgctl -f json find [PATTERN] --type function --scope pkg --limit 50 rgctl -f json find --type function --count-only +rgctl -f json find --annotation @MessageDriven --type class +rgctl -f json find --annotation @Stateful,@Stateless,@Singleton --type class +rgctl -f json find '*MDB*' --type class --limit 50 # bare-name suffix scan rgctl -f json callers --depth 1 --file PATH --class NAME --line N rgctl -f json callees --depth 1 rgctl -f json relations [SYMBOL] --edge annotatedwith --from-type function --to-type annotation --scope pkg rgctl -f json relations --edge extends --from-type class # seedless rgctl -f json inventory --by type # includes zero-count kinds rgctl -f json inventory --by edge +rgctl -f json inventory --by import-prefix # javax.ejb / javax.jms / org.eclipse … +rgctl -f json status # snapshot presence, digest, node/edge counts +rgctl -f json resources # persistence.xml + weblogic/jboss/web/beans (no Kantra) +rgctl -f json rules run ./rules/ [--target quarkus] # post-index Kantra eval rgctl -f json query find … # alias namespace ``` **Purpose:** Deterministic mmap structured query (no Cypher, no `MemoryBackend` hydrate). Prefer these over `gql` for agent work. Relations `total` is distinct `(source,target,edge)`; duplicates collapse with `occurrences` (`schema_version` ≥ 2). `inventory --by edge` uses the same rule: `count` = distinct, `occurrences` = raw stored edges. -**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). +**Migration probes (Coolstore-shaped):** +1. `status` — is `.rgctl/` fresh? +2. `inventory --by import-prefix` — EE surface census +3. `find --annotation @MessageDriven|@SessionScoped|…` — blockers without package guess +4. `find '*MDB*'` / `'*Remote*'` — suffix scan before reading files +5. `resources` — persistence provider, JNDI DS, weblogic/jboss bindings +6. `rules run ./rules/` or `discover --with-kantra` — fire `when:` catalog (M2) +7. `callers InitialContext` — JNDI usage sites -**Flags:** `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). `--file` / `--class` / `--line` disambiguate. Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. +**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). `resources` / `status` do not rediscover. `rules run` requires a snapshot; Kantra stays opt-in. -**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. +**Flags:** `--annotation` inverts `AnnotatedWith` (OR list; `@` optional). `--show-attributes` needs annotation-arg indexing (errors honestly until indexed). `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). `--file` / `--class` / `--line` disambiguate. Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. + +**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. Do not treat Kantra as the only search path — use annotation/import/resources first. **Agent should report:** counts, lean names/files, keyed edge pairs — not full node dumps. -**See:** OpenSpec `add-structured-query-cli`; GQL demotion is a follow-up change (latency + cardinality gates cleared). +**See:** OpenSpec `add-migration-search-primitives` (+ `add-structured-query-cli`); GQL demotion is a follow-up change. --- diff --git a/skills/rgctl/workflows/kantra.md b/skills/rgctl/workflows/kantra.md index c3066cda..fcd6ed9b 100644 --- a/skills/rgctl/workflows/kantra.md +++ b/skills/rgctl/workflows/kantra.md @@ -13,6 +13,9 @@ Native evaluation of [Konveyor Kantra](https://github.com/konveyor/kantra) rules ```bash rgctl discover . -l java --with-kantra # violations: .rgctl/kantra_findings.json + +# Post-index (when snapshot already exists): +rgctl -f json rules run ./rules/ --target quarkus # rules in graph: KantraRule / KantraRuleset nodes (GQL) ``` diff --git a/src/cli/mod.rs b/src/cli/mod.rs index af801f33..80691b2a 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -40,6 +40,9 @@ mod slice; pub mod slice_output; mod stage_profile; mod structured_query; +mod session_status; +mod resources; +mod rules; pub use args::OutputFormat; @@ -271,6 +274,14 @@ pub enum Commands { /// Return counts only #[arg(long = "count-only")] count_only: bool, + + /// Invert AnnotatedWith: entities carrying this annotation (`@Foo` or `Foo[,Bar…]`) + #[arg(long = "annotation", value_name = "NAME[,NAME…]")] + annotation: Option, + + /// Include annotation argument text when indexed (requires --annotation) + #[arg(long = "show-attributes")] + show_attributes: bool, }, /// Incoming CALLS neighbors for a symbol @@ -385,7 +396,7 @@ pub enum Commands { /// Aggregate symbol/edge counts (includes zero-count schema kinds for type/edge) Inventory { - /// Aggregation dimension: type | edge | lang | file | community + /// Aggregation dimension: type | edge | lang | file | community | import-prefix #[arg(long = "by", default_value = "type")] by: String, @@ -402,6 +413,22 @@ pub enum Commands { exclude_scope: bool, }, + /// Session graph status (snapshot presence, digest, node/edge counts; no rediscover) + Status, + + /// Parse deployment/persistence descriptors (persistence.xml, weblogic/jboss/web/beans) + Resources { + /// Optional extra descriptor file to include + #[arg(long = "file", value_name = "PATH")] + file: Option, + }, + + /// Evaluate Konveyor-shaped rules against the session (Kantra engine) + Rules { + #[command(subcommand)] + action: RulesCommands, + }, + /// Alias namespace for structured query verbs (`query find`, `query callers`, …) Query { #[command(subcommand)] @@ -681,6 +708,28 @@ pub enum Commands { }, } +#[derive(Subcommand)] +pub enum RulesCommands { + /// Evaluate a ruleset directory (or catalog) against the current session graph + Run { + /// Ruleset directory (`ruleset.yaml` + `*.yaml`), like `--kantra-rules` + #[arg(value_name = "DIR")] + rules_dir: std::path::PathBuf, + + /// Filter by `konveyor.io/target` label + #[arg(long = "target", value_name = "NAME")] + target: Option, + + /// Index KantraRule nodes only; skip evaluation + #[arg(long = "index-only")] + index_only: bool, + + /// Override with a rulesets tree (mutually exclusive with DIR as single ruleset when set) + #[arg(long = "catalog", value_name = "ROOT")] + catalog: Option, + }, +} + #[derive(Subcommand)] pub enum QueryCommands { /// Alias for `rgctl find` @@ -705,6 +754,10 @@ pub enum QueryCommands { limit: Option, #[arg(long = "count-only")] count_only: bool, + #[arg(long = "annotation", value_name = "NAME[,NAME…]")] + annotation: Option, + #[arg(long = "show-attributes")] + show_attributes: bool, }, /// Alias for `rgctl callers` Callers { @@ -1188,6 +1241,8 @@ impl Cli { exact, limit, count_only, + annotation, + show_attributes, } => structured_query::run_find( &ctx, pattern, @@ -1204,6 +1259,8 @@ impl Cli { }, exact, count_only, + annotation, + show_attributes, ), Commands::Callers { symbol, @@ -1310,6 +1367,16 @@ impl Cli { limit: None, }, ), + Commands::Status => session_status::run_status(&ctx), + Commands::Resources { file } => resources::run_resources(&ctx, file), + Commands::Rules { action } => match action { + RulesCommands::Run { + rules_dir, + target, + index_only, + catalog, + } => rules::run_rules(&ctx, rules_dir, target, index_only, catalog), + }, Commands::Query { action } => match action { QueryCommands::Find { pattern, @@ -1322,6 +1389,8 @@ impl Cli { exact, limit, count_only, + annotation, + show_attributes, } => structured_query::run_find( &ctx, pattern, @@ -1338,6 +1407,8 @@ impl Cli { }, exact, count_only, + annotation, + show_attributes, ), QueryCommands::Callers { symbol, @@ -1752,6 +1823,9 @@ fn command_label_for(command: &Commands) -> &'static str { Commands::Callees { .. } => "callees", Commands::Relations { .. } => "relations", Commands::Inventory { .. } => "inventory", + Commands::Status => "status", + Commands::Resources { .. } => "resources", + Commands::Rules { .. } => "rules", Commands::Query { action } => match action { QueryCommands::Find { .. } => "query find", QueryCommands::Callers { .. } => "query callers", diff --git a/src/cli/resources.rs b/src/cli/resources.rs new file mode 100644 index 00000000..74625237 --- /dev/null +++ b/src/cli/resources.rs @@ -0,0 +1,367 @@ +//! `rgctl resources` — first-party parse of JEE descriptors (no Kantra). + +use super::context::CliContext; +use super::OutputFormat; +use anyhow::Result; +use serde::Serialize; +use std::fs; +use std::path::{Path, PathBuf}; + +const RESOURCES_SCHEMA_VERSION: u32 = 1; + +#[derive(Debug, Default, Serialize)] +struct ResourcesResult { + schema_version: u32, + command: &'static str, + persistence: Vec, + datasources: Vec, + jms: Vec, + ejb_bindings: Vec, + cdi: Vec, + files_scanned: Vec, +} + +#[derive(Debug, Clone, Serialize)] +struct PersistenceUnit { + file: String, + name: Option, + provider: Option, + jta_data_source: Option, + non_jta_data_source: Option, + schema_generation: Vec, +} + +#[derive(Debug, Clone, Serialize)] +struct NamedBinding { + kind: String, + name: String, + #[serde(skip_serializing_if = "Option::is_none")] + jndi: Option, + file: String, +} + +#[derive(Debug, Clone, Serialize)] +struct CdiHint { + file: String, + bean_discovery_mode: Option, +} + +/// `rgctl resources` +pub fn run_resources(ctx: &CliContext, extra_file: Option) -> Result<()> { + let mut out = ResourcesResult { + schema_version: RESOURCES_SCHEMA_VERSION, + command: "resources", + ..Default::default() + }; + + let mut files = discover_descriptor_files(&ctx.repo)?; + if let Some(f) = extra_file { + files.push(PathBuf::from(f)); + } + + for path in &files { + let rel = path + .strip_prefix(&ctx.repo) + .unwrap_or(path) + .to_string_lossy() + .replace('\\', "/"); + out.files_scanned.push(rel.clone()); + let Ok(text) = fs::read_to_string(path) else { + continue; + }; + let name = path + .file_name() + .and_then(|s| s.to_str()) + .unwrap_or("") + .to_ascii_lowercase(); + if name == "persistence.xml" || rel.ends_with("META-INF/persistence.xml") { + parse_persistence(&text, &rel, &mut out.persistence); + } else if name.starts_with("weblogic") || name.starts_with("jboss-") { + parse_server_bindings(&text, &rel, &mut out); + } else if name == "web.xml" { + parse_web_xml(&text, &rel, &mut out); + } else if name == "beans.xml" { + parse_beans_xml(&text, &rel, &mut out.cdi); + } + } + + if ctx.format == OutputFormat::Json { + let v = serde_json::to_value(&out)?; + ctx.emit_json_value(&v)?; + } else { + ctx.stdout_line(&format!( + "resources: {} persistence, {} datasources, {} jms, {} ejb, {} cdi ({} files)", + out.persistence.len(), + out.datasources.len(), + out.jms.len(), + out.ejb_bindings.len(), + out.cdi.len(), + out.files_scanned.len() + ))?; + for p in &out.persistence { + ctx.stdout_line(&format!( + " persistence provider={} jta={:?}", + p.provider.as_deref().unwrap_or("?"), + p.jta_data_source + ))?; + } + } + Ok(()) +} + +fn discover_descriptor_files(repo: &Path) -> Result> { + let mut out = Vec::new(); + let walker = ignore::WalkBuilder::new(repo) + .hidden(false) + .git_ignore(true) + .build(); + for entry in walker.flatten() { + let path = entry.path(); + if !path.is_file() { + continue; + } + let name = path + .file_name() + .and_then(|s| s.to_str()) + .unwrap_or("") + .to_ascii_lowercase(); + let ok = name == "persistence.xml" + || name == "web.xml" + || name == "beans.xml" + || (name.starts_with("weblogic") && name.ends_with(".xml")) + || (name.starts_with("jboss-") && name.ends_with(".xml")); + if ok { + out.push(path.to_path_buf()); + } + } + out.sort(); + Ok(out) +} + +fn parse_persistence(text: &str, file: &str, out: &mut Vec) { + // Lightweight tag scrape (namespaces ignored). + for unit in split_tags(text, "persistence-unit") { + let name = attr_value(&unit, "name"); + let provider = tag_text(&unit, "provider"); + let jta = tag_text(&unit, "jta-data-source"); + let non_jta = tag_text(&unit, "non-jta-data-source"); + let mut schema = Vec::new(); + for prop in split_tags(&unit, "property") { + if let Some(n) = attr_value(&prop, "name") { + if n.contains("schema-generation") || n.contains("ddl") { + let v = attr_value(&prop, "value").unwrap_or_default(); + schema.push(format!("{n}={v}")); + } + } + } + out.push(PersistenceUnit { + file: file.into(), + name, + provider, + jta_data_source: jta, + non_jta_data_source: non_jta, + schema_generation: schema, + }); + } +} + +fn parse_server_bindings(text: &str, file: &str, out: &mut ResourcesResult) { + // weblogic / jboss: capture common JNDI-ish attributes and resource-ref names. + for (tag, kind) in [ + ("resource-description", "resource"), + ("resource-env-description", "resource-env"), + ("ejb-local-reference-description", "ejb"), + ("ejb-reference-description", "ejb"), + ("message-destination-description", "jms"), + ("connection-factory", "jms-factory"), + ("topic", "jms-topic"), + ("queue", "jms-queue"), + ("datasource", "datasource"), + ] { + for block in split_tags(text, tag) { + let name = tag_text(&block, "res-ref-name") + .or_else(|| tag_text(&block, "resource-env-ref-name")) + .or_else(|| tag_text(&block, "ejb-ref-name")) + .or_else(|| attr_value(&block, "name")) + .unwrap_or_else(|| tag.to_string()); + let jndi = tag_text(&block, "jndi-name") + .or_else(|| tag_text(&block, "lookup-name")) + .or_else(|| attr_value(&block, "jndi-name")); + let binding = NamedBinding { + kind: kind.into(), + name, + jndi, + file: file.into(), + }; + match kind { + "jms" | "jms-factory" | "jms-topic" | "jms-queue" => out.jms.push(binding), + "ejb" => out.ejb_bindings.push(binding), + "datasource" => out.datasources.push(binding), + _ => { + if binding + .jndi + .as_deref() + .unwrap_or("") + .contains("jdbc") + || binding.name.contains("jdbc") + || binding.name.contains("DataSource") + || binding.name.contains("DS") + { + out.datasources.push(binding); + } else if binding.name.contains("jms") + || binding + .jndi + .as_deref() + .unwrap_or("") + .contains("jms") + { + out.jms.push(binding); + } else { + out.ejb_bindings.push(binding); + } + } + } + } + } + // Coolstore-shaped: plain jdbc/CoolstoreDS + if out.datasources.is_empty() { + if let Some(jndi) = tag_text(text, "jndi-name") { + if jndi.contains("jdbc") || jndi.contains("DS") { + out.datasources.push(NamedBinding { + kind: "jndi".into(), + name: jndi.clone(), + jndi: Some(jndi), + file: file.into(), + }); + } + } + } +} + +fn parse_web_xml(text: &str, file: &str, out: &mut ResourcesResult) { + for block in split_tags(text, "resource-ref") { + let name = tag_text(&block, "res-ref-name").unwrap_or_else(|| "resource-ref".into()); + let jndi = tag_text(&block, "lookup-name"); + out.datasources.push(NamedBinding { + kind: "resource-ref".into(), + name, + jndi, + file: file.into(), + }); + } +} + +fn parse_beans_xml(text: &str, file: &str, out: &mut Vec) { + let mode = attr_value(text, "bean-discovery-mode").or_else(|| { + // sometimes on beans root + text.find("bean-discovery-mode=\"") + .and_then(|i| { + let rest = &text[i + "bean-discovery-mode=\"".len()..]; + rest.split('"').next().map(|s| s.to_string()) + }) + }); + out.push(CdiHint { + file: file.into(), + bean_discovery_mode: mode, + }); +} + +fn split_tags(hay: &str, tag: &str) -> Vec { + let open = format!("<{tag}"); + let close = format!(""); + let mut out = Vec::new(); + let mut rest = hay; + while let Some(start) = rest.find(&open) { + let from = &rest[start..]; + if let Some(end) = from.find(&close) { + out.push(from[..end + close.len()].to_string()); + rest = &from[end + close.len()..]; + } else { + // self-closing or truncated + if let Some(gt) = from.find('>') { + out.push(from[..=gt].to_string()); + rest = &from[gt + 1..]; + } else { + break; + } + } + } + out +} + +fn tag_text(hay: &str, tag: &str) -> Option { + let open = format!("<{tag}"); + let close = format!(""); + let start = hay.find(&open)?; + let after_open = &hay[start..]; + let gt = after_open.find('>')?; + let body_start = &after_open[gt + 1..]; + let end = body_start.find(&close)?; + Some(body_start[..end].trim().to_string()) +} + +fn attr_value(hay: &str, attr: &str) -> Option { + let key = format!("{attr}=\""); + let i = hay.find(&key)?; + let rest = &hay[i + key.len()..]; + let end = rest.find('"')?; + Some(rest[..end].to_string()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_persistence_eclipselink_coolstore() { + let xml = r#" + + + org.eclipse.persistence.jpa.PersistenceProvider + jdbc/CoolstoreDS + + + + + "#; + let mut units = Vec::new(); + parse_persistence(xml, "META-INF/persistence.xml", &mut units); + assert_eq!(units.len(), 1); + assert!(units[0] + .provider + .as_deref() + .unwrap() + .contains("eclipse.persistence")); + assert_eq!( + units[0].jta_data_source.as_deref(), + Some("jdbc/CoolstoreDS") + ); + assert!(!units[0].schema_generation.is_empty()); + } + + #[test] + fn parse_weblogic_jndi() { + let xml = r#" + + + orders + jms/orders + + + jdbc/CoolstoreDS + jdbc/CoolstoreDS + + "#; + let mut out = ResourcesResult::default(); + parse_server_bindings(xml, "WEB-INF/weblogic-ejb-jar.xml", &mut out); + assert!(!out.jms.is_empty() || !out.datasources.is_empty() || !out.ejb_bindings.is_empty()); + assert!( + out.datasources.iter().any(|d| d.name.contains("CoolstoreDS")) + || out.datasources.iter().any(|d| d + .jndi + .as_deref() + .unwrap_or("") + .contains("CoolstoreDS")) + ); + } +} diff --git a/src/cli/rules.rs b/src/cli/rules.rs new file mode 100644 index 00000000..7ab62ef6 --- /dev/null +++ b/src/cli/rules.rs @@ -0,0 +1,192 @@ +//! `rgctl rules run` — post-index Kantra evaluation against the session snapshot. + +use super::context::CliContext; +use super::kantra_discover::{ + preload_discovered_sources, resolve_kantra_catalog, run_kantra_index, run_kantra_violates, +}; +use super::stage_profile::DiscoverStageReport; +use super::OutputFormat; +use anyhow::{Context, Result, bail}; +use rgctl_graph::schema::{EdgeType, NodeType}; +use rgctl_graph::snapshot::SNAPSHOT_FILE; +use rgctl_kantra::{ + EvalContext, EvalEdge, EvalGraph, EvalNode, KantraEngine, KantraFileCache, ViolationResolver, + cache_dir, ruleset_hash, +}; +use std::collections::HashSet; +use std::fs; +use std::path::{Path, PathBuf}; + +/// `rgctl rules run ` (and related flags). +pub fn run_rules( + ctx: &CliContext, + rules_dir: PathBuf, + target: Option, + index_only: bool, + catalog: Option, +) -> Result<()> { + if !rules_dir.is_dir() && catalog.is_none() { + bail!("rules path is not a directory: {}", rules_dir.display()); + } + let store = &ctx.repo; + let snapshot = rgctl_graph::paths::artifact_path(store, SNAPSHOT_FILE); + if !snapshot.exists() { + bail!("Graph snapshot not found (run `rgctl discover` first)"); + } + + let rules_opt = if catalog.is_some() { + None + } else { + Some(rules_dir.as_path()) + }; + let catalog_opt = catalog.as_deref(); + + let mut profile = DiscoverStageReport::default(); + run_kantra_index(store, rules_opt, catalog_opt, &mut profile)?; + + if index_only { + if ctx.format == OutputFormat::Json { + let v = serde_json::json!({ + "schema_version": 1, + "command": "rules", + "action": "index-only", + "kantra_index_secs": profile.kantra_index.secs, + }); + ctx.emit_json_value(&v)?; + } else { + ctx.stdout_line("rules: indexed KantraRule nodes (eval skipped)")?; + } + return Ok(()); + } + + let catalog = resolve_kantra_catalog(rules_opt, catalog_opt)?; + let (engine, _) = KantraEngine::from_catalog(catalog.clone(), target.as_deref()) + .map_err(|e| anyhow::anyhow!("kantra catalog: {e}"))?; + + let files = collect_source_files(&ctx.repo)?; + let sources = preload_discovered_sources(&ctx.repo, &files, None); + let graph = build_eval_graph_from_snapshot(&snapshot)?; + let rs_hash = ruleset_hash( + engine.catalog_id().unwrap_or("unknown"), + catalog.rules.len(), + ); + let mut file_cache = KantraFileCache::load(&cache_dir(store), &rs_hash); + let mut eval_ctx = EvalContext { + repo_root: &ctx.repo, + files: &files, + sources: &sources, + graph: &graph, + cache: Some(&mut file_cache), + cached_files: HashSet::new(), + }; + let (mut findings, _) = engine + .evaluate(&mut eval_ctx) + .map_err(|e| anyhow::anyhow!("kantra evaluate: {e}"))?; + let resolver = ViolationResolver::from_eval_nodes(&graph.nodes); + resolver.attach_node_ids(&mut findings.violations); + findings.command = "rules_run".into(); + file_cache.save(&cache_dir(store))?; + + let out_path = rgctl_graph::paths::artifact_path(store, "kantra_findings.json"); + if let Some(parent) = out_path.parent() { + fs::create_dir_all(parent)?; + } + let json = serde_json::to_string_pretty(&findings).context("serialize kantra findings")?; + fs::write(&out_path, &json).with_context(|| format!("write {}", out_path.display()))?; + + // Best-effort VIOLATES edges (snapshot rewrite). + let _ = run_kantra_violates(store, &mut profile); + + if ctx.format == OutputFormat::Json { + let v = serde_json::to_value(&findings)?; + ctx.emit_json_value(&v)?; + } else { + ctx.stdout_line(&format!( + "rules: {} violations, {} skipped → {}", + findings.violations.len(), + findings.skipped_rules.len(), + out_path.display() + ))?; + } + Ok(()) +} + +fn collect_source_files(repo: &Path) -> Result> { + let mut out = Vec::new(); + for entry in ignore::WalkBuilder::new(repo).git_ignore(true).build().flatten() { + let path = entry.path(); + if path.is_file() { + out.push(path.to_path_buf()); + } + } + Ok(out) +} + +fn build_eval_graph_from_snapshot(snapshot: &Path) -> Result { + let store = rgctl_graph::SnapshotNodeStore::open(snapshot)?; + let mut graph = EvalGraph::default(); + for id in store.all_node_ids() { + let Some(node) = store.get_node(id)? else { + continue; + }; + if !matches!( + node.node_type, + NodeType::Import + | NodeType::Class + | NodeType::Interface + | NodeType::Enum + | NodeType::Annotation + | NodeType::Function + | NodeType::Module + | NodeType::File + ) { + continue; + } + graph.nodes.push(EvalNode { + id: Some(node.id), + node_type: format!("{:?}", node.node_type), + name: node.name.to_string(), + qualified_name: node.qualified_name.as_ref().map(|s| s.to_string()), + file_path: node.file_path.as_ref().map(|s| s.to_string()), + start_line: node.start_line, + labels: node.labels.clone(), + }); + } + store.for_each_edge(|from, to, et| { + let edge_name = match et { + EdgeType::Extends => "EXTENDS", + EdgeType::Implements => "IMPLEMENTS", + EdgeType::AnnotatedWith => "ANNOTATED_WITH", + _ => return Ok(()), + }; + let Some(from_node) = store.get_node(from)? else { + return Ok(()); + }; + let Some(to_node) = store.get_node(to)? else { + return Ok(()); + }; + graph.edges.push(EvalEdge { + edge_type: edge_name.to_string(), + from_name: from_node.name.to_string(), + from_qualified: from_node + .qualified_name + .as_ref() + .map(|s| s.to_string()) + .unwrap_or_else(|| from_node.name.to_string()), + to_name: to_node.name.to_string(), + to_qualified: to_node + .qualified_name + .as_ref() + .map(|s| s.to_string()) + .unwrap_or_else(|| to_node.name.to_string()), + file_path: from_node + .file_path + .as_ref() + .map(|s| s.to_string()) + .unwrap_or_default(), + line: from_node.start_line.unwrap_or(1), + }); + Ok(()) + })?; + Ok(graph) +} diff --git a/src/cli/session_status.rs b/src/cli/session_status.rs new file mode 100644 index 00000000..932997be --- /dev/null +++ b/src/cli/session_status.rs @@ -0,0 +1,108 @@ +//! Session graph status: cheap freshness / presence check (no rediscover). + +use super::context::CliContext; +use super::OutputFormat; +use anyhow::Result; +use serde::Serialize; +use std::path::Path; + +const STATUS_SCHEMA_VERSION: u32 = 1; + +#[derive(Debug, Serialize)] +struct SessionStatus { + schema_version: u32, + command: &'static str, + /// `ok` | `missing` + status: &'static str, + repo: String, + #[serde(skip_serializing_if = "Option::is_none")] + snapshot: Option, + #[serde(skip_serializing_if = "Option::is_none")] + digest: Option, + #[serde(skip_serializing_if = "Option::is_none")] + nodes: Option, + #[serde(skip_serializing_if = "Option::is_none")] + edges: Option, + kantra_findings: bool, + #[serde(skip_serializing_if = "Option::is_none")] + kantra_findings_path: Option, + #[serde(skip_serializing_if = "Option::is_none")] + message: Option, +} + +fn kantra_findings_path(repo: &Path) -> std::path::PathBuf { + rgctl_graph::paths::artifact_path(repo, "kantra_findings.json") +} + +/// `rgctl status` — report whether `.rgctl/` has a usable snapshot. +pub fn run_status(ctx: &CliContext) -> Result<()> { + let findings = kantra_findings_path(&ctx.repo); + let kantra_present = findings.is_file(); + let kantra_path = kantra_present.then(|| findings.display().to_string()); + + let session = ctx.snapshot_session()?; + let payload = match session { + Some(s) => SessionStatus { + schema_version: STATUS_SCHEMA_VERSION, + command: "status", + status: "ok", + repo: ctx.repo.display().to_string(), + snapshot: Some( + rgctl_graph::paths::artifact_path( + &ctx.repo, + rgctl_graph::snapshot::SNAPSHOT_FILE, + ) + .display() + .to_string(), + ), + digest: Some(s.digest.to_string()), + nodes: Some(s.store.node_count()), + edges: Some(s.store.edge_count()), + kantra_findings: kantra_present, + kantra_findings_path: kantra_path, + message: None, + }, + None => SessionStatus { + schema_version: STATUS_SCHEMA_VERSION, + command: "status", + status: "missing", + repo: ctx.repo.display().to_string(), + snapshot: None, + digest: None, + nodes: None, + edges: None, + kantra_findings: kantra_present, + kantra_findings_path: kantra_path, + message: Some("Graph snapshot not found; run `rgctl discover` first".into()), + }, + }; + + if ctx.format == OutputFormat::Json { + let v = serde_json::to_value(&payload)?; + ctx.emit_json_value(&v)?; + } else if payload.status == "ok" { + ctx.stdout_line(&format!( + "status=ok nodes={} edges={} digest={}{}", + payload.nodes.unwrap_or(0), + payload.edges.unwrap_or(0), + payload.digest.as_deref().unwrap_or("?"), + if payload.kantra_findings { + " kantra_findings=yes" + } else { + "" + } + ))?; + } else { + ctx.stdout_line( + payload + .message + .as_deref() + .unwrap_or("Graph snapshot missing"), + )?; + } + + if payload.status == "missing" { + anyhow::bail!("graph snapshot missing"); + } + Ok(()) +} diff --git a/src/cli/structured_query.rs b/src/cli/structured_query.rs index d49fbfa0..2a561de8 100644 --- a/src/cli/structured_query.rs +++ b/src/cli/structured_query.rs @@ -42,6 +42,8 @@ impl SharedQueryArgs { limit: self.limit, count_only: false, exact: false, + annotation_names: None, + show_attributes: false, }) } } @@ -109,6 +111,8 @@ pub fn run_find( shared: SharedQueryArgs, exact: bool, count_only: bool, + annotation: Option, + show_attributes: bool, ) -> Result<()> { let store = open_store(ctx)?; let node_type = type_name @@ -119,6 +123,14 @@ pub fn run_find( let mut filters = shared.into_filters(node_type)?; filters.exact = exact; filters.count_only = count_only; + filters.show_attributes = show_attributes; + if let Some(raw) = annotation { + let list = rgctl_graph::parse_annotation_list(&raw); + if list.is_empty() { + anyhow::bail!("--annotation requires at least one name (e.g. @MessageDriven)"); + } + filters.annotation_names = Some(list); + } // Default limit for find when not counting if filters.limit.is_none() && !count_only { filters.limit = Some(50); From 9fa81f477fcabe06a12f2a88e5c6c8090bcc316e Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Fri, 2 Oct 2026 15:09:22 +0200 Subject: [PATCH 10/18] add status commands Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- crates/rgctl-extraction/src/graph_builder.rs | 35 ++ crates/rgctl-extraction/src/lib.rs | 2 +- crates/rgctl-graph/src/paths.rs | 8 - crates/rgctl-graph/src/structured_query.rs | 135 ++++++- crates/rgctl-pipeline/Cargo.toml | 1 + crates/rgctl-pipeline/src/pipeline.rs | 14 + docs/Code_structure.md | 4 +- docs/agents/USER_AGENTS_TEMPLATE.md | 2 +- docs/design/blast-radius-design.md | 2 +- docs/faq.md | 2 +- docs/glossary.md | 2 +- docs/guides/discovering-and-indexing.md | 2 +- docs/guides/markdown-context-graph.md | 2 +- docs/installation.md | 10 +- docs/releases/v0.4.9.md | 4 +- skills/rgctl/SKILL.md | 9 +- .../rgctl/references/command-encyclopedia.md | 12 +- src/cli/discover.rs | 74 ++++ src/cli/migrate_cache.rs | 73 ---- src/cli/mod.rs | 44 +-- src/cli/resources.rs | 367 ------------------ src/cli/rules.rs | 16 - src/cli/structured_query.rs | 28 ++ 23 files changed, 309 insertions(+), 539 deletions(-) delete mode 100644 src/cli/migrate_cache.rs delete mode 100644 src/cli/resources.rs diff --git a/crates/rgctl-extraction/src/graph_builder.rs b/crates/rgctl-extraction/src/graph_builder.rs index 612a6e01..97f47655 100644 --- a/crates/rgctl-extraction/src/graph_builder.rs +++ b/crates/rgctl-extraction/src/graph_builder.rs @@ -66,6 +66,16 @@ pub struct GraphBuilder { active_tracker_key: Option, /// Node ids for the active file batch — flushed once in [`Self::end_file_batch`]. active_tracker_ids: Vec, + /// AnnotatedWith argument text keyed for sidecar (columnar edges drop properties). + annotation_args: Vec, +} + +/// One annotation usage with argument text (written to `.rgctl/annotation_args.json`). +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize, PartialEq, Eq)] +pub struct AnnotationArgEntry { + pub source_id: String, + pub annotation: String, + pub arguments: String, } #[derive(Debug, Default)] @@ -733,6 +743,26 @@ impl GraphBuilder { relation.location.start_line.to_string(), ); } + if relation.relation_type == RelationType::AnnotatedWith + && let Some(args) = relation + .metadata + .get("arguments") + .and_then(|v| v.as_str()) + .filter(|s| !s.is_empty()) + { + let ann = relation + .to + .rsplit(['.', '/', ':']) + .next() + .filter(|s| !s.is_empty()) + .unwrap_or(relation.to.as_str()); + edge = edge.with_property("arguments".into(), args.to_string()); + self.annotation_args.push(AnnotationArgEntry { + source_id: from.to_string(), + annotation: ann.to_string(), + arguments: args.to_string(), + }); + } self.commit_edge(edge); } Ok(()) @@ -1216,6 +1246,11 @@ impl GraphBuilder { (self.nodes, self.edges) } + /// Take collected annotation argument entries (for `.rgctl/annotation_args.json`). + pub fn take_annotation_args(&mut self) -> Vec { + std::mem::take(&mut self.annotation_args) + } + /// Finish spill writers and return a [`FinishedSpill`] for columnar compile. pub fn finish_spill(mut self) -> Result { if let Some(err) = self.spill_error.take() { diff --git a/crates/rgctl-extraction/src/lib.rs b/crates/rgctl-extraction/src/lib.rs index 10dfff0b..4c39e255 100644 --- a/crates/rgctl-extraction/src/lib.rs +++ b/crates/rgctl-extraction/src/lib.rs @@ -9,5 +9,5 @@ pub mod usage_detector; pub use discovery::{DiscoveryConfig, FileDiscoverer}; pub use extractor::{ExtractionTail, Extractor, FileExtraction, SymbolPass1Prep}; -pub use graph_builder::GraphBuilder; +pub use graph_builder::{AnnotationArgEntry, GraphBuilder}; pub use manifests::{DependencyDeclaration, extract_manifest}; diff --git a/crates/rgctl-graph/src/paths.rs b/crates/rgctl-graph/src/paths.rs index 9e9435b3..f125f2e0 100644 --- a/crates/rgctl-graph/src/paths.rs +++ b/crates/rgctl-graph/src/paths.rs @@ -166,14 +166,6 @@ pub fn migrate_legacy_daemon_home(home_root: &Path) -> std::io::Result<()> { Ok(()) } -/// Path to cached artifacts for a daemon-era repo name: `~/.rgctl/cache/{name}/.rgctl/`. -pub fn daemon_cache_artifacts(name: &str) -> Option { - let home = legacy_daemon_home()?; - let _ = migrate_legacy_daemon_home(&home); - let root = home.join(".rgctl").join("cache").join(name); - Some(root.join(ARTIFACT_DIR_NAME)) -} - #[cfg(test)] mod tests { use super::*; diff --git a/crates/rgctl-graph/src/structured_query.rs b/crates/rgctl-graph/src/structured_query.rs index c271b942..00e63451 100644 --- a/crates/rgctl-graph/src/structured_query.rs +++ b/crates/rgctl-graph/src/structured_query.rs @@ -143,6 +143,9 @@ pub struct EntityRow { pub line: Option, /// Node UUID (string) pub id: String, + /// Annotation argument text when `--show-attributes` and args are indexed + #[serde(skip_serializing_if = "Option::is_none")] + pub attributes: Option, } /// Keyed edge row (direction-stable; never positional). @@ -267,6 +270,8 @@ pub struct QueryFilters { pub annotation_names: Option>, /// Request annotation argument payloads when indexed (`--show-attributes`). pub show_attributes: bool, + /// Optional map `source_id\0annotation_simple` → arguments (from sidecar). + pub annotation_arg_index: Option>, } /// Session over an open snapshot store. @@ -293,7 +298,36 @@ impl<'a> StructuredQuery<'a> { file: node.file_path.as_ref().map(|s| s.to_string()), line: node.start_line, id: node.id.to_string(), + attributes: None, + } + } + + fn project_with_annotation_attrs( + node: &Node, + annots: &[String], + arg_index: Option<&HashMap>, + ) -> EntityRow { + let mut row = Self::project(node); + if let Some(idx) = arg_index { + let mut parts = Vec::new(); + for a in annots { + let key = format!("{}\0{a}", node.id); + if let Some(args) = idx.get(&key) { + parts.push(format!("@{a}{args}")); + } else { + // Also try bare annotation key variants + let simple = normalize_annotation_name(a); + let key2 = format!("{}\0{simple}", node.id); + if let Some(args) = idx.get(&key2) { + parts.push(format!("@{simple}{args}")); + } + } + } + if !parts.is_empty() { + row.attributes = Some(parts.join("; ")); + } } + row } fn matches_scope(node: &Node, scope: Option<&str>, mode: ScopeMode) -> bool { @@ -491,21 +525,27 @@ impl<'a> StructuredQuery<'a> { /// Entity search (`rgctl find`). pub fn find(&self, pattern: Option<&str>, filters: &QueryFilters) -> Result { if filters.show_attributes { - // Argument indexing is Phase C; refuse inventing empty lookup= fields. - if filters + let has_annots = filters .annotation_names .as_ref() .map(|v| !v.is_empty()) - .unwrap_or(false) + .unwrap_or(false); + if !has_annots { + return Err(Error::InvalidQuery( + "--show-attributes requires --annotation".into(), + )); + } + if filters + .annotation_arg_index + .as_ref() + .map(|m| m.is_empty()) + .unwrap_or(true) { return Err(Error::InvalidQuery( - "annotation attributes are not indexed yet; omit --show-attributes or re-discover after arg indexing lands" + "annotation attributes are not indexed yet; omit --show-attributes or re-discover after a Java index that writes .rgctl/annotation_args.json" .into(), )); } - return Err(Error::InvalidQuery( - "--show-attributes requires --annotation and indexed annotation arguments".into(), - )); } if let Some(annots) = filters.annotation_names.as_ref() { @@ -600,7 +640,43 @@ impl<'a> StructuredQuery<'a> { Ok(()) })?; - Self::finish_find(candidates, filters) + Self::finish_find_annotated(candidates, filters, &normalized) + } + + fn finish_find_annotated( + candidates: Vec, + filters: &QueryFilters, + annots: &[String], + ) -> Result { + let total = candidates.len(); + let limit = filters.limit.unwrap_or(total); + let arg_index = filters.annotation_arg_index.as_ref(); + let entities: Vec = if filters.count_only { + Vec::new() + } else { + candidates + .into_iter() + .take(limit) + .map(|n| { + if filters.show_attributes { + Self::project_with_annotation_attrs(&n, annots, arg_index) + } else { + Self::project(&n) + } + }) + .collect() + }; + let returned = if filters.count_only { + total.min(limit) + } else { + entities.len() + }; + Ok(FindResult { + schema_version: STRUCTURED_QUERY_SCHEMA_VERSION, + returned, + total, + entities, + }) } fn finish_find(candidates: Vec, filters: &QueryFilters) -> Result { @@ -1670,6 +1746,49 @@ mod tests { assert!(msg.contains("not indexed") || msg.contains("attributes")); } + #[test] + fn show_attributes_attaches_from_index() { + let (_dir, store) = sample_store(); + let q = StructuredQuery::new(&store); + // Resolve OrderMDB id from a plain find first. + let base = q + .find( + None, + &QueryFilters { + annotation_names: Some(vec!["MessageDriven".into()]), + node_type: Some(NodeType::Class), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(base.total, 1); + let id = base.entities[0].id.clone(); + let mut idx = HashMap::new(); + idx.insert( + format!("{id}\0MessageDriven"), + "(mappedName=\"jms/orders\")".into(), + ); + let res = q + .find( + None, + &QueryFilters { + annotation_names: Some(vec!["MessageDriven".into()]), + node_type: Some(NodeType::Class), + show_attributes: true, + annotation_arg_index: Some(idx), + ..Default::default() + }, + ) + .unwrap(); + assert!( + res.entities[0] + .attributes + .as_deref() + .unwrap_or("") + .contains("jms/orders") + ); + } + #[test] fn inventory_import_prefix_census() { let (_dir, store) = sample_store(); diff --git a/crates/rgctl-pipeline/Cargo.toml b/crates/rgctl-pipeline/Cargo.toml index 8b24bbeb..98703498 100644 --- a/crates/rgctl-pipeline/Cargo.toml +++ b/crates/rgctl-pipeline/Cargo.toml @@ -16,6 +16,7 @@ rayon = "1.7" crossbeam = "0.8" tracing = "0.1" uuid = { version = "1", features = ["v4", "serde"] } +serde_json = "1" [dev-dependencies] tempfile = { workspace = true } diff --git a/crates/rgctl-pipeline/src/pipeline.rs b/crates/rgctl-pipeline/src/pipeline.rs index 0f5dbf83..409eed89 100644 --- a/crates/rgctl-pipeline/src/pipeline.rs +++ b/crates/rgctl-pipeline/src/pipeline.rs @@ -189,9 +189,23 @@ impl ProcessingPipeline { let edges_created = builder.edge_count(); let content_store = builder.take_content_store(); let node_path_mapping = builder.take_tracker_mapping(); + let annotation_args = builder.take_annotation_args(); let spill_start = Instant::now(); let finished = builder.finish_spill()?; let digest = write_columnar_from_spill(finished, snapshot_path)?; + if !annotation_args.is_empty() + && let Some(parent) = snapshot_path.parent() + { + let args_path = parent.join("annotation_args.json"); + let payload = serde_json::json!({ + "schema_version": 1, + "command": "annotation_args", + "entries": annotation_args, + }); + if let Ok(json) = serde_json::to_string_pretty(&payload) { + let _ = std::fs::write(&args_path, json); + } + } let spill_elapsed = spill_start.elapsed(); tracing::info!( resolution_index_secs = index_elapsed.as_secs_f64(), diff --git a/docs/Code_structure.md b/docs/Code_structure.md index f050e3b6..5b83613c 100644 --- a/docs/Code_structure.md +++ b/docs/Code_structure.md @@ -138,7 +138,7 @@ flowchart TB - **`src/main.rs`** — process entry, dispatches to CLI. - **`src/cli/`** — subcommands: `discover`, `blast-radius`, `serve`, `gql`, `slice`, `inspect`, `metrics`, `semantic`, `communities`, `cpg`, `check`, `export`. - **`src/cli/http_serve.rs`** — `serve`: dashboard + `POST /api/query` (foreground HTTP for one repo). -- **`src/cli/migrate_cache.rs`** — `migrate-cache`: copy legacy `~/.rgctl/cache/` into in-repo `.rgctl/`. +- **`src/cli/session_status.rs`** — `status`: cheap snapshot / digest / node-edge summary. - **`src/cli/*_output.rs`** — typed JSON serializers (`blast_radius_output`, `discover_output`, `gql_output`, …). Commands assemble domain results from workspace crates and serialize here; **do not** embed algorithm logic in output modules. - **`src/languages/`** — wires the active language **bundle** into a `LanguageRegistry` at runtime. - Re-exports **`rgctl-core`** for library users (`use rgctl::analysis`, etc.). @@ -228,7 +228,7 @@ Single home for **graph algorithms and semantic analysis**: | `discover` | `pipeline`, `extraction`, `registry`, `graph`, `analysis`, `incremental`, `export`, `project-config`; stdout JSON via `discover_output` when `-f json` | | `blast-radius` | `analysis` (engine + macro index + depth filter), `graph` (columnar snapshot mmap); CLI orchestration in `blast_radius.rs` | | `serve` | `http_serve` — foreground HTTP dashboard + `/api/query` | -| `migrate-cache` | `rgctl-graph` paths + filesystem copy from legacy daemon cache | +| `status` | Session graph presence / digest / counts | | `gql` | `gql`, `graph` | | `slice` | `analysis` (CFG, PDG, slicing), reads source from disk | | `inspect` | `graph`, `analysis` | diff --git a/docs/agents/USER_AGENTS_TEMPLATE.md b/docs/agents/USER_AGENTS_TEMPLATE.md index a25d1e68..0b31d181 100644 --- a/docs/agents/USER_AGENTS_TEMPLATE.md +++ b/docs/agents/USER_AGENTS_TEMPLATE.md @@ -48,7 +48,7 @@ export REPO=/path/to/repo rgctl -r "$REPO" -f json find --type function --limit 20 ``` -Upgrading from an old daemon install: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/` into the repo (see [installation.md](../installation.md)). +Upgrading from an old daemon install: copy `~/.rgctl/cache/{name}/.rgctl/` into the repo as `.rgctl/` manually, or re-run `rgctl discover .` (see [installation.md](../installation.md)). --- diff --git a/docs/design/blast-radius-design.md b/docs/design/blast-radius-design.md index a6f66019..dd3b3e22 100644 --- a/docs/design/blast-radius-design.md +++ b/docs/design/blast-radius-design.md @@ -52,7 +52,7 @@ flowchart TB **Query tiers** (`src/cli/blast_radius.rs`): T0 blast lookup cache hit → in-process mmap engine path → full hydrate for `--with-slices` / `--policy-file`. -> **Retired:** Background daemon mode and the per-repo `query.sock` blast client are removed. All queries run in-process against `{repo}/.rgctl/`. Legacy daemon caches: `rgctl migrate-cache`. +> **Retired:** Background daemon mode and the per-repo `query.sock` blast client are removed. All queries run in-process against `{repo}/.rgctl/`. --- diff --git a/docs/faq.md b/docs/faq.md index 0a73f446..c2acb8e4 100644 --- a/docs/faq.md +++ b/docs/faq.md @@ -4,7 +4,7 @@ Short answers for common first-hour questions. Commands → [User Guide](user-gu ### Where are `.rgctl/` artifacts stored? -**Default:** `{repo}/.rgctl/` next to the source tree. Legacy background-daemon caches under `~/.rgctl/cache/{reponame}/` can be copied with `rgctl migrate-cache`. See [Installation — Migrating from daemon cache](installation.md#migrating-from-daemon-cache). +**Default:** `{repo}/.rgctl/` next to the source tree. See [Installation](installation.md). ### I ran `discover` but queried the wrong repo diff --git a/docs/glossary.md b/docs/glossary.md index d87323f2..3b202133 100644 --- a/docs/glossary.md +++ b/docs/glossary.md @@ -7,7 +7,7 @@ | **CPG** | Code Property Graph — hybrid of L_repo (CALL/type) and L_proc (CFG/PDG). CLI: `cpg`. | | **CFG** | Control-flow graph of a function (basic blocks and branches). | | **Discover** | Index a repository into `{repo}/.rgctl/` artifacts. | -| **migrate-cache** | Copy legacy `~/.rgctl/cache/{name}/.rgctl/` into the current repo. | +| **`.rgctl/`** | In-repo artifact directory written by `discover` (snapshots, findings, dashboard, …). | | **Fusion** | Late re-ranking of semantic hits with graph signals (blast, PageRank, sketches). | | **GQL** | rgctl graph query language (`MATCH` / macros) over the knowledge graph. | | **Hamming distance** | Bitwise distance used for packed semantic embedding retrieval. | diff --git a/docs/guides/discovering-and-indexing.md b/docs/guides/discovering-and-indexing.md index dd6a99e9..039a87f3 100644 --- a/docs/guides/discovering-and-indexing.md +++ b/docs/guides/discovering-and-indexing.md @@ -29,7 +29,7 @@ This guide uses the **CoolStore** — a Java EE e-commerce application. It lives **Pitfall:** `rgctl -r example/coolstore discover` does **not** index `example/coolstore`. The positional `.` becomes the session root (usually your **shell cwd**), so `-r` is ignored. That can scan the wrong tree and fail on large parent directories. -**Artifacts:** `discover` writes snapshots under **`{repo}/.rgctl/`**. Add `.rgctl/` to `.gitignore`. Legacy daemon caches under `~/.rgctl/cache/` can be copied with `rgctl migrate-cache`. +**Artifacts:** `discover` writes snapshots under **`{repo}/.rgctl/`**. Add `.rgctl/` to `.gitignore`. ## Step-by-Step diff --git a/docs/guides/markdown-context-graph.md b/docs/guides/markdown-context-graph.md index d5755d9b..f10a4ff2 100644 --- a/docs/guides/markdown-context-graph.md +++ b/docs/guides/markdown-context-graph.md @@ -41,7 +41,7 @@ Maintainers who already fetch profile corpora can skip the clone: `./scripts/fet **Prerequisites:** `rgctl` on your `PATH`. For large exports (17k+ Obsidian notes), use a **release** binary — [download the latest release](https://github.com/sshaaf/rgctl/releases) or [build from source](../installation.md) (`cargo build --release --bin rgctl`). See [Installation](../installation.md). -Artifacts are written to `{repo}/.rgctl/` next to the checkout. If you still have a legacy daemon cache under `~/.rgctl/cache/`, run `rgctl migrate-cache`. See [Installation — Migrating from daemon cache](../installation.md#migrating-from-daemon-cache). +Artifacts are written to `{repo}/.rgctl/` next to the checkout. See [Installation](../installation.md). ## What rgctl indexes diff --git a/docs/installation.md b/docs/installation.md index 210224b9..13d51e68 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -219,13 +219,7 @@ See the [HTTP Server and Dashboard guide](guides/http-server-and-dashboard.md) a ### Migrating from daemon cache -If you previously used the background daemon, artifacts may still be under `~/.rgctl/cache/{reponame}/.rgctl/`. Copy them into the repo: - -```bash -cd /path/to/repo -rgctl migrate-cache # uses repo directory name as cache key -rgctl migrate-cache --name coolstore --force # explicit cache name -``` +Background daemon mode is retired. Artifacts live only under `{repo}/.rgctl/`. If you still have files under `~/.rgctl/cache/{reponame}/.rgctl/`, copy that directory into the repo as `.rgctl/` manually (or re-run `rgctl discover .`). --- @@ -364,7 +358,7 @@ If empty, revisit [Add to PATH](#add-to-path). For GUI apps (Cursor, VS Code), n ### Queries fail with "no graph found" -Run `discover` first on the repo you mean to query. Artifacts should appear at `{repo}/.rgctl/`. If you still have a legacy daemon cache, run `rgctl migrate-cache`. +Run `discover` first on the repo you mean to query. Artifacts should appear at `{repo}/.rgctl/`. Old daemon caches under `~/.rgctl/cache/` can be copied into the repo by hand, or just rediscover. ### Slow `discover` on large repositories diff --git a/docs/releases/v0.4.9.md b/docs/releases/v0.4.9.md index 6cc4c6f2..5145c9e6 100644 --- a/docs/releases/v0.4.9.md +++ b/docs/releases/v0.4.9.md @@ -40,7 +40,7 @@ Background daemon mode and all daemon routing are **removed**. Every command run | `--no-daemon`, `--daemon-home`, `--fail-if-no-daemon` | Default is always in-repo artifacts | | `serve --daemon`, `serve --mode mcp` | `rgctl serve` (foreground HTTP + dashboard on one repo) | | MCP tools (`rgctl_status`, `rgctl_query`, …) | `rgctl -f json ` subprocesses ([AGENTS.md](../../AGENTS.md)) | -| `~/.rgctl/cache/{name}/.rgctl/` daemon cache | `rgctl migrate-cache` copies legacy cache into the repo | +| `~/.rgctl/cache/{name}/.rgctl/` daemon cache | Copy into `{repo}/.rgctl/` manually or re-run `discover` | **Agent workflow:** `discover` once → spawn `rgctl -f json gql|blast-radius|…` (or long-lived `rgctl serve` for HTTP). Re-run **`rgctl install --skill --force`** for updated skill + `AGENTS.md`. @@ -65,7 +65,7 @@ Restored `scripts/build-dashboard.sh` and `rgctl_wasm_*` asset naming; bundle te ## Breaking changes (upgrade from v0.4.8) -1. **Daemon and MCP removed** — see migration above; run `rgctl migrate-cache` if you still have `~/.rgctl/cache/`. +1. **Daemon and MCP removed** — see migration above; re-run `discover` (or copy `~/.rgctl/cache/…/.rgctl/` into the repo by hand) if you still have a legacy cache. 2. **Artifacts in repo** — add `.rgctl/` to `.gitignore`; expect `{repo}/.rgctl/` after `discover`. 3. **MCP configs obsolete** — remove `.cursor/mcp.json` / Claude MCP entries for `serve --mode mcp`; use CLI subprocesses or HTTP `serve`. 4. **Refresh agent skill** — `rgctl install --skill --force`. diff --git a/skills/rgctl/SKILL.md b/skills/rgctl/SKILL.md index 75743771..748246dd 100644 --- a/skills/rgctl/SKILL.md +++ b/skills/rgctl/SKILL.md @@ -39,7 +39,7 @@ rgctl -r "$REPO" -f json … For many queries in one session, optional: `rgctl serve` + `POST /api/query` (see [HTTP API](../../docs/http-api.md)). -Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/` into the repo. +Legacy daemon cache under `~/.rgctl/cache/` is obsolete; run `rgctl discover .` in the repo to build `{repo}/.rgctl/`. ## Agent Loop @@ -93,7 +93,7 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | Evaluate migration rules | `discover . --with-kantra` or `rules run ./rules/` | | Filter by migration target | `discover . --with-kantra --kantra-target quarkus` / `rules run ./rules/ --target quarkus` | | CI / custom ruleset | `discover . --with-kantra --kantra-rules PATH` / `rules run PATH` | -| Index rules only | `discover . --with-kantra --kantra-index-only` / `rules run PATH --index-only` | +| Index rules only | `discover . --with-kantra --kantra-index-only` | | List indexed rules (GQL) | `gql "MATCH (r:KantraRule) RETURN r LIMIT 20"` | | Rules for one target label | `gql` with `` r.`konveyor.io/target` `` property (backticks) | | Read violations artifact | `.rgctl/kantra_findings.json` | @@ -115,7 +115,6 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | Classes with annotation | `find --annotation @MessageDriven --type class` | | javax import worklist | `find "import javax*" --type import --scope ` | | Annotation pairs (seedless) | `relations --edge annotatedwith --from-type function --to-type annotation --scope ` | -| Deployment / persistence config | `resources` (persistence.xml, weblogic/jboss/web/beans) | | Find callers/callees | `callers --depth 1` / `callees ` | | Outside callers of a module | `callers --scope --scope-mode outside` | | EXTENDS / IMPLEMENTS inventory | `relations --edge extends --from-type class` (omit SYMBOL) | @@ -125,7 +124,7 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | Refresh community labels | `communities label --write` | | Ad-hoc Cypher (experimental) | `gql "MATCH …"` — uncanny valley; prefer verbs above | -**Migration probe order:** `status` → `inventory --by import-prefix` → `find --annotation …` / suffix globs → `resources` → `rules run` / `--with-kantra` → `callers InitialContext`. +**Migration probe order:** `status` → `inventory --by import-prefix` → `find --annotation …` / suffix globs → `rules run` / `--with-kantra` → `callers InitialContext`. **Complexity honesty:** exact name = hash index; prefix/`*mid*`/`--scope` may scan keys/columns until better indexes land. Module re-index is still a strong speed lever. Annotation **arguments** (e.g. `@Path("/x")`) are not indexed yet (`--show-attributes` errors until they are). **See:** [Command Encyclopedia](references/command-encyclopedia.md) (find/callers/relations/inventory), [GQL Reference](references/gql-reference.md) (legacy), [Semantic Search Guide](../../docs/guides/semantic-search.md) @@ -187,7 +186,7 @@ Needs `discover --with-cfg`. `--function` is method name, not class. | Symptom | Fix | |---------|-----| -| No `.rgctl/` in repo | Run `cd repo && rgctl discover .`; or `rgctl migrate-cache` from legacy daemon cache | +| No `.rgctl/` in repo | Run `cd repo && rgctl discover .` | | slice/inspect/cpg fails | Re-discover with `--with-cfg` | | semantic query fails | `semantic index` | | Ambiguous symbol | Add `--class` or `--file` on callers/find | diff --git a/skills/rgctl/references/command-encyclopedia.md b/skills/rgctl/references/command-encyclopedia.md index 19afe33b..37c50f93 100644 --- a/skills/rgctl/references/command-encyclopedia.md +++ b/skills/rgctl/references/command-encyclopedia.md @@ -118,8 +118,9 @@ rgctl -f json inventory --by type # includes zero-count kinds rgctl -f json inventory --by edge rgctl -f json inventory --by import-prefix # javax.ejb / javax.jms / org.eclipse … rgctl -f json status # snapshot presence, digest, node/edge counts -rgctl -f json resources # persistence.xml + weblogic/jboss/web/beans (no Kantra) +rgctl discover . --find '*coolstore*' # locate candidate project roots (no index) rgctl -f json rules run ./rules/ [--target quarkus] # post-index Kantra eval +rgctl -f json find --annotation @Resource --show-attributes # needs annotation_args.json from discover rgctl -f json query find … # alias namespace ``` @@ -130,15 +131,14 @@ rgctl -f json query find … # alias namespace 2. `inventory --by import-prefix` — EE surface census 3. `find --annotation @MessageDriven|@SessionScoped|…` — blockers without package guess 4. `find '*MDB*'` / `'*Remote*'` — suffix scan before reading files -5. `resources` — persistence provider, JNDI DS, weblogic/jboss bindings -6. `rules run ./rules/` or `discover --with-kantra` — fire `when:` catalog (M2) -7. `callers InitialContext` — JNDI usage sites +5. `rules run ./rules/` or `discover --with-kantra` — fire `when:` catalog (M2) +6. `callers InitialContext` — JNDI usage sites -**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). `resources` / `status` do not rediscover. `rules run` requires a snapshot; Kantra stays opt-in. +**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). `status` does not rediscover. `rules run` requires a snapshot; Kantra stays opt-in. **Flags:** `--annotation` inverts `AnnotatedWith` (OR list; `@` optional). `--show-attributes` needs annotation-arg indexing (errors honestly until indexed). `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). `--file` / `--class` / `--line` disambiguate. Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. -**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. Do not treat Kantra as the only search path — use annotation/import/resources first. +**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. Do not treat Kantra as the only search path — use annotation/import first. **Agent should report:** counts, lean names/files, keyed edge pairs — not full node dumps. diff --git a/src/cli/discover.rs b/src/cli/discover.rs index 8851918f..e4561e57 100644 --- a/src/cli/discover.rs +++ b/src/cli/discover.rs @@ -16,6 +16,8 @@ use std::sync::Arc; #[derive(Clone)] pub struct DiscoverArgs { pub path: Option, + /// When set, list matching project-root candidates and exit (no index). + pub find_roots: Option, pub languages: Option, pub exclude: Option, /// Secret scanning. Default off. @@ -89,6 +91,10 @@ pub fn run(ctx: &CliContext, args: DiscoverArgs) -> Result<()> { let limits = super::discover_limits::DiscoverLimits::from_cli(args.with_limits.as_deref())?; let path = resolve_session_root(ctx, args.path.as_deref()); + if let Some(pat) = args.find_roots.as_deref() { + return run_find_roots(ctx, &path, pat); + } + if let Some(files) = &args.files { return run_files_update(ctx, &path, files.clone(), &args); } @@ -196,3 +202,71 @@ fn run_files_update(ctx: &CliContext, path: &str, files: Vec, args: &Dis } Ok(()) } + +/// List directories under `root` whose path/name matches a glob (project-root locator). +fn run_find_roots(ctx: &CliContext, root: &str, pattern: &str) -> Result<()> { + let root_path = Path::new(root); + let mut hits: Vec = Vec::new(); + let markers = [ + "pom.xml", + "build.gradle", + "build.gradle.kts", + "Cargo.toml", + "package.json", + "go.mod", + "settings.gradle", + ]; + for entry in ignore::WalkBuilder::new(root_path) + .max_depth(Some(6)) + .git_ignore(true) + .build() + .flatten() + { + let path = entry.path(); + if !path.is_dir() { + continue; + } + let name = path + .file_name() + .and_then(|s| s.to_str()) + .unwrap_or(""); + let rel = path + .strip_prefix(root_path) + .unwrap_or(path) + .to_string_lossy() + .replace('\\', "/"); + let name_ok = rgctl_graph::glob_match(pattern, name); + let rel_ok = rgctl_graph::glob_match(pattern, &rel); + if !(name_ok || rel_ok) { + continue; + } + let looks_like_project = markers.iter().any(|m| path.join(m).is_file()) + || path.join("src").is_dir() + || path.join("pom.xml").is_file(); + if looks_like_project || name_ok { + hits.push(if rel.is_empty() { + ".".into() + } else { + rel + }); + } + } + hits.sort(); + hits.dedup(); + if ctx.format == OutputFormat::Json { + ctx.emit_json_value(&serde_json::json!({ + "schema_version": 1, + "command": "discover_find", + "pattern": pattern, + "roots": hits, + "returned": hits.len(), + }))?; + } else if hits.is_empty() { + ctx.stdout_line("discover --find: (no matching project roots)")?; + } else { + for h in hits { + ctx.stdout_line(&h)?; + } + } + Ok(()) +} diff --git a/src/cli/migrate_cache.rs b/src/cli/migrate_cache.rs deleted file mode 100644 index 0729057a..00000000 --- a/src/cli/migrate_cache.rs +++ /dev/null @@ -1,73 +0,0 @@ -//! `rgctl migrate-cache` — copy daemon-era cache artifacts into a repo tree. - -use super::context::CliContext; -use anyhow::{Context, Result, bail}; -use std::fs; -use std::path::{Path, PathBuf}; - -pub struct MigrateCacheArgs { - pub name: Option, - pub from: Option, - pub force: bool, -} - -pub fn run(ctx: &CliContext, args: MigrateCacheArgs) -> Result<()> { - let repo = ctx.repo.canonicalize().unwrap_or_else(|_| ctx.repo.clone()); - let dest = rgctl_graph::paths::artifact_dir(&repo); - if dest.exists() && !args.force { - bail!( - "destination {} already exists (pass --force to overwrite)", - dest.display() - ); - } - - let source = match args.from { - Some(p) => p, - None => { - let name = args - .name - .or_else(|| { - repo.file_name() - .and_then(|s| s.to_str()) - .map(str::to_string) - }) - .context("pass --name or use a repo path with a directory name")?; - rgctl_graph::paths::daemon_cache_artifacts(&name) - .with_context(|| format!("cannot resolve cache for {name:?} (set RGCTL_HOME or HOME)"))? - } - }; - - if !source.is_dir() { - bail!("cache source not found: {}", source.display()); - } - - if dest.exists() { - fs::remove_dir_all(&dest) - .with_context(|| format!("remove {}", dest.display()))?; - } - if let Some(parent) = dest.parent() { - fs::create_dir_all(parent)?; - } - copy_dir_recursive(&source, &dest)?; - eprintln!( - "[rgctl] migrated cache {} → {}", - source.display(), - dest.display() - ); - Ok(()) -} - -fn copy_dir_recursive(from: &Path, to: &Path) -> Result<()> { - fs::create_dir_all(to)?; - for entry in fs::read_dir(from)? { - let entry = entry?; - let ty = entry.file_type()?; - let dest_path = to.join(entry.file_name()); - if ty.is_dir() { - copy_dir_recursive(&entry.path(), &dest_path)?; - } else if ty.is_file() { - fs::copy(entry.path(), &dest_path)?; - } - } - Ok(()) -} diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 80691b2a..02700bf9 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -32,7 +32,6 @@ pub mod metrics_output; mod pipeline_session; pub mod pipeline_status; mod policy_file; -mod migrate_cache; mod semantic; mod semantic_api; pub mod semantic_output; @@ -41,7 +40,6 @@ pub mod slice_output; mod stage_profile; mod structured_query; mod session_status; -mod resources; mod rules; pub use args::OutputFormat; @@ -102,6 +100,10 @@ pub enum Commands { #[arg(value_name = "PATH")] path: Option, + /// Locate candidate project roots under PATH matching a glob (e.g. `*coolstore*`); no full index + #[arg(long = "find", value_name = "GLOB")] + find_roots: Option, + #[arg(short = 'l', long = "languages")] languages: Option, @@ -416,13 +418,6 @@ pub enum Commands { /// Session graph status (snapshot presence, digest, node/edge counts; no rediscover) Status, - /// Parse deployment/persistence descriptors (persistence.xml, weblogic/jboss/web/beans) - Resources { - /// Optional extra descriptor file to include - #[arg(long = "file", value_name = "PATH")] - file: Option, - }, - /// Evaluate Konveyor-shaped rules against the session (Kantra engine) Rules { #[command(subcommand)] @@ -646,21 +641,6 @@ pub enum Commands { dashboard_only: bool, }, - /// Copy daemon-era cache artifacts into `{repo}/.rgctl/` - MigrateCache { - /// Cache entry name under `~/.rgctl/cache/` (default: repo directory name) - #[arg(long)] - name: Option, - - /// Explicit cache `.rgctl/` source directory - #[arg(long, value_name = "PATH")] - from: Option, - - /// Overwrite existing `{repo}/.rgctl/` - #[arg(long)] - force: bool, - }, - /// Diff two columnar graph snapshots (cold diff profiling / compare path) Diff { /// Base snapshot file or directory containing `graph.snapshot.bin` @@ -720,10 +700,6 @@ pub enum RulesCommands { #[arg(long = "target", value_name = "NAME")] target: Option, - /// Index KantraRule nodes only; skip evaluation - #[arg(long = "index-only")] - index_only: bool, - /// Override with a rulesets tree (mutually exclusive with DIR as single ruleset when set) #[arg(long = "catalog", value_name = "ROOT")] catalog: Option, @@ -1118,6 +1094,7 @@ impl Cli { let result = match self.command { Commands::Discover { path, + find_roots, languages, exclude, verbose: _, @@ -1150,6 +1127,7 @@ impl Cli { &ctx, discover::DiscoverArgs { path, + find_roots, languages, exclude: join_exclude_patterns(&exclude), with_security, @@ -1368,14 +1346,12 @@ impl Cli { }, ), Commands::Status => session_status::run_status(&ctx), - Commands::Resources { file } => resources::run_resources(&ctx, file), Commands::Rules { action } => match action { RulesCommands::Run { rules_dir, target, - index_only, catalog, - } => rules::run_rules(&ctx, rules_dir, target, index_only, catalog), + } => rules::run_rules(&ctx, rules_dir, target, catalog), }, Commands::Query { action } => match action { QueryCommands::Find { @@ -1774,10 +1750,6 @@ impl Cli { force, }, ), - Commands::MigrateCache { name, from, force } => migrate_cache::run( - &ctx, - migrate_cache::MigrateCacheArgs { name, from, force }, - ), Commands::Diff { base, head } => diff::run( &ctx, diff::DiffArgs { base, head }, @@ -1824,7 +1796,6 @@ fn command_label_for(command: &Commands) -> &'static str { Commands::Relations { .. } => "relations", Commands::Inventory { .. } => "inventory", Commands::Status => "status", - Commands::Resources { .. } => "resources", Commands::Rules { .. } => "rules", Commands::Query { action } => match action { QueryCommands::Find { .. } => "query find", @@ -1861,7 +1832,6 @@ fn command_label_for(command: &Commands) -> &'static str { Commands::PrCheck { .. } => "pr-check", Commands::Export { .. } => "export", Commands::Install { .. } => "install", - Commands::MigrateCache { .. } => "migrate-cache", Commands::Diff { .. } => "diff", Commands::Serve { .. } => "serve", } diff --git a/src/cli/resources.rs b/src/cli/resources.rs deleted file mode 100644 index 74625237..00000000 --- a/src/cli/resources.rs +++ /dev/null @@ -1,367 +0,0 @@ -//! `rgctl resources` — first-party parse of JEE descriptors (no Kantra). - -use super::context::CliContext; -use super::OutputFormat; -use anyhow::Result; -use serde::Serialize; -use std::fs; -use std::path::{Path, PathBuf}; - -const RESOURCES_SCHEMA_VERSION: u32 = 1; - -#[derive(Debug, Default, Serialize)] -struct ResourcesResult { - schema_version: u32, - command: &'static str, - persistence: Vec, - datasources: Vec, - jms: Vec, - ejb_bindings: Vec, - cdi: Vec, - files_scanned: Vec, -} - -#[derive(Debug, Clone, Serialize)] -struct PersistenceUnit { - file: String, - name: Option, - provider: Option, - jta_data_source: Option, - non_jta_data_source: Option, - schema_generation: Vec, -} - -#[derive(Debug, Clone, Serialize)] -struct NamedBinding { - kind: String, - name: String, - #[serde(skip_serializing_if = "Option::is_none")] - jndi: Option, - file: String, -} - -#[derive(Debug, Clone, Serialize)] -struct CdiHint { - file: String, - bean_discovery_mode: Option, -} - -/// `rgctl resources` -pub fn run_resources(ctx: &CliContext, extra_file: Option) -> Result<()> { - let mut out = ResourcesResult { - schema_version: RESOURCES_SCHEMA_VERSION, - command: "resources", - ..Default::default() - }; - - let mut files = discover_descriptor_files(&ctx.repo)?; - if let Some(f) = extra_file { - files.push(PathBuf::from(f)); - } - - for path in &files { - let rel = path - .strip_prefix(&ctx.repo) - .unwrap_or(path) - .to_string_lossy() - .replace('\\', "/"); - out.files_scanned.push(rel.clone()); - let Ok(text) = fs::read_to_string(path) else { - continue; - }; - let name = path - .file_name() - .and_then(|s| s.to_str()) - .unwrap_or("") - .to_ascii_lowercase(); - if name == "persistence.xml" || rel.ends_with("META-INF/persistence.xml") { - parse_persistence(&text, &rel, &mut out.persistence); - } else if name.starts_with("weblogic") || name.starts_with("jboss-") { - parse_server_bindings(&text, &rel, &mut out); - } else if name == "web.xml" { - parse_web_xml(&text, &rel, &mut out); - } else if name == "beans.xml" { - parse_beans_xml(&text, &rel, &mut out.cdi); - } - } - - if ctx.format == OutputFormat::Json { - let v = serde_json::to_value(&out)?; - ctx.emit_json_value(&v)?; - } else { - ctx.stdout_line(&format!( - "resources: {} persistence, {} datasources, {} jms, {} ejb, {} cdi ({} files)", - out.persistence.len(), - out.datasources.len(), - out.jms.len(), - out.ejb_bindings.len(), - out.cdi.len(), - out.files_scanned.len() - ))?; - for p in &out.persistence { - ctx.stdout_line(&format!( - " persistence provider={} jta={:?}", - p.provider.as_deref().unwrap_or("?"), - p.jta_data_source - ))?; - } - } - Ok(()) -} - -fn discover_descriptor_files(repo: &Path) -> Result> { - let mut out = Vec::new(); - let walker = ignore::WalkBuilder::new(repo) - .hidden(false) - .git_ignore(true) - .build(); - for entry in walker.flatten() { - let path = entry.path(); - if !path.is_file() { - continue; - } - let name = path - .file_name() - .and_then(|s| s.to_str()) - .unwrap_or("") - .to_ascii_lowercase(); - let ok = name == "persistence.xml" - || name == "web.xml" - || name == "beans.xml" - || (name.starts_with("weblogic") && name.ends_with(".xml")) - || (name.starts_with("jboss-") && name.ends_with(".xml")); - if ok { - out.push(path.to_path_buf()); - } - } - out.sort(); - Ok(out) -} - -fn parse_persistence(text: &str, file: &str, out: &mut Vec) { - // Lightweight tag scrape (namespaces ignored). - for unit in split_tags(text, "persistence-unit") { - let name = attr_value(&unit, "name"); - let provider = tag_text(&unit, "provider"); - let jta = tag_text(&unit, "jta-data-source"); - let non_jta = tag_text(&unit, "non-jta-data-source"); - let mut schema = Vec::new(); - for prop in split_tags(&unit, "property") { - if let Some(n) = attr_value(&prop, "name") { - if n.contains("schema-generation") || n.contains("ddl") { - let v = attr_value(&prop, "value").unwrap_or_default(); - schema.push(format!("{n}={v}")); - } - } - } - out.push(PersistenceUnit { - file: file.into(), - name, - provider, - jta_data_source: jta, - non_jta_data_source: non_jta, - schema_generation: schema, - }); - } -} - -fn parse_server_bindings(text: &str, file: &str, out: &mut ResourcesResult) { - // weblogic / jboss: capture common JNDI-ish attributes and resource-ref names. - for (tag, kind) in [ - ("resource-description", "resource"), - ("resource-env-description", "resource-env"), - ("ejb-local-reference-description", "ejb"), - ("ejb-reference-description", "ejb"), - ("message-destination-description", "jms"), - ("connection-factory", "jms-factory"), - ("topic", "jms-topic"), - ("queue", "jms-queue"), - ("datasource", "datasource"), - ] { - for block in split_tags(text, tag) { - let name = tag_text(&block, "res-ref-name") - .or_else(|| tag_text(&block, "resource-env-ref-name")) - .or_else(|| tag_text(&block, "ejb-ref-name")) - .or_else(|| attr_value(&block, "name")) - .unwrap_or_else(|| tag.to_string()); - let jndi = tag_text(&block, "jndi-name") - .or_else(|| tag_text(&block, "lookup-name")) - .or_else(|| attr_value(&block, "jndi-name")); - let binding = NamedBinding { - kind: kind.into(), - name, - jndi, - file: file.into(), - }; - match kind { - "jms" | "jms-factory" | "jms-topic" | "jms-queue" => out.jms.push(binding), - "ejb" => out.ejb_bindings.push(binding), - "datasource" => out.datasources.push(binding), - _ => { - if binding - .jndi - .as_deref() - .unwrap_or("") - .contains("jdbc") - || binding.name.contains("jdbc") - || binding.name.contains("DataSource") - || binding.name.contains("DS") - { - out.datasources.push(binding); - } else if binding.name.contains("jms") - || binding - .jndi - .as_deref() - .unwrap_or("") - .contains("jms") - { - out.jms.push(binding); - } else { - out.ejb_bindings.push(binding); - } - } - } - } - } - // Coolstore-shaped: plain jdbc/CoolstoreDS - if out.datasources.is_empty() { - if let Some(jndi) = tag_text(text, "jndi-name") { - if jndi.contains("jdbc") || jndi.contains("DS") { - out.datasources.push(NamedBinding { - kind: "jndi".into(), - name: jndi.clone(), - jndi: Some(jndi), - file: file.into(), - }); - } - } - } -} - -fn parse_web_xml(text: &str, file: &str, out: &mut ResourcesResult) { - for block in split_tags(text, "resource-ref") { - let name = tag_text(&block, "res-ref-name").unwrap_or_else(|| "resource-ref".into()); - let jndi = tag_text(&block, "lookup-name"); - out.datasources.push(NamedBinding { - kind: "resource-ref".into(), - name, - jndi, - file: file.into(), - }); - } -} - -fn parse_beans_xml(text: &str, file: &str, out: &mut Vec) { - let mode = attr_value(text, "bean-discovery-mode").or_else(|| { - // sometimes on beans root - text.find("bean-discovery-mode=\"") - .and_then(|i| { - let rest = &text[i + "bean-discovery-mode=\"".len()..]; - rest.split('"').next().map(|s| s.to_string()) - }) - }); - out.push(CdiHint { - file: file.into(), - bean_discovery_mode: mode, - }); -} - -fn split_tags(hay: &str, tag: &str) -> Vec { - let open = format!("<{tag}"); - let close = format!(""); - let mut out = Vec::new(); - let mut rest = hay; - while let Some(start) = rest.find(&open) { - let from = &rest[start..]; - if let Some(end) = from.find(&close) { - out.push(from[..end + close.len()].to_string()); - rest = &from[end + close.len()..]; - } else { - // self-closing or truncated - if let Some(gt) = from.find('>') { - out.push(from[..=gt].to_string()); - rest = &from[gt + 1..]; - } else { - break; - } - } - } - out -} - -fn tag_text(hay: &str, tag: &str) -> Option { - let open = format!("<{tag}"); - let close = format!(""); - let start = hay.find(&open)?; - let after_open = &hay[start..]; - let gt = after_open.find('>')?; - let body_start = &after_open[gt + 1..]; - let end = body_start.find(&close)?; - Some(body_start[..end].trim().to_string()) -} - -fn attr_value(hay: &str, attr: &str) -> Option { - let key = format!("{attr}=\""); - let i = hay.find(&key)?; - let rest = &hay[i + key.len()..]; - let end = rest.find('"')?; - Some(rest[..end].to_string()) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn parse_persistence_eclipselink_coolstore() { - let xml = r#" - - - org.eclipse.persistence.jpa.PersistenceProvider - jdbc/CoolstoreDS - - - - - "#; - let mut units = Vec::new(); - parse_persistence(xml, "META-INF/persistence.xml", &mut units); - assert_eq!(units.len(), 1); - assert!(units[0] - .provider - .as_deref() - .unwrap() - .contains("eclipse.persistence")); - assert_eq!( - units[0].jta_data_source.as_deref(), - Some("jdbc/CoolstoreDS") - ); - assert!(!units[0].schema_generation.is_empty()); - } - - #[test] - fn parse_weblogic_jndi() { - let xml = r#" - - - orders - jms/orders - - - jdbc/CoolstoreDS - jdbc/CoolstoreDS - - "#; - let mut out = ResourcesResult::default(); - parse_server_bindings(xml, "WEB-INF/weblogic-ejb-jar.xml", &mut out); - assert!(!out.jms.is_empty() || !out.datasources.is_empty() || !out.ejb_bindings.is_empty()); - assert!( - out.datasources.iter().any(|d| d.name.contains("CoolstoreDS")) - || out.datasources.iter().any(|d| d - .jndi - .as_deref() - .unwrap_or("") - .contains("CoolstoreDS")) - ); - } -} diff --git a/src/cli/rules.rs b/src/cli/rules.rs index 7ab62ef6..9e2f31e0 100644 --- a/src/cli/rules.rs +++ b/src/cli/rules.rs @@ -22,7 +22,6 @@ pub fn run_rules( ctx: &CliContext, rules_dir: PathBuf, target: Option, - index_only: bool, catalog: Option, ) -> Result<()> { if !rules_dir.is_dir() && catalog.is_none() { @@ -44,21 +43,6 @@ pub fn run_rules( let mut profile = DiscoverStageReport::default(); run_kantra_index(store, rules_opt, catalog_opt, &mut profile)?; - if index_only { - if ctx.format == OutputFormat::Json { - let v = serde_json::json!({ - "schema_version": 1, - "command": "rules", - "action": "index-only", - "kantra_index_secs": profile.kantra_index.secs, - }); - ctx.emit_json_value(&v)?; - } else { - ctx.stdout_line("rules: indexed KantraRule nodes (eval skipped)")?; - } - return Ok(()); - } - let catalog = resolve_kantra_catalog(rules_opt, catalog_opt)?; let (engine, _) = KantraEngine::from_catalog(catalog.clone(), target.as_deref()) .map_err(|e| anyhow::anyhow!("kantra catalog: {e}"))?; diff --git a/src/cli/structured_query.rs b/src/cli/structured_query.rs index 2a561de8..3fe3f794 100644 --- a/src/cli/structured_query.rs +++ b/src/cli/structured_query.rs @@ -44,6 +44,7 @@ impl SharedQueryArgs { exact: false, annotation_names: None, show_attributes: false, + annotation_arg_index: None, }) } } @@ -53,6 +54,30 @@ fn open_store(ctx: &CliContext) -> Result Option> { + let path = rgctl_graph::paths::artifact_path(repo, "annotation_args.json"); + let bytes = std::fs::read(&path).ok()?; + let v: serde_json::Value = serde_json::from_slice(&bytes).ok()?; + let entries = v.get("entries")?.as_array()?; + let mut map = std::collections::HashMap::new(); + for e in entries { + let source_id = e.get("source_id")?.as_str()?; + let annotation = e.get("annotation")?.as_str()?; + let arguments = e.get("arguments")?.as_str()?; + let simple = rgctl_graph::parse_annotation_list(annotation) + .into_iter() + .next() + .unwrap_or_else(|| annotation.to_string()); + map.insert(format!("{source_id}\0{simple}"), arguments.to_string()); + } + if map.is_empty() { + None + } else { + Some(map) + } +} + fn emit_json(ctx: &CliContext, value: &T) -> Result<()> { let v = serde_json::to_value(value)?; ctx.emit_json_value(&v)?; @@ -131,6 +156,9 @@ pub fn run_find( } filters.annotation_names = Some(list); } + if show_attributes { + filters.annotation_arg_index = load_annotation_arg_index(&ctx.repo); + } // Default limit for find when not counting if filters.limit.is_none() && !count_only { filters.limit = Some(50); From 92655a022b0004d95fc803aa27ea957ba575eea3 Mon Sep 17 00:00:00 2001 From: Shaaf Syed <474256+sshaaf@users.noreply.github.com> Date: Fri, 2 Oct 2026 15:38:40 +0200 Subject: [PATCH 11/18] restructure skills, will break older ones Signed-off-by: Shaaf Syed <474256+sshaaf@users.noreply.github.com> --- AGENTS.md | 4 +- README.md | 6 +- agent-pack/manifest.yaml | 2 - .../skills/rgctl-discover/SKILL.md | 6 +- .../host-agents/skills/rgctl-flow/SKILL.md | 56 ++- .../host-agents/skills/rgctl-gate/SKILL.md | 13 +- .../host-agents/skills/rgctl-gql/SKILL.md | 35 -- .../host-agents/skills/rgctl-impact/SKILL.md | 18 +- .../host-agents/skills/rgctl-kantra/SKILL.md | 71 +++- .../host-agents/skills/rgctl-migrate/SKILL.md | 74 +++- .../host-agents/skills/rgctl-search/SKILL.md | 35 +- .../agents/host-agents/skills/rgctl/README.md | 76 ++-- .../agents/host-agents/skills/rgctl/SKILL.md | 81 ++-- .../rgctl/references/command-encyclopedia.md | 83 ++-- .../references/communities-and-policy.md | 47 ++- .../skills/rgctl/references/gql-reference.md | 181 -------- .../skills/rgctl/references/workflows.md | 397 +++++++++--------- .../host-claude/commands/rgctl/discover.md | 15 - .../agents/host-claude/commands/rgctl/flow.md | 15 - .../agents/host-claude/commands/rgctl/gate.md | 15 - .../agents/host-claude/commands/rgctl/gql.md | 15 - .../host-claude/commands/rgctl/impact.md | 15 - .../host-claude/commands/rgctl/kantra.md | 15 - .../host-claude/commands/rgctl/migrate.md | 15 - .../host-claude/commands/rgctl/search.md | 15 - .../skills/rgctl-discover/SKILL.md | 6 +- .../host-claude/skills/rgctl-flow/SKILL.md | 56 ++- .../host-claude/skills/rgctl-gate/SKILL.md | 13 +- .../host-claude/skills/rgctl-gql/SKILL.md | 35 -- .../host-claude/skills/rgctl-impact/SKILL.md | 18 +- .../host-claude/skills/rgctl-kantra/SKILL.md | 71 +++- .../host-claude/skills/rgctl-migrate/SKILL.md | 74 +++- .../host-claude/skills/rgctl-search/SKILL.md | 35 +- .../agents/host-claude/skills/rgctl/README.md | 76 ++-- .../agents/host-claude/skills/rgctl/SKILL.md | 81 ++-- .../rgctl/references/command-encyclopedia.md | 83 ++-- .../references/communities-and-policy.md | 47 ++- .../skills/rgctl/references/gql-reference.md | 181 -------- .../skills/rgctl/references/workflows.md | 397 +++++++++--------- .../host-codex/skills/rgctl-discover/SKILL.md | 6 +- .../host-codex/skills/rgctl-flow/SKILL.md | 56 ++- .../host-codex/skills/rgctl-gate/SKILL.md | 13 +- .../host-codex/skills/rgctl-gql/SKILL.md | 35 -- .../host-codex/skills/rgctl-impact/SKILL.md | 18 +- .../host-codex/skills/rgctl-kantra/SKILL.md | 71 +++- .../host-codex/skills/rgctl-migrate/SKILL.md | 74 +++- .../host-codex/skills/rgctl-search/SKILL.md | 35 +- .../agents/host-codex/skills/rgctl/README.md | 76 ++-- .../agents/host-codex/skills/rgctl/SKILL.md | 81 ++-- .../rgctl/references/command-encyclopedia.md | 83 ++-- .../references/communities-and-policy.md | 47 ++- .../skills/rgctl/references/gql-reference.md | 181 -------- .../skills/rgctl/references/workflows.md | 397 +++++++++--------- .../host-cursor/commands/rgctl-discover.md | 15 - .../agents/host-cursor/commands/rgctl-flow.md | 15 - .../agents/host-cursor/commands/rgctl-gate.md | 15 - .../agents/host-cursor/commands/rgctl-gql.md | 15 - .../host-cursor/commands/rgctl-impact.md | 15 - .../host-cursor/commands/rgctl-kantra.md | 15 - .../host-cursor/commands/rgctl-migrate.md | 15 - .../host-cursor/commands/rgctl-search.md | 15 - .../skills/rgctl-discover/SKILL.md | 6 +- .../host-cursor/skills/rgctl-flow/SKILL.md | 56 ++- .../host-cursor/skills/rgctl-gate/SKILL.md | 13 +- .../host-cursor/skills/rgctl-gql/SKILL.md | 35 -- .../host-cursor/skills/rgctl-impact/SKILL.md | 18 +- .../host-cursor/skills/rgctl-kantra/SKILL.md | 71 +++- .../host-cursor/skills/rgctl-migrate/SKILL.md | 74 +++- .../host-cursor/skills/rgctl-search/SKILL.md | 35 +- .../agents/host-cursor/skills/rgctl/README.md | 76 ++-- .../agents/host-cursor/skills/rgctl/SKILL.md | 81 ++-- .../rgctl/references/command-encyclopedia.md | 83 ++-- .../references/communities-and-policy.md | 47 ++- .../skills/rgctl/references/gql-reference.md | 181 -------- .../skills/rgctl/references/workflows.md | 397 +++++++++--------- agent-pack/out/manifest.json | 298 ++++++++++++- agent-pack/out/policy/rgctl-structural.mdc | 2 +- build.rs | 2 +- crates/rgctl-agent-pack-codegen/src/lib.rs | 106 +---- docs/Introduction.md | 8 +- docs/README.md | 2 +- docs/agents/USER_AGENTS_TEMPLATE.md | 9 +- docs/guides/README.md | 4 +- docs/guides/agent-commands.md | 96 ++--- docs/guides/agent-skill.md | 66 ++- docs/installation.md | 9 +- docs/json-api.md | 9 +- docs/releases/agent-pack-install.md | 18 +- docs/user-guide.md | 9 +- skills/rgctl/README.md | 62 +-- skills/rgctl/SKILL.md | 21 +- .../rgctl/references/command-encyclopedia.md | 65 +-- .../references/communities-and-policy.md | 47 ++- skills/rgctl/references/gql-reference.md | 181 -------- skills/rgctl/references/workflows.md | 92 +--- skills/rgctl/workflows/_advanced.md | 6 +- skills/rgctl/workflows/gql.md | 53 --- skills/rgctl/workflows/kantra.md | 17 +- skills/rgctl/workflows/search.md | 8 +- src/cli/agent_pack.rs | 69 --- src/cli/install.rs | 16 +- src/cli/install_output.rs | 10 +- src/cli/mod.rs | 8 +- tests/cli_output/install.rs | 13 +- tests/install_skill.rs | 34 +- website/src/app/agents/page.tsx | 49 +-- website/src/app/demo/page.tsx | 2 +- website/src/app/docs/page.tsx | 4 +- website/src/app/install/page.tsx | 15 +- website/src/app/page.tsx | 8 +- 110 files changed, 2967 insertions(+), 3395 deletions(-) delete mode 100644 agent-pack/out/agents/host-agents/skills/rgctl-gql/SKILL.md delete mode 100644 agent-pack/out/agents/host-agents/skills/rgctl/references/gql-reference.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/discover.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/flow.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/gate.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/gql.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/impact.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/kantra.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/migrate.md delete mode 100644 agent-pack/out/agents/host-claude/commands/rgctl/search.md delete mode 100644 agent-pack/out/agents/host-claude/skills/rgctl-gql/SKILL.md delete mode 100644 agent-pack/out/agents/host-claude/skills/rgctl/references/gql-reference.md delete mode 100644 agent-pack/out/agents/host-codex/skills/rgctl-gql/SKILL.md delete mode 100644 agent-pack/out/agents/host-codex/skills/rgctl/references/gql-reference.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-discover.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-flow.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-gate.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-gql.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-impact.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-kantra.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-migrate.md delete mode 100644 agent-pack/out/agents/host-cursor/commands/rgctl-search.md delete mode 100644 agent-pack/out/agents/host-cursor/skills/rgctl-gql/SKILL.md delete mode 100644 agent-pack/out/agents/host-cursor/skills/rgctl/references/gql-reference.md delete mode 100644 skills/rgctl/references/gql-reference.md delete mode 100644 skills/rgctl/workflows/gql.md diff --git a/AGENTS.md b/AGENTS.md index 013ec7d1..3b8298e0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -6,7 +6,7 @@ **Your goal when contributing here:** preserve ingest scale, query correctness, memory discipline, and deterministic artifacts under `.rgctl/` — not add convenience at the cost of Tokio blocking, whole-repo clones, or ungated cold regressions. -> **Looking for how to *use* rgctl on another codebase?** Install skills (`rgctl install --skill --with-commands`) or copy [docs/agents/USER_AGENTS_TEMPLATE.md](docs/agents/USER_AGENTS_TEMPLATE.md) into *that* repo’s `AGENTS.md`. See [docs/guides/agent-commands.md](docs/guides/agent-commands.md). +> **Looking for how to *use* rgctl on another codebase?** Install skills (`rgctl install --skill`) or copy [docs/agents/USER_AGENTS_TEMPLATE.md](docs/agents/USER_AGENTS_TEMPLATE.md) into *that* repo’s `AGENTS.md`. See [docs/guides/agent-commands.md](docs/guides/agent-commands.md). --- @@ -197,7 +197,7 @@ cargo build --release Code-daemon / ONNX weights: `git lfs pull` when using that embedder feature. -Dogfood fixtures: `rgctl-tests/` (e.g. ecommerce-*). Consumer agent pack: `rgctl install --skill --with-commands --tools cursor`. +Dogfood fixtures: `rgctl-tests/` (e.g. ecommerce-*). Consumer agent pack: `rgctl install --skill --tools cursor`. --- diff --git a/README.md b/README.md index 85f75b24..3fdf0d0d 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rgctl -f json blast-radius MyService rgctl -f json gql 'MATCH (a:Function)-[:CALLS]->(b) RETURN a,b LIMIT 20' # Use with your favorite LLM agent -rgctl install --skill --with-commands --tools cursor,claude,codex,agents +rgctl install --skill --tools cursor,claude,codex,agents ``` https://github.com/user-attachments/assets/15ec6d91-f716-4cbd-a873-e982ba3c6dca @@ -97,10 +97,10 @@ Always prefer **`-f json`** for agents and scripts ([JSON API](docs/json-api.md) ## Use with coding agents -Install the bundled pack (skills + slash commands) into your IDE tooling: +Install the bundled pack (skills) into your IDE tooling: ```bash -rgctl install --skill --with-commands --tools cursor,claude,codex,agents +rgctl install --skill --tools cursor,claude,codex,agents ``` Then: **discover once → query with `-f json`**. See [Agent commands](docs/guides/agent-commands.md). diff --git a/agent-pack/manifest.yaml b/agent-pack/manifest.yaml index c3b34271..2a7804fc 100644 --- a/agent-pack/manifest.yaml +++ b/agent-pack/manifest.yaml @@ -9,8 +9,6 @@ workflows: title: Data flow and slices - id: search title: Semantic and structural search - - id: gql - title: Graph query language - id: migrate title: Migration roadmap - id: kantra diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-discover/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-discover/SKILL.md index 62094801..7cc1b194 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-discover/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-discover/SKILL.md @@ -3,7 +3,7 @@ name: rgctl-discover description: "Index and discover. Use for rgctl discover workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Discover workflow @@ -18,7 +18,9 @@ metadata: **Fast path:** If `.rgctl/` exists and the user did not ask to rebuild, do **not** re-run discover. -Common flags: `--with-cfg`, `--with-kantra`, `--export-migration-hints` (migration plan is the **migrate** workflow, not discover alone). +Common flags: `--with-cfg` (CFG/PDG archive), `--with-ast-skeleton`, `--with-dfg-loops` (loop-carried PDG tags). Migration plan output is the **migrate** workflow; Konveyor rules are the **kantra** workflow — do not conflate them with a plain index. + +Artifacts live at `{repo}/.rgctl/`. Check CFG readiness with `rgctl -f json cpg status` before slice/PDG workflows. ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-flow/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-flow/SKILL.md index d620f4bd..0c2ae23e 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-flow/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-flow/SKILL.md @@ -3,19 +3,67 @@ name: rgctl-flow description: "Data flow and slices. Use for rgctl flow workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Flow workflow **When:** Slices, PDG, taint, CPG data flows. Requires `discover --with-cfg`. +Check readiness: `rgctl -f json cpg status`. + +### AST skeleton + +**User intent:** *"Inspect the AST skeleton of `updateQuantity` to check its structure"* + ```bash -rgctl -r "$REPO" -f json slice FILE --line N --variable V [--function F] [--direction backward|forward] -rgctl -r "$REPO" -f json cpg flows FILE --line N --variable V --function F +rgctl discover . --with-ast-skeleton +rgctl -f json cpg ast updateQuantity ``` -Check readiness: `rgctl -f json cpg status`. +Coarse skeleton (`kind`, lines, `label`) — **not** a typed signature API (`params` / `return_type` are not emitted). + +### Status + line slice + +**User intent:** *"Confirm the CFG archive is ready, then slice how `quantity` is used in `updateQuantity`"* + +```bash +rgctl -f json cpg status +rgctl -f json cpg slice src/cart/CartService.ts \ + --line 50 --variable quantity --function updateQuantity --view pdg +``` + +**`cpg slice` has no `--symbol`.** For whole-function CFG/PDG, use `inspect cfg|pdg` or `cpg pdg `. + +CLI alias: `rgctl -f json slice FILE --line N --variable V [--function F] [--direction backward|forward]`. + +### Field mutations + +**User intent:** *"Check where `ShoppingCart` object fields are mutated"* + +```bash +rgctl -f json cpg mutations --type ShoppingCart --exclude-ctors +``` + +### Data flows + +**User intent:** *"Trace how the `quantity` variable flows into database queries"* + +```bash +rgctl -f json cpg flows src/cart/CartService.ts \ + --line 50 --variable quantity --function updateQuantity --direction forward +``` + +### Loop-carried DFG + +**User intent:** *"Check for loop-carried dependencies that prevent parallelization"* + +```bash +rgctl discover . --with-cfg --with-dfg-loops +rgctl -f json inspect BatchProcessor.process pdg --edge-layer data +``` + +`--with-dfg-loops` **tags** edges during discover — it does not print a dedicated loop-hazard array. Look for `loop_carried` on PDG data deps. ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-gate/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-gate/SKILL.md index 07ab7e10..59eb2f5f 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-gate/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-gate/SKILL.md @@ -3,15 +3,26 @@ name: rgctl-gate description: "CI and policy gates. Use for rgctl gate workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Gate workflow **When:** Policy checks and temporal PR gates. +### Policy check + +**User intent:** *"Validate changes against project policies before committing"* + ```bash rgctl -r "$REPO" -f json check --policy-file policy.json +``` + +Blast-radius policy schema (`max_impact_nodes`, `forbidden_crossings`, …) — see [docs/policy-format.md](../../docs/policy-format.md). Named rules like `no-controller-direct-db-access` are **not** built-in ids. Report `passed` + `violations`. + +### Temporal PR gate + +```bash rgctl -r "$REPO" -f json pr-check --policy-file rgctl-pr-policy.json --base-ref origin/main --head-ref HEAD --strict rgctl -r "$REPO" -f json check --temporal --policy-file policy.json --base-ref origin/main --head-ref HEAD ``` diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-gql/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-gql/SKILL.md deleted file mode 100644 index da90d1d6..00000000 --- a/agent-pack/out/agents/host-agents/skills/rgctl-gql/SKILL.md +++ /dev/null @@ -1,35 +0,0 @@ ---- -name: rgctl-gql -description: "Graph query language. Use for rgctl gql workflow. Spawn rgctl -f json; parse schema_version from stdout." -rgctl-managed: true -metadata: - generatedBy: "rgctl 0.4.13" ---- - -# GQL workflow - -**When:** Ad-hoc graph queries, inventories, call neighborhoods. - -```bash -rgctl -r "$REPO" -f json gql 'MATCH (n:Function) WHERE n.name LIKE "*Service*" RETURN n LIMIT 20' -rgctl -r "$REPO" -f json gql --macro-name all_functions unused -``` - -Use **qualified_name** / FQN for classes, not bare `n.name` when disambiguating. Always use **LIMIT** on broad patterns. Explain macros before inventing raw GQL. - - -## Agent loop - -1. Parse the user question (natural language). -2. Run `rgctl -f json …` (or `rgctl serve` + HTTP for repeated queries). -3. Parse `schema_version` and payload from **stdout** only. -4. Summarize facts; do not dump raw JSON. -5. Re-query if the graph may be stale after edits. - -**Never** redirect stderr to `/dev/null`. If `.rgctl/` exists and the question is structural, use rgctl before ripgrep or bulk file reads. - -```bash -export REPO=/path/to/repo -rgctl -r "$REPO" -f json -``` - diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-impact/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-impact/SKILL.md index 3fa5b702..46287a0b 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-impact/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-impact/SKILL.md @@ -3,18 +3,30 @@ name: rgctl-impact description: "Blast radius and impact. Use for rgctl impact workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Impact workflow **When:** Before refactors, renames, or API changes. +### Blast radius + +**User intent:** *"What's the impact if I change the signature of `updateQuantity`?"* + ```bash -rgctl -r "$REPO" -f json blast-radius SYMBOL [--depth N] [--class NAME] [--file PATH] +rgctl -r "$REPO" -f json blast-radius updateQuantity --depth 2 ``` -Disambiguate symbols with `--class` or `--file` when names collide. Report hop depth and top callers/callees from JSON payload. +Report `metrics.score`, `topology.direct_callers`, impact size. Add `--class` / `--file` if ambiguous. + +### Relationship between two symbols + +**User intent:** *"What's the relationship between A and B?"* + +1. Resolve symbols → bounded CALLS/DEPENDSON traversal +2. Report hops, shared neighbors, files +3. If no direct path but asymmetric dependency, fall back to `blast-radius` on each ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-kantra/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-kantra/SKILL.md index d200c8c1..c5bbe010 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-kantra/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-kantra/SKILL.md @@ -3,7 +3,7 @@ name: rgctl-kantra description: "Konveyor Kantra rules. Use for rgctl kantra workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Kantra workflow @@ -12,15 +12,74 @@ metadata: **This workflow is not migration roadmap export.** Do not present `migration_plan.json` as the main Kantra deliverable. +Native evaluation of [Konveyor Kantra](https://github.com/konveyor/kantra) rules against the rgctl graph and source cache. Release builds embed Konveyor `stable/java` (~2.6k rules); no external Kantra CLI required. + +### Default Kantra discover + +**User intent:** *"Run Konveyor migration rules on this Java codebase"* + +```bash +rgctl discover . -l java --with-kantra +# violations: .rgctl/kantra_findings.json + +# Post-index (when snapshot already exists): +rgctl -f json rules run ./rules/ --target quarkus +# rules in graph: find --type kantrarule / inventory --by type +``` + +Report `catalog_id`, `evaluated_rules`, violation count, sample hits (`rule_id`, `file`, `line`, `matched_by`), and top `skipped_rules` reasons. + +### Target-filtered eval + +**User intent:** *"What Quarkus migration rules apply?" / "Audit for Spring Boot 3+"* + +```bash +rgctl discover . -l java --with-kantra --kantra-target quarkus +# or: --kantra-target spring-boot3+ +``` + +`target_filter` appears in `kantra_findings.json`. Only rules with `konveyor.io/target=` labels are evaluated. + +### Rules inventory + +**User intent:** *"List migration rules indexed in the graph" / "How many Kantra rules?"* + +```bash +rgctl -f json find --type kantrarule --limit 50 +rgctl -f json inventory --by type # KantraRule / KantraRuleset counts +# after full eval: rule → code links +rgctl -f json relations --edge violates --from-type kantrarule --limit 50 +``` + +Line-level detail and enrichment live in `kantra_findings.json` (preferred over edge dumps for violations). + +### Fixture / CI override + +**User intent:** *"Run a small custom ruleset in CI"* + ```bash -rgctl -r "$REPO" discover . -l java --with-kantra -rgctl -r "$REPO" discover . -l java --with-kantra --kantra-target quarkus -rgctl -r "$REPO" -f json gql 'MATCH (r:KantraRule) RETURN r LIMIT 20' +rgctl discover . --with-kantra --kantra-rules tests/fixtures/kantra-rules ``` -Overrides: `--kantra-rules DIR`, `--kantra-catalog ROOT`, `--kantra-index-only` (index without eval). +Mutually exclusive with `--kantra-catalog`. Embedded catalog is the default when neither override is set. + +### Index only + +**User intent:** *"Index rules into the graph without running eval"* + +```bash +rgctl discover . --with-kantra --kantra-index-only +``` + +Useful when you only need structured rule inventory (`find --type kantrarule`). Eval stage is skipped; `kantra_findings.json` is not written. + +**Pitfalls:** + +- Does **not** require `--with-cfg` +- Many upstream Konveyor rules use unsupported providers (`builtin.xml`, `java.dependency`) or Windup-style regex — expect a large `skipped_rules` list with full catalog +- Re-run discover after rule/catalog changes; kantra index rewrites `graph.snapshot.bin` at end of pipeline -For extraction ordering after violations, use the **migrate** workflow separately. +**See:** [User guide — Kantra](../../docs/user-guide.md#kantra-migration-rules---with-kantra), [JSON API](../../docs/json-api.md#kantra_findingsjson), [KANTRA_ARCHITECTURE_OPTIONS.md](../../KANTRA_ARCHITECTURE_OPTIONS.md) ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-migrate/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-migrate/SKILL.md index 8395c20b..a22cdd75 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-migrate/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-migrate/SKILL.md @@ -3,7 +3,7 @@ name: rgctl-migrate description: "Migration roadmap. Use for rgctl migrate workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Migrate workflow @@ -12,16 +12,78 @@ metadata: **This workflow is not Kantra.** Do not treat `--with-kantra` or `kantra_findings.json` as the main deliverable here. +### Migration plan + +**User intent:** *"Generate a complete migration plan for this codebase"* + ```bash -rgctl -r "$REPO" discover . --with-cfg --with-harmonic --export-migration-hints \ +rgctl discover . --with-cfg --with-security --with-taint \ + --with-dashboard --with-harmonic --export-migration-hints \ --migration-preset hybrid_default --migration-order scheduled -rgctl -r "$REPO" -f json metrics --pagerank +# read .rgctl/migration_plan.json (and/or dashboard Migration tab via serve --open) +``` + +Discover stdout (`-f json`) is **telemetry** — not the plan body. Report path + preset/order used + top `packages[]` by priority/step. + +**Migration presets:** + +- `hybrid_default` - Balanced approach (default) +- `foundational_first` - Migrate core/base libraries first +- `dense_cluster` - Tackle tightly-coupled modules together +- `risk_mitigation` - Minimize blast radius per step + +**Migration orders:** + +- `scheduled` - Dependency-aware sequence (default) +- `priority` - Highest-impact packages first + +### Hotspots + +**User intent:** *"Which core functions are bottlenecks / central dependencies?"* + +```bash +rgctl -f json metrics --pagerank +``` + +Report `.pagerank.top` nodes + why they are risky to change. Resolve UUIDs to function names using `cpg function`. + +### CPG export + +**User intent:** *"Export a GraphSON archive to preserve the baseline before refactoring"* + +```bash +rgctl cpg export --format graphson --output cpg.json --path-contains src/ +``` + +Writes a **file**; success is typically a text summary. Needs prior `discover --with-cfg` for a useful L_proc-rich export. + +### Migration feature-flag cheat sheet + +| Flag | Enables | +|------|---------| +| `--with-cfg` | CFG/PDG/dominance archive (slice, inspect, cpg PDG) | +| `--with-taint` | Discover-time taint (implies CFG as needed) | +| `--with-security` | Secret scanning | +| `--with-dashboard` | `.rgctl/dashboard/` bundle | +| `--with-harmonic` | Harmonic centrality (migration ranking; expensive) | +| `--export-migration-hints` | Write `migration_plan.json` | +| `--with-ast-skeleton` | AST skeleton for `cpg ast` | +| `--with-dfg-loops` | Tag loop-carried data deps on PDG | +| `--migration-preset ` | Strategy: `hybrid_default`, `foundational_first`, `dense_cluster`, `risk_mitigation` | +| `--migration-order ` | Roadmap sort: `scheduled` (dependency-aware), `priority` (score rank) | + +Migration-oriented discover (heavy): + +```bash +rgctl discover . --with-cfg --with-security --with-taint \ + --with-dashboard --with-harmonic --export-migration-hints \ + --migration-preset foundational_first --migration-order scheduled +# then read .rgctl/migration_plan.json (or dashboard copy) ``` -Presets: `hybrid_default`, `foundational_first`, `dense_cluster`, `risk_mitigation`. -Orders: `scheduled` (dependency-aware), `priority` (score rank). +Choose `--migration-preset` to match user intent. Use `--migration-order priority` when the user wants highest-impact packages first instead of a dependency-safe sequence. -Report plan path, preset/order, and top packages — not raw discover telemetry. +For extraction ordering after violations, run the **kantra** workflow separately when Konveyor rules apply. ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl-search/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl-search/SKILL.md index 06385e82..25d706b1 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl-search/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl-search/SKILL.md @@ -3,7 +3,7 @@ name: rgctl-search description: "Semantic and structural search. Use for rgctl search workflow. Spawn rgctl -f json; parse schema_version from stdout." rgctl-managed: true metadata: - generatedBy: "rgctl 0.4.13" + generatedBy: "rgctl 0.0.0-dev" --- # Search workflow @@ -11,12 +11,37 @@ metadata: **When:** Natural-language or intent-based code location (requires `semantic index`). ```bash -rgctl -r "$REPO" -f json semantic query "…" [--limit 10] -rgctl -r "$REPO" -f json semantic query "…" --scope community --limit 10 -rgctl -r "$REPO" -f json gql --macro-name all_communities unused +rgctl -r "$REPO" semantic index # opt-in; default vocab. extras: --embedder code-daemon|hash +rgctl -r "$REPO" -f json semantic query "checkout flow" --limit 10 ``` -Fusion is on by default for semantic query; use GQL for exact graph patterns. +Fusion is on by default for semantic query. For exact graph patterns use structured verbs (`find`, `callers`, `relations`, `inventory`) — not freeform Cypher. + +### NL function search + +**User intent:** *"Where is the code that handles our checkout flow?"* + +Report top `hits[]` (`name`, `score`, `file_path`). + +### Community semantic + +**User intent:** *"Which architectural subsystem owns checkout?"* + +```bash +rgctl -r "$REPO" -f json semantic query "checkout" --scope community --limit 10 +``` + +Hits are pooled **community** results (same `hits[]` contract). + +### Concept search with empty hits + +If `find` / name globs return 0 for a concept (e.g., "ingress", "gateway"): + +1. Try `communities list` and grep labels +2. Try `semantic query ""` +3. Broaden with `find '*Gateway*' --type class` or `inventory --by type` + +Concepts often live in package/directory paths or type names, not bare function names. ## Agent loop diff --git a/agent-pack/out/agents/host-agents/skills/rgctl/README.md b/agent-pack/out/agents/host-agents/skills/rgctl/README.md index 3d832943..ede0ff9f 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl/README.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl/README.md @@ -4,23 +4,22 @@ A skill for answering structural questions about codebases using the rgctl CLI g ## Quick Stats -- **Main skill:** 352 lines (57% reduction from original 814 lines) -- **Total documentation:** 1,505 lines (85% more comprehensive coverage) -- **Files:** 5 reference files + main skill -- **Workflow families:** 6 + Kantra rules -- **NL routing examples:** 25+ common user utterances +- **Main skill:** router + structured verb tables +- **Reference files:** command encyclopedia, workflows (assembled), communities & policy +- **Workflow families:** discover, impact, flow, search, migrate, kantra, gate +- **Agent query path:** `find` / `callers` / `callees` / `relations` / `inventory` / `status` (no Cypher) ## Structure ``` skills/rgctl/ -├── SKILL.md # Main skill (352 lines) +├── SKILL.md # Main skill ├── README.md # This file +├── workflows/ # Source for workflow skills + references/workflows.md (assembled at build) └── references/ - ├── command-encyclopedia.md # All commands with JSON samples (19KB) - ├── workflows.md # Migration, Kantra rules, refactor, audit scenarios - ├── gql-reference.md # GQL patterns & limitations (4.7KB) - └── communities-and-policy.md # Community detection + CI policy (13KB) + ├── command-encyclopedia.md # All commands with JSON samples + ├── workflows.md # Generated from workflows/ at build (do not edit by hand) + └── communities-and-policy.md # Community detection + CI policy ``` ## What's Covered @@ -29,76 +28,57 @@ skills/rgctl/ - When to use rgctl - **CLI subprocess workflow** — spawn `rgctl -f json` for agents -- **6 workflow families:** +- **Workflow families:** 1. Discovery & Indexing 1b. Konveyor Kantra rules (`--with-kantra`) - 2. Query & Search (includes communities + KantraRule GQL) + 2. Query & Search (structured verbs + communities) 3. Impact & Safety (includes policy checks) 4. Metrics & Analysis 5. Code Analysis (CFG/PDG/slicing) 6. Export & Visualization -- **NL routing table** (20+ user utterances → commands) -- Common scenarios (migration, pre-refactor safety) +- **NL routing table** (user utterances → commands) - Failure playbook ### References (Loaded On-Demand) #### command-encyclopedia.md -- All 15+ commands with full details +- Structured query verbs and domain commands - JSON sample responses - Prerequisites and pitfalls - "What to report" guidelines -#### workflows.md -- Migration & audit workflows (incl. Konveyor Kantra `--with-kantra`) -- Intent discovery & subsystem mapping -- Pre-refactor safety analysis -- CI gates & policy -- Advanced patterns - -#### gql-reference.md -- Cypher subset capabilities -- Macros (all_functions, all_communities) -- Valid edge types -- LIKE pattern matching limitations -- Common patterns & troubleshooting +#### workflows.md (generated) +- Assembled from `workflows/*.md` when the agent pack is built (`cargo build`) +- Edit fragments under `workflows/` (e.g. `migrate.md`, `kantra.md`); order comes from `agent-pack/manifest.yaml` #### communities-and-policy.md -- **Community Detection:** - - What communities are (implicit architecture) - - Commands (list, query, label, semantic scope) - - Use cases (microservice extraction, ownership) - - 5 complete workflows -- **CI Policy Checks:** - - Policy schema (max_impact_nodes, centrality, forbidden_crossings) - - CI integration (GitHub Actions, GitLab) - - Crafting policies (calibration, gradual tightening) - - 4 complete workflows -- Combined workflows using both features +- **Community Detection:** list, semantic scope, ownership workflows +- **CI Policy Checks:** schema, CI integration, calibration ## Design Principles -✅ **Progressive disclosure** - Main skill <500 lines, details in references +✅ **Progressive disclosure** - Main skill lean, details in references ✅ **Workflow-centric** - Organized by user intent, not commands -✅ **CLI-first** - Agents use `rgctl -f json` subprocesses (optional `serve` HTTP) -✅ **Clear routing** - Natural language → tool mapping -✅ **Comprehensive** - All features documented with examples -✅ **Integration** - Shows how features work together +✅ **CLI-first** - Agents use `rgctl -f json` structured verbs +✅ **Clear routing** - Natural language → command mapping +✅ **No Cypher in skill surface** - Agents must not invent MATCH strings ## Installation -From another repo: +From a target repository (not the rgctl source tree unless you are dogfooding): ```bash -rgctl install --skill +rgctl install --skill --tools cursor,claude,codex,antigravity,agents ``` -This writes `.claude/skills/rgctl/`, `.agents/skills/rgctl/`, and `.cursor/skills/rgctl/` from the embedded skill. +Installs meta skill `rgctl`, workflow skills (`rgctl-discover`, …). See [Agent commands guide](../../docs/guides/agent-commands.md). + +**Maintainers:** edit workflow bodies under `workflows/`; regenerate `references/workflows.md` with `assemble_workflows_reference` (see `rgctl-agent-pack-codegen` test `workflows_reference_matches_fragments`). ## See Also - [User Guide](../../docs/user-guide.md) - Complete CLI tutorial - [Agent recipes](../../docs/agent-recipes.md) - Copy-paste CLI workflows -- [HTTP API](../../docs/http-api.md) - Optional `rgctl serve` for repeated queries +- [HTTP API](../../docs/http-api.md) - Optional `rgctl serve` for dashboard - [JSON API](../../docs/json-api.md) - Schema specifications - [All Guides](../../docs/guides/README.md) - Feature-specific guides diff --git a/agent-pack/out/agents/host-agents/skills/rgctl/SKILL.md b/agent-pack/out/agents/host-agents/skills/rgctl/SKILL.md index f0a8b0b0..f92dd988 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl/SKILL.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl/SKILL.md @@ -37,15 +37,15 @@ rgctl -r "$REPO" -f json … **Critical:** Parse `schema_version` + payload from **stdout**. **Never use `2>/dev/null`** — it swallows rgctl errors. -For many queries in one session, optional: `rgctl serve` + `POST /api/query` (see [HTTP API](../../docs/http-api.md)). +For interactive exploration, optional: `rgctl serve --open` (dashboard). Agents should still spawn CLI structured verbs (`find` / `callers` / `relations` / `inventory` / …). -Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/` into the repo. +Legacy daemon cache under `~/.rgctl/cache/` is obsolete; run `rgctl discover .` in the repo to build `{repo}/.rgctl/`. ## Agent Loop ```text 1. USER PROMPT → natural language (not a CLI string) -2. SUBPROCESS → rgctl -f json (or HTTP /api/query) +2. SUBPROCESS → rgctl -f json 3. GRAPH FACTS → parse schema_version + payload 4. LLM REASONING → summarize using "what to report" guidelines 5. ACTION → edit / plan / check — re-query if graph may be stale @@ -90,30 +90,44 @@ Legacy daemon cache: `rgctl migrate-cache` copies `~/.rgctl/cache/{name}/.rgctl/ | User Intent | CLI Command | |-------------|-------------| -| Evaluate migration rules | `discover . --with-kantra` | -| Filter by migration target | `discover . --with-kantra --kantra-target quarkus` | -| CI / custom ruleset | `discover . --with-kantra --kantra-rules PATH` | -| List indexed rules (GQL) | `gql "MATCH (r:KantraRule) RETURN r LIMIT 20"` | -| Rules for one target label | `gql` with `` r.`konveyor.io/target` `` property (backticks) | +| Evaluate migration rules | `discover . --with-kantra` or `rules run ./rules/` | +| Filter by migration target | `discover . --with-kantra --kantra-target quarkus` / `rules run ./rules/ --target quarkus` | +| CI / custom ruleset | `discover . --with-kantra --kantra-rules PATH` / `rules run PATH` | +| Index rules only | `discover . --with-kantra --kantra-index-only` | +| List indexed rules | `find --type kantrarule --limit 50` / `inventory --by type` | +| Rule → code links | `relations --edge violates --from-type kantrarule` | | Read violations artifact | `.rgctl/kantra_findings.json` | **See:** [User guide — Kantra](../../docs/user-guide.md#kantra-migration-rules---with-kantra), [JSON — kantra_findings](../../docs/json-api.md#kantra_findingsjson) ### 2. Query & Search +**Prefer structured verbs** (mmap; no Cypher). Parse `-f json` from **stdout** (`schema_version`); never `2>/dev/null`. + | User Intent | CLI Command | |-------------|-------------| -| Inventory functions | `gql --macro-name all_functions unused` | -| Find callers/callees | `gql "MATCH (a)-[:CALLS]->(b) WHERE ..."` | +| Session / index freshness | `status` | +| Schema / counts (incl. zeros) | `inventory --by type` or `inventory --by edge` | +| Import prefix census | `inventory --by import-prefix` | +| Count functions | `find --type function --count-only` | +| Find by name/type | `find "User*" --type class --limit 50` | +| Suffix scan (MDB / Remote) | `find '*MDB*' --type class` | +| Classes with annotation | `find --annotation @MessageDriven --type class` | +| javax import worklist | `find "import javax*" --type import --scope ` | +| Annotation pairs (seedless) | `relations --edge annotatedwith --from-type function --to-type annotation --scope ` | +| Find callers/callees | `callers --depth 1` / `callees ` | +| Outside callers of a module | `callers --scope --scope-mode outside` | +| EXTENDS / IMPLEMENTS inventory | `relations --edge extends --from-type class` (omit SYMBOL) | | Natural-language search | `semantic query "checkout flow"` | | List communities | `communities list` | -| Community members | `gql "MATCH (f) WHERE f.community_id='12'"` | | Subsystem ownership | `semantic query "X" --scope community` | | Refresh community labels | `communities label --write` | +| Community census | `inventory --by community` | -**GQL limitations:** no `COUNT`/`ORDER BY`; LIKE prefix/suffix only; CALLS misses dynamic dispatch; Konveyor labels need backticks in `WHERE`. +**Migration probe order:** `status` → `inventory --by import-prefix` → `find --annotation …` / suffix globs → `rules run` / `--with-kantra` → `callers InitialContext`. -**See:** [GQL Reference](references/gql-reference.md), [Semantic Search Guide](../../docs/guides/semantic-search.md) +**Complexity honesty:** exact name = hash index; prefix/`*mid*`/`--scope` may scan keys/columns until better indexes land. Module re-index is still a strong speed lever. Annotation **arguments** (e.g. `@Path("/x")`) need `--show-attributes` when `annotation_args.json` is present. +**See:** [Command Encyclopedia](references/command-encyclopedia.md) (find/callers/relations/inventory/status), [Semantic Search Guide](../../docs/guides/semantic-search.md) ### 3. Impact & Safety @@ -164,19 +178,21 @@ Needs `discover --with-cfg`. `--function` is method name, not class. | "Where is checkout flow?" | `semantic query "checkout flow" --limit 10` | | "Impact if I change X" | `blast-radius X --depth 2` | | "Validate against policy" | `check --policy-file policy.json` | -| "Who calls X" | `gql "MATCH (a)-[:CALLS*1..3]->(b) WHERE a.name='X' RETURN a,b"` | +| "Who calls X" | `callers X --depth 2` (impact → `blast-radius X`) | +| "javax imports / annotations" | `find "import javax*" --type import`; `relations --edge annotatedwith --from-type function --to-type annotation` | | "Where is X mutated?" | `cpg mutations --type X --exclude-ctors` | ## Failure Playbook | Symptom | Fix | |---------|-----| -| No `.rgctl/` in repo | Run `cd repo && rgctl discover .`; or `rgctl migrate-cache` from legacy daemon cache | +| No `.rgctl/` in repo | Run `cd repo && rgctl discover .` | | slice/inspect/cpg fails | Re-discover with `--with-cfg` | | semantic query fails | `semantic index` | -| Ambiguous symbol | Add `--class` or `--file`; disambiguate via GQL | +| Ambiguous symbol | Add `--class` or `--file` on callers/find | | `check` exit 1 | Report violations (JSON still on stdout) | -| GQL LIKE returns 0 | Try `communities list`, `semantic query`, or broader type patterns | +| find/relations empty | Run `inventory --by type` / `--by edge` (zeros mean unpopulated schema); check `--scope` | +| Name glob returns 0 | Try `semantic query` / `communities list` / broader `find '*X*'` types | ## Artifacts @@ -205,7 +221,6 @@ rgctl -r "$REPO" -f json … - **[Command Encyclopedia](references/command-encyclopedia.md)** — Full command reference - **[Workflows](references/workflows.md)** — Worked scenarios -- **[GQL Reference](references/gql-reference.md)** — GQL patterns - **[Communities & Policy](references/communities-and-policy.md)** — CI policy checks ## External Documentation @@ -213,30 +228,30 @@ rgctl -r "$REPO" -f json … - [User Guide](../../docs/user-guide.md) — Complete CLI tutorial - [JSON API](../../docs/json-api.md) — Schema specifications - [Agent Recipes](../../docs/agent-recipes.md) — Copy-paste recipes -- [AGENTS.md](../../AGENTS.md) — Minimal agent contract +- [USER_AGENTS_TEMPLATE.md](../../docs/agents/USER_AGENTS_TEMPLATE.md) — paste into consumer repos +- [AGENTS.md](../../AGENTS.md) — contributor agent README (rgctl source tree) - [Policy Format](../../docs/policy-format.md) — CI policy schema ## Installation ```bash -rgctl install --skill +rgctl install --skill --tools cursor,claude,codex,antigravity,agents ``` -Writes `.claude/skills/rgctl/`, `.agents/skills/rgctl/`, and `.cursor/skills/rgctl/` from the embedded skill in the binary. +Installs workflow skills (`rgctl-discover`, `rgctl-migrate`, `rgctl-kantra`, …) and meta-skill `rgctl` for each selected adapter. Omitting `--tools` installs **cursor, claude, codex, agents, antigravity**; use `--tools all` for the full registry. See [docs/guides/agent-commands.md](../../docs/guides/agent-commands.md). Workflow source: `skills/rgctl/workflows/`; keep `references/workflows.md` in sync via `cargo test -p rgctl-agent-pack-codegen workflows_reference_matches_fragments`. -## Workflow slash commands (generated) +## Workflow skills (generated) -| Intent | Command | -|--------|---------| -| Index and discover | `/rgctl-discover` | -| Blast radius and impact | `/rgctl-impact` | -| Data flow and slices | `/rgctl-flow` | -| Semantic and structural search | `/rgctl-search` | -| Graph query language | `/rgctl-gql` | -| Migration roadmap | `/rgctl-migrate` | -| Konveyor Kantra rules | `/rgctl-kantra` | -| CI and policy gates | `/rgctl-gate` | +| Intent | Skill | +|--------|-------| +| Index and discover | `rgctl-discover` | +| Blast radius and impact | `rgctl-impact` | +| Data flow and slices | `rgctl-flow` | +| Semantic and structural search | `rgctl-search` | +| Migration roadmap | `rgctl-migrate` | +| Konveyor Kantra rules | `rgctl-kantra` | +| CI and policy gates | `rgctl-gate` | **Migrate** (roadmap / `migration_plan.json`) and **Kantra** (rules / `kantra_findings.json`) are separate workflows — do not conflate. -rgctl-managed router note generatedBy rgctl 0.4.13 +rgctl-managed router note generatedBy rgctl 0.0.0-dev diff --git a/agent-pack/out/agents/host-agents/skills/rgctl/references/command-encyclopedia.md b/agent-pack/out/agents/host-agents/skills/rgctl/references/command-encyclopedia.md index 5e48b95c..857fe2af 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl/references/command-encyclopedia.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl/references/command-encyclopedia.md @@ -7,7 +7,7 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json ## Table of Contents - [discover](#discover) -- [gql](#gql) +- [find / callers / callees / relations / inventory](#find--callers--callees--relations--inventory) - [blast-radius](#blast-radius) - [slice](#slice) - [inspect](#inspect) @@ -90,9 +90,9 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json } ``` -**GQL companion:** After discover, query indexed rules with `gql "MATCH (r:KantraRule) RETURN r"`. Konveyor labels are properties — filter with backticks: `` r.`konveyor.io/target` ``. +**Structured companion:** After discover, list indexed rules with `find --type kantrarule` / `inventory --by type`. Rule→code links after full eval: `relations --edge violates --from-type kantrarule`. Prefer `kantra_findings.json` for line-level violations. -**Pitfalls:** Full embedded catalog skips many rules (unsupported providers, Windup regex). Use `kantra_findings.json` for violation details; GQL `VIOLATES` edges link rules to code nodes after full eval (not `--kantra-index-only`). +**Pitfalls:** Full embedded catalog skips many rules (unsupported providers, Windup regex). Use `kantra_findings.json` for violation details; `VIOLATES` edges exist after full eval (not `--kantra-index-only`). **Agent should report:** `catalog_id`, `target_filter`, violation count, representative hits, skip summary — not full JSON dump. @@ -100,44 +100,49 @@ Samples below are truncated where noted. Field names match live CLI / `docs/json --- -## gql +## find / callers / callees / relations / inventory -**Command:** `rgctl -f json gql ''` or `rgctl -f json gql --macro-name unused` +**Commands:** -**Purpose:** Inventory, callers/callees, communities, path/relationship queries. +```bash +rgctl -f json find [PATTERN] --type function --scope pkg --limit 50 +rgctl -f json find --type function --count-only +rgctl -f json find --annotation @MessageDriven --type class +rgctl -f json find --annotation @Stateful,@Stateless,@Singleton --type class +rgctl -f json find '*MDB*' --type class --limit 50 # bare-name suffix scan +rgctl -f json callers --depth 1 --file PATH --class NAME --line N +rgctl -f json callees --depth 1 +rgctl -f json relations [SYMBOL] --edge annotatedwith --from-type function --to-type annotation --scope pkg +rgctl -f json relations --edge extends --from-type class # seedless +rgctl -f json inventory --by type # includes zero-count kinds +rgctl -f json inventory --by edge +rgctl -f json inventory --by import-prefix # javax.ejb / javax.jms / org.eclipse … +rgctl -f json status # snapshot presence, digest, node/edge counts +rgctl discover . --find '*coolstore*' # locate candidate project roots (no index) +rgctl -f json rules run ./rules/ [--target quarkus] # post-index Kantra eval +rgctl -f json find --annotation @Resource --show-attributes # needs annotation_args.json from discover +rgctl -f json query find … # alias namespace +``` -**Prerequisites:** `discover` done. Virtual `:Community` needs analysis overlay from discover. +**Purpose:** Deterministic mmap structured query (no Cypher, no `MemoryBackend` hydrate). **Agents must use these verbs** — do not invent MATCH strings. Relations `total` is distinct `(source,target,edge)`; duplicates collapse with `occurrences` (`schema_version` ≥ 2). `inventory --by edge` uses the same rule: `count` = distinct, `occurrences` = raw stored edges. -**Sample** (macro `all_functions`): +**Migration probes (Coolstore-shaped):** +1. `status` — is `.rgctl/` fresh? +2. `inventory --by import-prefix` — EE surface census +3. `find --annotation @MessageDriven|@SessionScoped|…` — blockers without package guess +4. `find '*MDB*'` / `'*Remote*'` — suffix scan before reading files +5. `rules run ./rules/` or `discover --with-kantra` — fire `when:` catalog (M2) +6. `callers InitialContext` — JNDI usage sites -```json -{ - "schema_version": 1, - "count": 260, - "rows": [ - [{ "binding": "f", "node": "addItem", "type": "Function", - "file": "…/controller/CartController.java" }] - ], - "explain": false -} -``` +**Prerequisites:** `discover` done (columnar `graph.snapshot.bin`). `status` does not rediscover. `rules run` requires a snapshot; Kantra stays opt-in. -**Useful patterns:** +**Flags:** `--annotation` inverts `AnnotatedWith` (OR list; `@` optional). `--show-attributes` needs annotation-arg indexing (errors honestly until indexed). `--scope` + `--scope-mode inside|outside|crossing` (or `--exclude-scope`). `--file` / `--class` / `--line` disambiguate. Edge rows use keyed `source`/`target` (never positional). Omit `SYMBOL` on `relations` for set-wide typed-edge scans. -```bash -# Incoming callers of X -rgctl -f json gql "MATCH (a:Function)-[:CALLS]->(b:Function) WHERE b.name = 'checkout' RETURN a,b LIMIT 20" -# Outgoing callees of X -rgctl -f json gql "MATCH (a:Function)-[:CALLS]->(b:Function) WHERE a.name = 'checkout' RETURN a,b LIMIT 20" -# Name search (prefix or suffix only — *middle* silently returns 0) -rgctl -f json gql "MATCH (n:Function) WHERE n.name LIKE '*Service' RETURN n LIMIT 20" -# Communities macro -rgctl -f json gql --macro-name all_communities unused -``` +**Pitfalls:** Exact name is O(1) hash; prefix/contains/`--scope` may scan. Ambiguous symbols emit candidates (`error: ambiguous_symbol` JSON under `-f json`). Annotation argument values are not in the graph yet. Warm caches invalidate wall-time claims — label cold vs warm. Do not scrape stderr; parse `schema_version` on stdout. Do not treat Kantra as the only search path — use annotation/import first. -**Pitfalls:** `--macro-name` still needs a positional query arg — pass `unused`. `--explain` plan is text-mode only. rgctl GQL is a **subset of Cypher** — no `COUNT`, `ORDER BY`, `GROUP BY`, or aggregation functions. CALLS edges are static — interface / dynamic dispatch (receiver methods, virtual calls, trait impls) may not appear; if a `CALLS*1..N` query returns 0 edges for a method you know is called, fall back to `grep` for call sites. If LIKE on function names returns 0 for a concept (e.g. "ingress", "gateway"), it likely lives in package/directory names, type names, or community labels — try `communities list`, `semantic query`, or broaden the LIKE to non-Function node types before concluding nothing exists. +**Agent should report:** counts, lean names/files, keyed edge pairs — not full node dumps. -**Agent should report:** matching symbols, files, hop relationships — not raw row dumps. +**See:** OpenSpec `add-migration-search-primitives` (+ `add-structured-query-cli`). --- @@ -247,7 +252,7 @@ rgctl -f json slice src/main/java/com/example/ecommerce/service/CartService.java **Purpose:** Raw CFG / PDG / dominator view for one function. -**Prerequisites:** `discover --with-cfg`. Symbol only — **no** `--class` (disambiguate via blast-radius / GQL first). +**Prerequisites:** `discover --with-cfg`. Symbol only — **no** `--class` (disambiguate via `find` / `blast-radius` / `callers` first). **Layer flags:** `cfg --prune` drops unreachable blocks before display. `pdg --edge-layer data|control` filters to one dependence type (default `all`); `--def-use` adds def-use variable lists per node. `dom --frontiers` prints dominance frontiers instead of just the tree. @@ -322,7 +327,7 @@ rgctl -f json cpg function '' rgctl -f json blast-radius '' ``` -Loop over `top[]` UUIDs and resolve each. GQL `WHERE n.id = ''` does **not** work (node id is not a queryable property). +Loop over `top[]` UUIDs and resolve each with `cpg function` / `blast-radius` (node id is not a `find` name). **Agent should report:** top hotspot symbols (resolve UUIDs first), modularity/community count when requested. @@ -337,7 +342,7 @@ rgctl semantic index [--embedder vocab|hash|onnx|code-daemon] [--embed-bodies] [ [--dimensions N] [--incremental] [--diffuse] [--diffuse-alpha F] [--diffuse-iters N] [--diffuse-bidirectional] rgctl semantic distill --matrix PATH [--embedder code-daemon|hash|onnx] [--tokens PATH] [--dimensions N] rgctl -f json semantic query "…" [--limit N] [--scope function|community] \ - [--expand neighbors|blast|gql|all] [--expand-depth N] [--no-fusion] [--candidate-pool N] [--keyword-and] + [--expand neighbors|blast|all] [--expand-depth N] [--no-fusion] [--candidate-pool N] [--keyword-and] ``` **Purpose:** Natural-language / keyword find of functions (and community-scoped search), with optional one-shot expansion into graph context. @@ -346,7 +351,7 @@ rgctl -f json semantic query "…" [--limit N] [--scope function|community] \ **Index tuning:** `--dimensions` (default 256, multiple of 8) trades index size for precision. `--incremental` (default true) reuses embeddings for unchanged `code_hash`. `--diffuse` blends each embedding toward its call-graph neighbors' mean (Jacobi iterations via `--diffuse-alpha`/`--diffuse-iters`; `--diffuse-bidirectional` includes callers, not just callees) — useful when bare-name/docstring signal is weak and callers/callees disambiguate intent; `--no-diffuse` forces it off. -**Query expansion:** `--expand neighbors` pulls CALLS neighbors of top hits, `--expand blast` runs blast-radius on top hits, `--expand gql` returns a ready GQL query, `--expand all` does all three — use when the user's NL query implies "and show me what's connected," so you skip a manual follow-up call. `--expand-depth` controls hop depth for `neighbors`/`gql` expansion (default 1). `--no-fusion` returns pure Hamming top-k (skip late-fusion re-ranking — rarely needed). `--candidate-pool` widens/narrows the pre-fusion candidate set (default 256). `--keyword-and` requires all query keywords to match entry metadata (stricter than default OR). +**Query expansion:** `--expand neighbors` pulls CALLS neighbors of top hits, `--expand blast` runs blast-radius on top hits, `--expand all` combines those — use when the user's NL query implies "and show me what's connected," so you skip a manual follow-up call. `--expand-depth` controls hop depth for `neighbors` expansion (default 1). Prefer follow-up `callers` / `blast-radius` over any Cypher expand mode. `--no-fusion` returns pure Hamming top-k (skip late-fusion re-ranking — rarely needed). `--candidate-pool` widens/narrows the pre-fusion candidate set (default 256). `--keyword-and` requires all query keywords to match entry metadata (stricter than default OR). **Sample** (default vocab, query `checkout cart`): @@ -374,7 +379,7 @@ rgctl -f json semantic query "…" [--limit N] [--scope function|community] \ **Pitfalls:** Query without index fails. Restart `serve` after rebuilding index for dashboard search. **Large repos (100K+ nodes):** `--scope community` may return only singleton communities because label-propagation produces very granular clusters. For subsystem ownership on large repos, prefer `communities list` + grep labels over `--scope community`. -**Agent should report:** top hit names, files, scores (`score` / `fused_score`); keep `node_id` for follow-up GQL — not every hit. +**Agent should report:** top hit names, files, scores (`score` / `fused_score`); keep `node_id` for follow-up `callers` / `blast-radius` — not every hit. --- @@ -399,7 +404,7 @@ rgctl -f json semantic query "…" [--limit N] [--scope function|community] \ } ``` -**Agent should report:** top labels + sizes; use GQL `community_id` for members. +**Agent should report:** top labels + sizes; use `inventory --by community` for census; explore ownership via `semantic query --scope community` / `blast-radius` (responses may include `community_id`). --- @@ -523,7 +528,7 @@ rgctl cpg export --format graphson --output cpg.json [--path-contains src/] \ **Prerequisites:** `discover` done. -**Pitfalls (critical):** `--query` uses **filter** syntax — `name:Foo`, `type:Function`, `all` — **not** GQL `MATCH … RETURN`. Agents must not pass MATCH strings to `--query`. +**Pitfalls (critical):** `--query` uses **filter** syntax — `name:Foo`, `type:Function`, `all` — **not** Cypher `MATCH … RETURN`. Agents must not pass MATCH strings to `--query`. **Agent should report:** output path + format; confirm filter used. diff --git a/agent-pack/out/agents/host-agents/skills/rgctl/references/communities-and-policy.md b/agent-pack/out/agents/host-agents/skills/rgctl/references/communities-and-policy.md index 1838f662..1804f37c 100644 --- a/agent-pack/out/agents/host-agents/skills/rgctl/references/communities-and-policy.md +++ b/agent-pack/out/agents/host-agents/skills/rgctl/references/communities-and-policy.md @@ -65,11 +65,15 @@ rgctl -f json communities list #### Query Community Members -Once you have a community ID, list its members: +Once you have a community ID from `communities list`, explore ownership and members via structured verbs: **CLI:** ```bash -rgctl -f json gql "MATCH (f:Function) WHERE f.community_id = '12715' RETURN f LIMIT 20" +rgctl -f json inventory --by community +rgctl -f json semantic query "