diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 768b04f5..15603022 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -149,8 +149,34 @@ jobs: run: | set -euo pipefail cd dist - shasum -a 256 rgctl-* > SHA256SUMS.txt + # Artifacts may land flat (merge-multiple) or under per-target dirs. + mapfile -t archives < <( + find . -type f \( -name 'rgctl-*.tar.gz' -o -name 'rgctl-*.zip' \) | sort + ) + if [[ ${#archives[@]} -eq 0 ]]; then + echo "error: no rgctl archives under dist/; refusing empty SHA256SUMS.txt" >&2 + find . -type f | head -50 >&2 + exit 1 + fi + # Flatten into dist/ so softprops uploads a single flat file set. + for f in "${archives[@]}"; do + base="$(basename "$f")" + if [[ "$f" != "./$base" ]]; then + mv -f "$f" "./$base" + fi + done + : > SHA256SUMS.txt + for f in rgctl-*.tar.gz rgctl-*.zip; do + [[ -f "$f" ]] || continue + shasum -a 256 "$f" >> SHA256SUMS.txt + done + if [[ ! -s SHA256SUMS.txt ]]; then + echo "error: SHA256SUMS.txt is empty" >&2 + exit 1 + fi + echo "--- SHA256SUMS.txt ---" cat SHA256SUMS.txt + echo "lines=$(wc -l < SHA256SUMS.txt)" - name: Resolve release tag id: meta diff --git a/.gitignore b/.gitignore index 1b1c1afa..f6a8fc4c 100644 --- a/.gitignore +++ b/.gitignore @@ -2,6 +2,7 @@ .cursor/ .claude/ .opencode/ +.agent # OpenSpec change proposals (local only — do not commit) openspec/ .scratch/ diff --git a/README.md b/README.md index 76d134cb..ddb80d7f 100644 --- a/README.md +++ b/README.md @@ -29,16 +29,15 @@ The LLM reasons on **summaries and facts**, not raw repo grep — fewer tokens, ## Quick Start -**1. Install** from [GitHub Releases](https://github.com/sshaaf/rgctl/releases/latest) (binary **`rgctl`**) or build from source ([Installation docs](docs/installation.md)): +**1. Install** from [GitHub Releases](https://github.com/sshaaf/rgctl/releases/latest) (binary **`rgctl`**) or build from source ([Installation docs](docs/installation.md) — glibc / Ubuntu 22.04 caveat, Rust **1.88+**, and `--no-default-features` if ONNX/`ort` link fails): ```bash git clone https://github.com/sshaaf/rgctl.git cd rgctl git lfs pull # only if you use `semantic index --embedder code-daemon` (~206 MB) cargo build --release --bin rgctl - +# If ort-sys fails: cargo build --release --bin rgctl --no-default-features ``` - **2. Discover (Index your repo):** Run this once to build the graph and reachability caches. Artifacts land in `{repo}/.rgctl/`. diff --git a/crates/rgctl-analysis/src/ast_skeleton.rs b/crates/rgctl-analysis/src/ast_skeleton.rs index 77edacd4..f760435d 100644 --- a/crates/rgctl-analysis/src/ast_skeleton.rs +++ b/crates/rgctl-analysis/src/ast_skeleton.rs @@ -5,7 +5,7 @@ use crate::language_profile::{function_kinds_for, parse_source}; use rgctl_error::{Error, Result}; -use rgctl_plugin_helpers::extract_name_from_node; +use rgctl_plugin_helpers::{ecmascript_function_symbol_name, extract_name_from_node}; use serde::{Deserialize, Serialize}; use std::fs; use std::path::{Path, PathBuf}; @@ -140,11 +140,13 @@ pub fn build_function_skeleton( let bytes = source.as_bytes(); let tree = parse_source(language, bytes)?; let kinds = function_kinds_for(language)?; - let func = find_function(tree.root_node(), bytes, function_name, kinds).ok_or_else(|| { - Error::NotFound(format!( - "function '{function_name}' not found for AST skeleton" - )) - })?; + let func = find_function(tree.root_node(), bytes, function_name, kinds, language).ok_or_else( + || { + Error::NotFound(format!( + "function '{function_name}' not found for AST skeleton" + )) + }, + )?; let mut nodes = Vec::new(); let root_id = 0u32; nodes.push(AstSkeletonNode { @@ -169,17 +171,22 @@ fn find_function<'a>( source: &[u8], name: &str, kinds: &[&str], + language: &str, ) -> Option> { if kinds.contains(&node.kind()) { - if let Ok(Some(n)) = extract_name_from_node(node, source) { - if n == name { - return Some(node); + let resolved = match language { + "javascript" | "js" | "typescript" | "ts" => { + ecmascript_function_symbol_name(node, source) } + _ => extract_name_from_node(node, source).ok().flatten(), + }; + if resolved.as_deref() == Some(name) { + return Some(node); } } let mut cursor = node.walk(); for child in node.children(&mut cursor) { - if let Some(found) = find_function(child, source, name, kinds) { + if let Some(found) = find_function(child, source, name, kinds, language) { return Some(found); } } diff --git a/crates/rgctl-analysis/src/cfg_builder.rs b/crates/rgctl-analysis/src/cfg_builder.rs index bae1ec34..80fdc832 100644 --- a/crates/rgctl-analysis/src/cfg_builder.rs +++ b/crates/rgctl-analysis/src/cfg_builder.rs @@ -4,7 +4,7 @@ use crate::cfg::{BasicBlock, BlockId, CfgEdgeType, ControlFlowGraph, Statement, use crate::def_use::extract_def_use; use crate::language_profile::{function_kinds_for, parse_source}; use rgctl_error::{Error, Result}; -use rgctl_plugin_helpers::extract_name_from_node; +use rgctl_plugin_helpers::{ecmascript_function_symbol_name, extract_name_from_node}; use smallvec::SmallVec; use std::collections::{HashMap, HashSet}; use tracing::warn; @@ -23,7 +23,7 @@ pub fn build_cfg_for_function( let tree = parse_language(language, bytes)?; let function_kinds = function_kinds_for(language)?; let func_node = - find_function_by_name(tree.root_node(), bytes, function_name, function_kinds) + find_function_by_name(tree.root_node(), bytes, function_name, function_kinds, language) .ok_or_else(|| Error::NotFound(format!("function '{function_name}' not found")))?; build_cfg_from_function_node(language, func_node, bytes, function_name) } @@ -83,8 +83,9 @@ pub fn build_cfg_for_function_in_tree( function_name: &str, ) -> Result { let function_kinds = function_kinds_for(language)?; - let func_node = find_function_by_name(tree.root_node(), source, function_name, function_kinds) - .ok_or_else(|| Error::NotFound(format!("function '{function_name}' not found")))?; + let func_node = + find_function_by_name(tree.root_node(), source, function_name, function_kinds, language) + .ok_or_else(|| Error::NotFound(format!("function '{function_name}' not found")))?; build_cfg_from_function_node(language, func_node, source, function_name) } @@ -96,7 +97,7 @@ pub fn index_function_locations( let tree = parse_language(language, source)?; let function_kinds = function_kinds_for(language)?; let mut index = HashMap::new(); - collect_function_locations(tree.root_node(), source, function_kinds, &mut index); + collect_function_locations(tree.root_node(), source, function_kinds, language, &mut index); Ok((tree, index)) } @@ -104,12 +105,13 @@ fn collect_function_locations( root: Node<'_>, source: &[u8], function_kinds: &[&str], + language: &str, out: &mut HashMap, ) { let mut stack = vec![root]; while let Some(node) = stack.pop() { if function_kinds.contains(&node.kind()) { - if let Ok(Some(func_name)) = extract_name_from_node(node, source) { + if let Some(func_name) = callable_name_for_cfg(node, source, language) { out.entry(func_name).or_insert(FunctionLocation { start_byte: node.start_byte(), end_byte: node.end_byte(), @@ -127,11 +129,22 @@ fn parse_language(language: &str, source: &[u8]) -> Result { parse_source(language, source) } +/// Resolve the Function name CFG should look up, matching language-plugin naming. +fn callable_name_for_cfg(node: Node<'_>, source: &[u8], language: &str) -> Option { + match language { + "javascript" | "js" | "typescript" | "ts" => { + ecmascript_function_symbol_name(node, source) + } + _ => extract_name_from_node(node, source).ok().flatten(), + } +} + fn find_function_by_name<'a>( node: Node<'a>, source: &[u8], name: &str, function_kinds: &[&str], + language: &str, ) -> Option> { // Java instance initializer blocks are bare `block` children of `class_body`, // extracted as synthetic `N` functions (not a distinct CST kind). @@ -157,7 +170,7 @@ fn find_function_by_name<'a>( let mut stack = vec![node]; while let Some(node) = stack.pop() { if function_kinds.contains(&node.kind()) { - if let Ok(Some(func_name)) = extract_name_from_node(node, source) { + if let Some(func_name) = callable_name_for_cfg(node, source, language) { if func_name == name { return Some(node); } @@ -6026,6 +6039,36 @@ function classify(v) { assert!(cfg.blocks.len() >= 5); } + #[test] + fn test_typescript_bound_arrow_cfg() { + let code = r#" +export const dup = (n: number): number => n + 1; +export const uniqueOne = (n: number): number => dup(n); + +export function dupDecl(n: number): number { return n + 1; } +export function uniqueDecl(n: number): number { return dupDecl(n); } +"#; + let arrow_cfg = build_cfg_for_function("typescript", code, "uniqueOne").unwrap(); + assert!( + arrow_cfg.blocks.len() >= 2, + "bound arrow should have a CFG, got {} blocks", + arrow_cfg.blocks.len() + ); + let decl_cfg = build_cfg_for_function("typescript", code, "uniqueDecl").unwrap(); + assert!(decl_cfg.blocks.len() >= 2); + } + + #[test] + fn test_javascript_bound_arrow_cfg() { + let code = r#" +const tickDown = (n) => n - 1; +function tickDecl(n) { return n - 1; } +"#; + let cfg = build_cfg_for_function("javascript", code, "tickDown").unwrap(); + assert!(cfg.blocks.len() >= 2); + let _ = build_cfg_for_function("javascript", code, "tickDecl").unwrap(); + } + #[test] fn test_unsupported_language() { let result = build_cfg_for_function("brainfuck", "+++", "main"); diff --git a/crates/rgctl-analysis/src/language_profile.rs b/crates/rgctl-analysis/src/language_profile.rs index 1198436d..218a23e6 100644 --- a/crates/rgctl-analysis/src/language_profile.rs +++ b/crates/rgctl-analysis/src/language_profile.rs @@ -95,6 +95,7 @@ const PROFILES: &[LanguageAnalysisProfile] = &[ extensions: &["js", "jsx", "mjs", "cjs"], function_kinds: &[ "function_declaration", + "function_expression", "method_definition", "arrow_function", ], @@ -107,6 +108,7 @@ const PROFILES: &[LanguageAnalysisProfile] = &[ extensions: &["ts", "tsx"], function_kinds: &[ "function_declaration", + "function_expression", "method_definition", "arrow_function", ], diff --git a/crates/rgctl-analysis/src/macro_call_lookup.rs b/crates/rgctl-analysis/src/macro_call_lookup.rs index 6d18a189..27d0890c 100644 --- a/crates/rgctl-analysis/src/macro_call_lookup.rs +++ b/crates/rgctl-analysis/src/macro_call_lookup.rs @@ -92,11 +92,7 @@ pub fn parse_fqn_symbol( let scope_hint = &input[..idx]; let target_name = &input[idx + 2..]; - if scope_hint.contains('/') - || scope_hint.contains('\\') - || scope_hint.ends_with(".java") - || scope_hint.ends_with(".rs") - { + if looks_like_file_scope(scope_hint) { ParsedSymbol { target_name: target_name.to_string(), class_filter: explicit_class, @@ -118,6 +114,19 @@ pub fn parse_fqn_symbol( } } +fn looks_like_file_scope(scope: &str) -> bool { + if scope.contains('/') || scope.contains('\\') { + return true; + } + matches!( + scope.rsplit_once('.').map(|(_, ext)| ext), + Some( + "java" | "rs" | "ts" | "tsx" | "js" | "jsx" | "mjs" | "cjs" | "py" | "go" | "php" + | "cs" | "c" | "h" | "cpp" | "cc" | "cxx" | "hpp" | "hh" | "rb" + ) + ) +} + /// Extract lowercase language id from graph node metadata or file extension. pub fn language_from_node(node: &Node) -> String { node.get_property("language") @@ -829,6 +838,24 @@ mod tests { assert_eq!(parsed.file_filter.as_deref(), Some("src/Foo.java")); } + #[test] + fn parse_fqn_typescript_file_scope() { + let parsed = parse_fqn_symbol("fixtures/caseB_one.ts::dup", None, None); + assert_eq!(parsed.target_name, "dup"); + assert_eq!( + parsed.file_filter.as_deref(), + Some("fixtures/caseB_one.ts") + ); + assert!(parsed.class_filter.is_none()); + } + + #[test] + fn parse_fqn_bare_ts_filename_is_file_scope() { + let parsed = parse_fqn_symbol("caseB_one.ts::dup", None, None); + assert_eq!(parsed.file_filter.as_deref(), Some("caseB_one.ts")); + assert!(parsed.class_filter.is_none()); + } + #[test] fn resolve_symbol_uuid_filters_by_class() { let candidates = vec![ diff --git a/crates/rgctl-lang-javascript/src/plugin.rs b/crates/rgctl-lang-javascript/src/plugin.rs index 34c3d6e5..261be1c4 100644 --- a/crates/rgctl-lang-javascript/src/plugin.rs +++ b/crates/rgctl-lang-javascript/src/plugin.rs @@ -9,7 +9,8 @@ use rgctl_plugin_api::{ SourceLocation, Symbol, SymbolType, }; use rgctl_plugin_helpers::{ - extract_cjs_require_symbols, extract_class_extends_relations, extract_import_symbols, + bound_function_expression_name, extract_cjs_require_symbols, extract_class_extends_relations, + extract_import_symbols, is_ecmascript_function_node, is_function_expression_kind, simple_type_name, type_name_from_node, }; use rgctl_semantic::type_inference::TypeInferencer; @@ -56,11 +57,17 @@ impl JavaScriptPlugin { let mut cursor = node.walk(); let mut name = None; let mut parameters = Vec::new(); + // Arrow / function expressions put the binding on a parent; their first + // identifier child is often a parameter (e.g. `x => x`), not a name. + let is_expr = is_function_expression_kind(node.kind()) + || (node.kind() == "function" + && node.is_named() + && node.child_by_field_name("name").is_none()); for child in node.children(&mut cursor) { match child.kind() { "identifier" | "property_identifier" => { - if name.is_none() { + if name.is_none() && !is_expr { name = Some(child.utf8_text(source)?.to_string()); } } @@ -71,18 +78,27 @@ impl JavaScriptPlugin { } } - let raw_name = name.unwrap_or_else(|| "anonymous".to_string()); - - // Infer types for parameters - let function_source = node.utf8_text(source).unwrap_or(""); - let inferencer = TypeInferencer::new(); - let inferred_types = inferencer.infer_javascript(function_source); - - // Update parameters with inferred types - for param in &mut parameters { - if param.param_type.is_none() { - if let Some(inference) = inferred_types.get(¶m.name) { - param.param_type = Some(format!("{:?}", inference.inferred)); + let raw_name = name + .or_else(|| bound_function_expression_name(node, source)) + .unwrap_or_else(|| { + if is_expr { + format!("anonymous@L{}", node.start_position().row + 1) + } else { + "anonymous".to_string() + } + }); + + // Skip regex-based inference when there are no params (common for arrows / + // keyword-filtered nodes). Compiling/running inference per function was a + // Node-corpus hot-path cost (AGENTS.md clone / CPU discipline). + if !parameters.is_empty() { + let function_source = node.utf8_text(source).unwrap_or(""); + let inferred_types = TypeInferencer::new().infer_javascript(function_source); + for param in &mut parameters { + if param.param_type.is_none() { + if let Some(inference) = inferred_types.get(¶m.name) { + param.param_type = Some(format!("{:?}", inference.inferred)); + } } } } @@ -347,10 +363,10 @@ impl JavaScriptPlugin { file_path: &str, symbols: &mut Vec, ) -> Result<()> { + if is_ecmascript_function_node(node) { + symbols.push(self.extract_function(node, source, file_path)?); + } match node.kind() { - "function_declaration" | "function" | "method_definition" | "arrow_function" => { - symbols.push(self.extract_function(node, source, file_path)?); - } "class_declaration" => { symbols.push(self.extract_class(node, source, file_path)?); } @@ -856,6 +872,55 @@ mod tests { assert_eq!(symbols.len(), 1); assert_eq!(symbols[0].symbol_type, SymbolType::Function); + assert_eq!(symbols[0].name, "multiply"); + } + + #[test] + fn test_extract_named_arrows_distinct() { + let plugin = JavaScriptPlugin::new().unwrap(); + let source = br#" +function declaredAdd(a, b) { return a + b; } +const arrowAdd = (a, b) => a + b; +const arrowHelper = () => 1; +const api = { fetchAll: async () => 0 }; +[1].map(x => x + 1); +"#; + let symbols = plugin + .extract_symbols(Path::new("arrows.js"), source) + .unwrap(); + let names: Vec<_> = symbols + .iter() + .filter(|s| s.symbol_type == SymbolType::Function) + .map(|s| s.name.as_str()) + .collect(); + assert!(names.contains(&"declaredAdd"), "{names:?}"); + assert!(names.contains(&"arrowAdd"), "{names:?}"); + assert!(names.contains(&"arrowHelper"), "{names:?}"); + assert!(names.contains(&"fetchAll"), "{names:?}"); + assert!( + names.iter().any(|n| n.starts_with("anonymous@L")), + "callback should stay span-disambiguated anonymous: {names:?}" + ); + assert_eq!( + names.iter().filter(|n| n.starts_with("anonymous@L")).count(), + 1, + "only the map callback should be anonymous, got {names:?}" + ); + } + + #[test] + fn test_function_declaration_has_no_anonymous_duplicate() { + let plugin = JavaScriptPlugin::new().unwrap(); + let source = b"export function gamma(n) { return n + 1; }"; + let symbols = plugin + .extract_symbols(Path::new("gamma.js"), source) + .unwrap(); + let fns: Vec<_> = symbols + .iter() + .filter(|s| s.symbol_type == SymbolType::Function) + .map(|s| s.name.as_str()) + .collect(); + assert_eq!(fns, vec!["gamma"], "unexpected functions: {fns:?}"); } #[test] diff --git a/crates/rgctl-lang-typescript/src/plugin.rs b/crates/rgctl-lang-typescript/src/plugin.rs index 3a596233..10593daa 100644 --- a/crates/rgctl-lang-typescript/src/plugin.rs +++ b/crates/rgctl-lang-typescript/src/plugin.rs @@ -6,7 +6,8 @@ use rgctl_plugin_api::*; use rgctl_plugin_api::{Error, Result}; use rgctl_plugin_helpers::{ - extract_class_extends_relations, extract_import_symbols, find_child_kind, simple_type_name, + bound_function_expression_name, extract_class_extends_relations, extract_import_symbols, + find_child_kind, is_ecmascript_function_node, is_function_expression_kind, simple_type_name, type_name_from_node, }; use std::collections::HashMap; @@ -161,11 +162,17 @@ impl TypeScriptPlugin { let mut parameters = Vec::new(); let mut return_type = None; let mut modifiers = Vec::new(); + // Arrow / function expressions put the binding on a parent; their first + // identifier child is often a parameter (e.g. `x => x`), not a name. + let is_expr = is_function_expression_kind(node.kind()) + || (node.kind() == "function" + && node.is_named() + && node.child_by_field_name("name").is_none()); for child in node.children(&mut cursor) { match child.kind() { "identifier" | "property_identifier" => { - if name.is_none() { + if name.is_none() && !is_expr { name = Some(child.utf8_text(source)?.to_string()); } } @@ -188,7 +195,15 @@ impl TypeScriptPlugin { } } - let raw_name = name.unwrap_or_else(|| "anonymous".to_string()); + let raw_name = name + .or_else(|| bound_function_expression_name(node, source)) + .unwrap_or_else(|| { + if is_expr { + format!("anonymous@L{}", node.start_position().row + 1) + } else { + "anonymous".to_string() + } + }); let is_constructor = raw_name == "constructor" && node.kind() == "method_definition"; let class_name = if is_constructor { self.find_containing_class_name(node, source) @@ -586,10 +601,10 @@ impl TypeScriptPlugin { symbols: &mut Vec, plugin: &TypeScriptPlugin, ) -> Result<()> { + if is_ecmascript_function_node(node) { + symbols.push(plugin.extract_function(node, source, file_path)?); + } match node.kind() { - "function_declaration" | "function" | "method_definition" | "arrow_function" => { - symbols.push(plugin.extract_function(node, source, file_path)?); - } "class_declaration" | "abstract_class_declaration" => { symbols.push(plugin.extract_class(node, source, file_path)?); } @@ -1264,6 +1279,78 @@ mod tests { assert_eq!(add_fn.parameters.len(), 2); } + #[test] + fn test_extract_arrow_function() { + let plugin = TypeScriptPlugin::new().unwrap(); + let source = b"const multiply = (x: number, y: number): number => x * y;"; + let symbols = plugin + .extract_symbols(Path::new("test.ts"), source) + .unwrap(); + + assert_eq!(symbols.len(), 1); + assert_eq!(symbols[0].symbol_type, SymbolType::Function); + assert_eq!(symbols[0].name, "multiply"); + } + + #[test] + fn test_extract_named_arrows_distinct() { + let plugin = TypeScriptPlugin::new().unwrap(); + let source = br#" +export function declaredAdd(a: number, b: number): number { return a + b; } +export const arrowAdd = (a: number, b: number): number => a + b; +const arrowHelper = (): number => 1; +const api = { fetchAll: async (): Promise => 0 }; +class C { foo = (): number => 2; } +[1].map((x) => x + 1); +"#; + let symbols = plugin + .extract_symbols(Path::new("arrows.ts"), source) + .unwrap(); + let names: Vec<_> = symbols + .iter() + .filter(|s| s.symbol_type == SymbolType::Function) + .map(|s| s.name.as_str()) + .collect(); + assert!(names.contains(&"declaredAdd"), "{names:?}"); + assert!(names.contains(&"arrowAdd"), "{names:?}"); + assert!(names.contains(&"arrowHelper"), "{names:?}"); + assert!(names.contains(&"fetchAll"), "{names:?}"); + assert!(names.contains(&"foo"), "{names:?}"); + assert!( + names.iter().any(|n| n.starts_with("anonymous@L")), + "callback should stay span-disambiguated anonymous: {names:?}" + ); + // Declarations must not also emit a body-less keyword duplicate. + let decl_dupes: Vec<_> = names + .iter() + .filter(|n| n.starts_with("anonymous@L")) + .collect(); + assert_eq!( + decl_dupes.len(), + 1, + "only the map callback should be anonymous, got {names:?}" + ); + } + + #[test] + fn test_function_declaration_has_no_anonymous_duplicate() { + let plugin = TypeScriptPlugin::new().unwrap(); + let source = br#" +export function gamma(n: number): number { + return n + 1; +} +"#; + let symbols = plugin + .extract_symbols(Path::new("gamma.ts"), source) + .unwrap(); + let fns: Vec<_> = symbols + .iter() + .filter(|s| s.symbol_type == SymbolType::Function) + .map(|s| s.name.as_str()) + .collect(); + assert_eq!(fns, vec!["gamma"], "unexpected functions: {fns:?}"); + } + #[test] fn test_extract_class() { let plugin = TypeScriptPlugin::new().unwrap(); diff --git a/crates/rgctl-plugin-helpers/src/ecmascript.rs b/crates/rgctl-plugin-helpers/src/ecmascript.rs index 492fdb8c..8b45d50f 100644 --- a/crates/rgctl-plugin-helpers/src/ecmascript.rs +++ b/crates/rgctl-plugin-helpers/src/ecmascript.rs @@ -375,6 +375,168 @@ pub fn find_child_kind<'a>(node: Node<'a>, kind: &str) -> Option> { None } +/// True for arrow / anonymous function expressions (not declarations or methods). +pub fn is_function_expression_kind(kind: &str) -> bool { + matches!(kind, "arrow_function" | "function_expression") +} + +/// True when `node` is a real JS/TS callable CST node (not the anonymous +/// `'function'` keyword token that appears under `function_declaration`). +/// +/// Tree-sitter emits that keyword as `kind == "function"` with `is_named() == false`. +/// Matching it as a Function symbol produces a body-less `anonymous@L{N}` duplicate +/// on every declaration. +pub fn is_ecmascript_function_node(node: Node<'_>) -> bool { + match node.kind() { + "function_declaration" + | "function_expression" + | "method_definition" + | "arrow_function" + | "generator_function" + | "generator_function_declaration" => true, + "function" => node.is_named(), + _ => false, + } +} + +/// Graph Function name for a JS/TS callable node — matches language-plugin naming. +/// +/// Bound arrows / function expressions use the parent binding; true callbacks use +/// `anonymous@L{line}`. Declarations and methods use the CST name field. +pub fn ecmascript_function_symbol_name(node: Node<'_>, source: &[u8]) -> Option { + if !is_ecmascript_function_node(node) { + return None; + } + + let is_expr = is_function_expression_kind(node.kind()) + || (node.kind() == "function" + && node.is_named() + && node.child_by_field_name("name").is_none()); + + if is_expr { + return Some( + bound_function_expression_name(node, source).unwrap_or_else(|| { + format!("anonymous@L{}", node.start_position().row + 1) + }), + ); + } + + if let Some(name_node) = node.child_by_field_name("name") { + if let Ok(text) = name_node.utf8_text(source) { + let trimmed = text.trim(); + if !trimmed.is_empty() { + return Some(trimmed.to_string()); + } + } + } + + // method_definition / fallbacks + let mut cursor = node.walk(); + for child in node.children(&mut cursor) { + if matches!( + child.kind(), + "identifier" | "property_identifier" | "private_property_identifier" + ) { + if let Ok(text) = child.utf8_text(source) { + let trimmed = text.trim(); + if !trimmed.is_empty() { + return Some(trimmed.to_string()); + } + } + } + } + None +} + +/// Resolve a display name for `arrow_function` / `function_expression` from a +/// parent binding: `variable_declarator`, object `pair`, or class field. +/// +/// Returns `None` for true callbacks (e.g. `arr.map(x => …)`). +pub fn bound_function_expression_name(node: Node, source: &[u8]) -> Option { + if !is_function_expression_kind(node.kind()) + && !(node.kind() == "function" + && node.is_named() + && node.child_by_field_name("name").is_none()) + { + return None; + } + + let mut current = node; + for _ in 0..12 { + let parent = current.parent()?; + match parent.kind() { + "variable_declarator" => { + return identifier_text(parent.child_by_field_name("name")?, source); + } + "pair" => { + let key = parent + .child_by_field_name("key") + .or_else(|| find_direct_child_kinds(parent, &["property_identifier", "identifier", "string", "number"]))?; + return property_key_text(key, source); + } + "public_field_definition" | "field_definition" | "property_definition" => { + let name = parent.child_by_field_name("name").or_else(|| { + find_direct_child_kinds( + parent, + &[ + "property_identifier", + "private_property_identifier", + "identifier", + ], + ) + })?; + return identifier_text(name, source); + } + "assignment_expression" => { + let left = parent.child_by_field_name("left")?; + return match left.kind() { + "identifier" => identifier_text(left, source), + "member_expression" | "subscript_expression" => left + .child_by_field_name("property") + .and_then(|p| property_key_text(p, source)), + _ => None, + }; + } + // Peel type / grouping wrappers without consuming a binding. + "parenthesized_expression" + | "as_expression" + | "type_assertion" + | "satisfies_expression" + | "non_null_expression" + | "await_expression" + | "ternary_expression" => { + current = parent; + } + _ => return None, + } + } + None +} + +fn find_direct_child_kinds<'a>(node: Node<'a>, kinds: &[&str]) -> Option> { + let mut cursor = node.walk(); + for child in node.children(&mut cursor) { + if kinds.contains(&child.kind()) { + return Some(child); + } + } + None +} + +fn identifier_text(node: Node, source: &[u8]) -> Option { + node.utf8_text(source).ok().map(|s| s.trim().to_string()).filter(|s| !s.is_empty()) +} + +fn property_key_text(node: Node, source: &[u8]) -> Option { + let raw = node.utf8_text(source).ok()?.trim(); + let trimmed = raw.trim_matches(['"', '\'', '`']); + if trimmed.is_empty() { + None + } else { + Some(trimmed.to_string()) + } +} + fn push_relation( from: &str, to: &str, @@ -514,4 +676,98 @@ mod tests { .any(|r| r.relation_type == RelationType::Extends && r.to == "Error") ); } + + fn find_kind<'a>(node: Node<'a>, kind: &str) -> Option> { + if node.kind() == kind { + return Some(node); + } + let mut cursor = node.walk(); + for child in node.children(&mut cursor) { + if let Some(found) = find_kind(child, kind) { + return Some(found); + } + } + None + } + + #[test] + fn test_bound_name_const_arrow() { + let source = "const multiply = (x, y) => x * y;"; + let tree = parse_ts(source); + let arrow = find_kind(tree.root_node(), "arrow_function").unwrap(); + assert_eq!( + bound_function_expression_name(arrow, source.as_bytes()).as_deref(), + Some("multiply") + ); + } + + #[test] + fn test_bound_name_object_property_arrow() { + let source = "const api = { fetchAll: async () => 1 };"; + let tree = parse_ts(source); + let arrow = find_kind(tree.root_node(), "arrow_function").unwrap(); + assert_eq!( + bound_function_expression_name(arrow, source.as_bytes()).as_deref(), + Some("fetchAll") + ); + } + + #[test] + fn test_bound_name_class_field_arrow() { + let source = "class C { foo = () => 2; }"; + let tree = parse_ts(source); + let arrow = find_kind(tree.root_node(), "arrow_function").unwrap(); + assert_eq!( + bound_function_expression_name(arrow, source.as_bytes()).as_deref(), + Some("foo") + ); + } + + #[test] + fn test_callback_arrow_has_no_bound_name() { + let source = "[1].map(x => x + 1);"; + let tree = parse_js(source); + let arrow = find_kind(tree.root_node(), "arrow_function").unwrap(); + assert_eq!(bound_function_expression_name(arrow, source.as_bytes()), None); + } + + #[test] + fn test_keyword_function_token_is_not_a_function_node() { + let source = "export function gamma(n: number): number { return n + 1; }"; + let tree = parse_ts(source); + let mut keyword = None; + let mut decl = None; + let mut stack = vec![tree.root_node()]; + while let Some(n) = stack.pop() { + if n.kind() == "function_declaration" { + decl = Some(n); + } + if n.kind() == "function" && !n.is_named() { + keyword = Some(n); + } + let mut c = n.walk(); + for ch in n.children(&mut c) { + stack.push(ch); + } + } + let keyword = keyword.expect("anonymous 'function' keyword token"); + let decl = decl.expect("function_declaration"); + assert!(!is_ecmascript_function_node(keyword)); + assert!(is_ecmascript_function_node(decl)); + assert_eq!( + ecmascript_function_symbol_name(decl, source.as_bytes()).as_deref(), + Some("gamma") + ); + } + + #[test] + fn test_cfg_name_for_bound_arrow() { + let source = "export const uniqueOne = (n: number): number => n + 1;"; + let tree = parse_ts(source); + let arrow = find_kind(tree.root_node(), "arrow_function").unwrap(); + assert_eq!( + ecmascript_function_symbol_name(arrow, source.as_bytes()).as_deref(), + Some("uniqueOne") + ); + } } diff --git a/crates/rgctl-plugin-helpers/src/lib.rs b/crates/rgctl-plugin-helpers/src/lib.rs index 78cf1ac2..9e6343c9 100644 --- a/crates/rgctl-plugin-helpers/src/lib.rs +++ b/crates/rgctl-plugin-helpers/src/lib.rs @@ -7,8 +7,9 @@ pub mod tree_sitter; pub use complexity::ComplexityCalculator; pub use ecmascript::{ - extract_cjs_require_symbols, extract_class_extends_relations, extract_import_symbols, - find_child_kind, simple_type_name, type_name_from_node, + bound_function_expression_name, ecmascript_function_symbol_name, extract_cjs_require_symbols, + extract_class_extends_relations, extract_import_symbols, find_child_kind, + is_ecmascript_function_node, is_function_expression_kind, simple_type_name, type_name_from_node, }; pub use python::{ containing_class_name, decorator_name_and_args, decorators_for_node, diff --git a/crates/rgctl-semantic/src/type_inference.rs b/crates/rgctl-semantic/src/type_inference.rs index 1897e2ba..fe8b9992 100644 --- a/crates/rgctl-semantic/src/type_inference.rs +++ b/crates/rgctl-semantic/src/type_inference.rs @@ -4,6 +4,7 @@ use regex::Regex; use std::collections::HashMap; +use std::sync::OnceLock; /// Inferred type for a variable or parameter. #[derive(Debug, Clone, PartialEq, Eq)] @@ -42,6 +43,18 @@ pub struct TypeInference { pub confidence: f64, } +fn python_def_params_re() -> &'static Regex { + static RE: OnceLock = OnceLock::new(); + RE.get_or_init(|| Regex::new(r"def\s+\w+\s*\(([^)]*)\)").expect("python def regex")) +} + +fn javascript_fn_params_re() -> &'static Regex { + static RE: OnceLock = OnceLock::new(); + RE.get_or_init(|| { + Regex::new(r"function\s+\w+\s*\(([^)]*)\)").expect("javascript function regex") + }) +} + /// Type inferencer for dynamically typed languages. pub struct TypeInferencer; @@ -55,19 +68,17 @@ impl TypeInferencer { pub fn infer_python(&self, source: &str) -> HashMap { let mut types: HashMap = HashMap::new(); - if let Ok(re) = Regex::new(r"def\s+\w+\s*\(([^)]*)\)") { - if let Some(cap) = re.captures(source) { - for param in cap[1].split(',') { - let name = param.trim().split(':').next().unwrap_or("").trim(); - if !name.is_empty() && name != "self" { - types.insert( - name.to_string(), - TypeInference { - inferred: InferredType::Unknown, - confidence: 0.3, - }, - ); - } + if let Some(cap) = python_def_params_re().captures(source) { + for param in cap[1].split(',') { + let name = param.trim().split(':').next().unwrap_or("").trim(); + if !name.is_empty() && name != "self" { + types.insert( + name.to_string(), + TypeInference { + inferred: InferredType::Unknown, + confidence: 0.3, + }, + ); } } } @@ -94,22 +105,28 @@ impl TypeInferencer { } /// Infer types from JavaScript/TypeScript source. + /// + /// Only matches classic `function name(...)` forms. Arrow / method bodies that do + /// not contain that pattern return empty (callers should skip when params are empty). pub fn infer_javascript(&self, source: &str) -> HashMap { let mut types = HashMap::new(); - if let Ok(re) = Regex::new(r"function\s+\w+\s*\(([^)]*)\)") { - if let Some(cap) = re.captures(source) { - for param in cap[1].split(',') { - let name = param.trim(); - if !name.is_empty() { - types.insert( - name.to_string(), - TypeInference { - inferred: InferredType::Unknown, - confidence: 0.3, - }, - ); - } + // Cheap reject: regex never matches arrows / methods without a `function` keyword. + if !source.contains("function") { + return types; + } + + if let Some(cap) = javascript_fn_params_re().captures(source) { + for param in cap[1].split(',') { + let name = param.trim(); + if !name.is_empty() { + types.insert( + name.to_string(), + TypeInference { + inferred: InferredType::Unknown, + confidence: 0.3, + }, + ); } } } @@ -166,6 +183,22 @@ def calculate(x, y): assert!(types["y"].inferred.is_numeric()); } + #[test] + fn test_javascript_skips_arrows_without_function_keyword() { + let inferencer = TypeInferencer::new(); + let types = inferencer.infer_javascript("const add = (x, y) => x + y;"); + assert!(types.is_empty()); + } + + #[test] + fn test_javascript_function_params() { + let inferencer = TypeInferencer::new(); + let types = inferencer.infer_javascript("function add(x, y) { return x + y; }"); + assert!(types.contains_key("x")); + assert!(types.contains_key("y")); + assert!(types["x"].inferred.is_numeric()); + } + #[test] fn test_type_mapping() { assert_eq!(TypeInferencer::map_to_idl_type("i32"), "int64"); diff --git a/docs/installation.md b/docs/installation.md index c92fb89e..de922c86 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -53,6 +53,8 @@ Pre-built binaries are published on the project **Releases** page: | Linux (x86_64) | `rgctl-*-x86_64-unknown-linux-gnu.tar.gz` | | Windows | `rgctl-*-x86_64-pc-windows-msvc.zip` | + **Linux glibc note:** the `x86_64-unknown-linux-gnu` release is linked against a relatively new glibc (CI runner). It may **not** run on older LTS hosts such as **Ubuntu 22.04**. If `./rgctl` fails with `GLIBC_… not found`, build from source on that host (see Option B), or track a future musl/manylinux asset. + 3. Extract the archive: ```bash @@ -69,6 +71,8 @@ Expand-Archive rgctl-*-x86_64-pc-windows-msvc.zip -DestinationPath . ### Option B -- Build from source +Requires **Rust 1.88+** (workspace `rust-version`; edition 2024). Check with `rustc --version`. + ```bash git clone https://github.com/sshaaf/rgctl.git cd rgctl @@ -78,6 +82,12 @@ cargo build --release --bin rgctl All **Tier 1** languages registered in [`languages.toml`](languages.toml) (Rust, Python, Ruby, PHP, JavaScript, TypeScript, Go, Java, C#, C, C++) plus markdown are always included in the binary — no per-language feature flags. +**Default features** include `semantic-onnx` (links `ort` / ONNX Runtime for the optional `code-daemon` embedder). If the build fails linking `ort-sys` or you do not need ONNX, build without default features (vocab embedder still works): + +```bash +cargo build --release --bin rgctl --no-default-features +``` + **Optional ONNX weights** (only for `--embedder code-daemon`): ```bash @@ -362,9 +372,11 @@ Start with the default mode (no extra flags). Add `--with-cfg`, `--with-taint`, ### Build from source fails -- Confirm Rust 1.88+: `rustc --version` +- Confirm **Rust 1.88+** (workspace `rust-version`): `rustc --version` - Update Rust: `rustup update` - Clean build: `cargo clean && cargo build --release --bin rgctl` +- ONNX / `ort-sys` link errors: `cargo build --release --bin rgctl --no-default-features` (disables default `semantic-onnx`; default `vocab` semantic search still works) +- Linux binary from GitHub Releases fails with missing `GLIBC_*`: the gnu release needs a newer glibc than Ubuntu 22.04; build from source on the target host (musl/manylinux assets are a tracked follow-up) --- diff --git a/docs/releases/unreleased.md b/docs/releases/unreleased.md index b09d03ea..3bfc4fd4 100644 --- a/docs/releases/unreleased.md +++ b/docs/releases/unreleased.md @@ -1,3 +1,15 @@ # Unreleased (post v0.4.16) + +## Fixes + +- **TypeScript / JavaScript named arrows** — `const` / `let` / object-property / class-field arrows and function expressions are named from their binding (no longer collapsed to a single `anonymous` per file). Truly unnamed callbacks use `anonymous@L{line}`. **Rediscover** TS/JS repos after upgrade so Function identities and call edges refresh. +- **CFG skip diagnostics** — discover `--with-cfg` summarizes skips by `unsupported_language` / `missing_source` / `analysis_error`; `-v` / profile logs each skip (path, symbol, reason). +- **Community flat-graph UX** — warn (soften `[✓] Detected N communities`) when Calls/Uses are sparse or communities ≥ ~90% of nodes. +- **Release checksums** — workflow refuses an empty `SHA256SUMS.txt` and flattens nested artifact dirs. +- **Discover `--exclude`** — repeatable `-e a -e b` while still accepting comma-separated values. + +## Docs + +- Installation: glibc / Ubuntu 22.04 caveat for gnu Linux releases, Rust **1.88+**, `--no-default-features` when ONNX/`ort` fails to link. Musl/manylinux prebuilt asset tracked in #97. diff --git a/docs/user-guide.md b/docs/user-guide.md index 67c96aff..bf301940 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -832,10 +832,12 @@ rgctl -r "$REPO" gql \ ## 9. Inspect CFG / PDG / dominance -`inspect` dumps semantic layers for an **indexed function symbol** (no `--class` flag — use a unique symbol or GQL to pick the right function). Run `discover --with-cfg` first. +`inspect` dumps semantic layers for an **indexed function symbol**. Prefer a unique name, or the same +`path/to/file.ts::symbol` / `Class::method` form as `blast-radius`. Run `discover --with-cfg` first. ```bash rgctl -r "$REPO" inspect checkout cfg +rgctl -r "$REPO" inspect "fixtures/caseB_one.ts::dup" cfg ``` ```text diff --git a/rgctl-tests/ecommerce-javascript/correctness/expected-facts.json b/rgctl-tests/ecommerce-javascript/correctness/expected-facts.json index 72b01b55..ee6fe8ed 100644 --- a/rgctl-tests/ecommerce-javascript/correctness/expected-facts.json +++ b/rgctl-tests/ecommerce-javascript/correctness/expected-facts.json @@ -282,6 +282,80 @@ "min_blocks": 6, "statements_contain": [] } + }, + "caseGamma": { + "match": { + "name": "gamma" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "gamma", + "file_suffix": "caseC2.js" + }, + "cfg": { + "exact_block_count": 2, + "exact_edge_count": 1 + } + }, + "caseUniqueOne": { + "match": { + "name": "uniqueOne" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "uniqueOne", + "file_suffix": "caseB_one.js" + }, + "exact_callees": [ + "dup" + ], + "cfg": { + "min_blocks": 2 + } + }, + "caseUniqueDecl": { + "match": { + "name": "uniqueDecl" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "uniqueDecl", + "file_suffix": "caseB_one.js" + }, + "cfg": { + "min_blocks": 2 + } + }, + "caseDupOne": { + "match": { + "name": "dup" + }, + "severity": "required", + "unique": false, + "identity": { + "exact_name": "dup", + "file_suffix": "caseB_one.js" + }, + "cfg": { + "min_blocks": 2 + } + }, + "caseArrowAdd": { + "match": { + "name": "arrowAdd" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "arrowAdd", + "file_suffix": "namedArrows.js" + }, + "cfg": { + "min_blocks": 2 + } } }, "invariants": { @@ -303,7 +377,10 @@ "symbols": [ "correctnessRoot", "correctnessMid", - "correctnessLeaf" + "correctnessLeaf", + "caseUniqueOne", + "caseGamma", + "caseArrowAdd" ] }, "B6_cfg_calls_subset_of_calls_edges": { diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseA.js b/rgctl-tests/ecommerce-javascript/fixtures/caseA.js new file mode 100644 index 00000000..1798a2cc --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseA.js @@ -0,0 +1,4 @@ +/** Tester case A — one declaration; expect only `alpha`. */ +export function alpha(n) { + return n + 1; +} diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseA2.js b/rgctl-tests/ecommerce-javascript/fixtures/caseA2.js new file mode 100644 index 00000000..479ffa81 --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseA2.js @@ -0,0 +1,8 @@ +/** Tester case A2 — two declarations; expect only `alpha2` and `beta`. */ +export function alpha2(n) { + return n + 1; +} + +function beta(n) { + return alpha2(n); +} diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseB_one.js b/rgctl-tests/ecommerce-javascript/fixtures/caseB_one.js new file mode 100644 index 00000000..64cb4df2 --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseB_one.js @@ -0,0 +1,10 @@ +/** Tester case B (file one). */ +export const dup = (n) => n + 1; +export const uniqueOne = (n) => dup(n); + +export function dupDecl(n) { + return n + 1; +} +export function uniqueDecl(n) { + return dupDecl(n); +} diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseB_two.js b/rgctl-tests/ecommerce-javascript/fixtures/caseB_two.js new file mode 100644 index 00000000..3792e987 --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseB_two.js @@ -0,0 +1,10 @@ +/** Tester case B (file two). */ +export const dup = (n) => n + 2; +export const uniqueTwo = (n) => dup(n); + +export function dupDecl2(n) { + return n + 2; +} +export function uniqueDecl2(n) { + return dupDecl2(n); +} diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseC1.js b/rgctl-tests/ecommerce-javascript/fixtures/caseC1.js new file mode 100644 index 00000000..5e08aaac --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseC1.js @@ -0,0 +1,3 @@ +/** Tester case C1 — no functions. */ +export const X = 5; +export const Y = "hello"; diff --git a/rgctl-tests/ecommerce-javascript/fixtures/caseC2.js b/rgctl-tests/ecommerce-javascript/fixtures/caseC2.js new file mode 100644 index 00000000..717fd5be --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/caseC2.js @@ -0,0 +1,9 @@ +/** Tester case C2 — one declaration; expect only `gamma`. */ +// comment +// comment + +const VALUE = 1; + +export function gamma(n) { + return n + VALUE; +} diff --git a/rgctl-tests/ecommerce-javascript/fixtures/namedArrows.js b/rgctl-tests/ecommerce-javascript/fixtures/namedArrows.js new file mode 100644 index 00000000..a9eded24 --- /dev/null +++ b/rgctl-tests/ecommerce-javascript/fixtures/namedArrows.js @@ -0,0 +1,21 @@ +/** Named-arrow fixture for extraction / blast-radius gates. */ + +function declaredAdd(a, b) { + return a + b; +} + +const arrowAdd = (a, b) => a + b; + +const arrowHelper = () => 1; + +const api = { + fetchAll: async () => 0, +}; + +function callArrows() { + return arrowAdd(1, 2) + arrowHelper(); +} + +[1].map((x) => x + 1); + +module.exports = { declaredAdd, arrowAdd, arrowHelper, api, callArrows }; diff --git a/rgctl-tests/ecommerce-typescript/correctness/expected-facts.json b/rgctl-tests/ecommerce-typescript/correctness/expected-facts.json index d576f72f..5e9cb33c 100644 --- a/rgctl-tests/ecommerce-typescript/correctness/expected-facts.json +++ b/rgctl-tests/ecommerce-typescript/correctness/expected-facts.json @@ -282,6 +282,80 @@ "min_blocks": 6, "statements_contain": [] } + }, + "caseGamma": { + "match": { + "name": "gamma" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "gamma", + "file_suffix": "caseC2.ts" + }, + "cfg": { + "exact_block_count": 2, + "exact_edge_count": 1 + } + }, + "caseUniqueOne": { + "match": { + "name": "uniqueOne" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "uniqueOne", + "file_suffix": "caseB_one.ts" + }, + "exact_callees": [ + "dup" + ], + "cfg": { + "min_blocks": 2 + } + }, + "caseUniqueDecl": { + "match": { + "name": "uniqueDecl" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "uniqueDecl", + "file_suffix": "caseB_one.ts" + }, + "cfg": { + "min_blocks": 2 + } + }, + "caseDupOne": { + "match": { + "name": "dup" + }, + "severity": "required", + "unique": false, + "identity": { + "exact_name": "dup", + "file_suffix": "caseB_one.ts" + }, + "cfg": { + "min_blocks": 2 + } + }, + "caseArrowAdd": { + "match": { + "name": "arrowAdd" + }, + "severity": "required", + "unique": true, + "identity": { + "exact_name": "arrowAdd", + "file_suffix": "namedArrows.ts" + }, + "cfg": { + "min_blocks": 2 + } } }, "invariants": { @@ -303,7 +377,10 @@ "symbols": [ "correctnessRoot", "correctnessMid", - "correctnessLeaf" + "correctnessLeaf", + "caseUniqueOne", + "caseGamma", + "caseArrowAdd" ] }, "B6_cfg_calls_subset_of_calls_edges": { diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseA.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseA.ts new file mode 100644 index 00000000..a5b803e4 --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseA.ts @@ -0,0 +1,4 @@ +/** Tester case A — one declaration; expect only `alpha`. */ +export function alpha(n: number): number { + return n + 1; +} diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseA2.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseA2.ts new file mode 100644 index 00000000..ec103412 --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseA2.ts @@ -0,0 +1,8 @@ +/** Tester case A2 — two declarations; expect only `alpha2` and `beta` (no anonymous dupes). */ +export function alpha2(n: number): number { + return n + 1; +} + +function beta(n: number): number { + return alpha2(n); +} diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseB_one.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseB_one.ts new file mode 100644 index 00000000..5806ad1a --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseB_one.ts @@ -0,0 +1,10 @@ +/** Tester case B (file one) — bound arrows must have CFG; `dup` is intentionally ambiguous with caseB_two. */ +export const dup = (n: number): number => n + 1; +export const uniqueOne = (n: number): number => dup(n); + +export function dupDecl(n: number): number { + return n + 1; +} +export function uniqueDecl(n: number): number { + return dupDecl(n); +} diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseB_two.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseB_two.ts new file mode 100644 index 00000000..d9e25197 --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseB_two.ts @@ -0,0 +1,10 @@ +/** Tester case B (file two) — second `dup` for path::symbol disambiguation. */ +export const dup = (n: number): number => n + 2; +export const uniqueTwo = (n: number): number => dup(n); + +export function dupDecl2(n: number): number { + return n + 2; +} +export function uniqueDecl2(n: number): number { + return dupDecl2(n); +} diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseC1.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseC1.ts new file mode 100644 index 00000000..f7ac7e47 --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseC1.ts @@ -0,0 +1,3 @@ +/** Tester case C1 — no functions; expect 0 Function nodes from this file. */ +export const X = 5; +export const Y = "hello"; diff --git a/rgctl-tests/ecommerce-typescript/fixtures/caseC2.ts b/rgctl-tests/ecommerce-typescript/fixtures/caseC2.ts new file mode 100644 index 00000000..5b67ef02 --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/caseC2.ts @@ -0,0 +1,9 @@ +/** Tester case C2 — one declaration; expect only `gamma` (no anonymous@L duplicate). */ +// comment +// comment + +const VALUE = 1; + +export function gamma(n: number): number { + return n + VALUE; +} diff --git a/rgctl-tests/ecommerce-typescript/fixtures/namedArrows.ts b/rgctl-tests/ecommerce-typescript/fixtures/namedArrows.ts new file mode 100644 index 00000000..d371df6c --- /dev/null +++ b/rgctl-tests/ecommerce-typescript/fixtures/namedArrows.ts @@ -0,0 +1,19 @@ +/** Named-arrow fixture for extraction / blast-radius gates. */ + +export function declaredAdd(a: number, b: number): number { + return a + b; +} + +export const arrowAdd = (a: number, b: number): number => a + b; + +export const arrowHelper = (): number => 1; + +export const api = { + fetchAll: async (): Promise => 0, +}; + +export function callArrows(): number { + return arrowAdd(1, 2) + arrowHelper(); +} + +[1].map((x) => x + 1); diff --git a/src/cli/discover_cfg.rs b/src/cli/discover_cfg.rs index 5377db1f..c7a65320 100644 --- a/src/cli/discover_cfg.rs +++ b/src/cli/discover_cfg.rs @@ -30,6 +30,35 @@ pub struct CfgAnalysisOptions { pub dfg_loops: bool, } +/// Why a function was not CFG-analyzed. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum CfgSkipReason { + /// Extension / language has no CFG builder. + UnsupportedLanguage, + /// Source text was not available for the function's file. + MissingSource, + /// Parse or CFG/PDG construction failed (or empty body). + AnalysisError, +} + +impl CfgSkipReason { + pub fn as_str(self) -> &'static str { + match self { + Self::UnsupportedLanguage => "unsupported_language", + Self::MissingSource => "missing_source", + Self::AnalysisError => "analysis_error", + } + } +} + +/// One skipped function (populated when `verbose` is set). +#[derive(Debug, Clone)] +pub struct CfgSkipEvent { + pub file_path: String, + pub symbol: String, + pub reason: CfgSkipReason, +} + /// Wall-clock totals for CFG sub-stages (sum of per-function work). #[derive(Debug, Clone, Copy, Default)] pub struct CfgStageProfile { @@ -44,6 +73,10 @@ pub struct CfgStageProfile { pub struct CfgAnalysisBatchResult { pub success_count: usize, pub error_count: usize, + pub skip_unsupported_language: usize, + pub skip_missing_source: usize, + pub skip_analysis_error: usize, + pub skips: Vec, pub total_flows: usize, pub vulnerable_flows: usize, pub cache_hits: usize, @@ -55,6 +88,30 @@ pub struct CfgAnalysisBatchResult { pub stage_profile: Option, } +impl CfgAnalysisBatchResult { + fn record_skip( + &mut self, + reason: CfgSkipReason, + file_path: &str, + symbol: &str, + verbose: bool, + ) { + self.error_count += 1; + match reason { + CfgSkipReason::UnsupportedLanguage => self.skip_unsupported_language += 1, + CfgSkipReason::MissingSource => self.skip_missing_source += 1, + CfgSkipReason::AnalysisError => self.skip_analysis_error += 1, + } + if verbose { + self.skips.push(CfgSkipEvent { + file_path: file_path.to_string(), + symbol: symbol.to_string(), + reason, + }); + } + } +} + #[derive(Default)] struct CfgStageTimings { build_cfg_ns: AtomicU64, @@ -195,7 +252,7 @@ pub fn run_cfg_analysis_batch( dfg_loops: options.dfg_loops, }; - let nested: Vec>> = with_large_stack(|| { + let nested: Vec>> = with_large_stack(|| { with_large_pool(options.thread_count, || { groups .par_iter() @@ -203,10 +260,20 @@ pub fn run_cfg_analysis_batch( .collect() }) }); - let flat: Vec> = nested.into_iter().flatten().collect(); + + let mut in_group = vec![false; functions.len()]; + for group in &groups { + for &idx in &group.func_indices { + in_group[idx] = true; + } + } let stage_profile = stage_ref.map(|stage| { - let analyzed = flat.iter().filter_map(|w| w.as_ref()).count(); + let analyzed = nested + .iter() + .flatten() + .filter(|w| w.is_ok()) + .count(); emit_stage_profile(stage, analyzed) }); if let (Some(log), true) = (timing_log, options.verbose) { @@ -215,27 +282,51 @@ pub fn run_cfg_analysis_batch( let mut saves: Vec = Vec::new(); let mut result = CfgAnalysisBatchResult::default(); - for work in flat { - match work { - None => result.error_count += 1, - Some(w) => { - if w.from_cache { - result.cache_hits += 1; - } else { - result.recomputed += 1; - } - if w.skip_persist { - result.skipped_unchanged += 1; - } - result.success_count += 1; - result.total_flows += w.flow_count; - result.vulnerable_flows += w.vulnerable_count; - if let Some(record) = w.archive_record { - result.archive_records.push(record); + + for (idx, func) in functions.iter().enumerate() { + if in_group[idx] { + continue; + } + let path = func.file_path.as_deref().unwrap_or(""); + result.record_skip( + CfgSkipReason::UnsupportedLanguage, + path, + &func.name, + options.verbose, + ); + } + + for (group, outcomes) in groups.iter().zip(nested.into_iter()) { + for (&func_idx, outcome) in group.func_indices.iter().zip(outcomes.into_iter()) { + let func = &functions[func_idx]; + match outcome { + Err(reason) => { + result.record_skip( + reason, + &group.file_path, + &func.name, + options.verbose, + ); } - if let Some(analysis) = w.analysis { - if !w.skip_persist { - saves.push(analysis); + Ok(w) => { + if w.from_cache { + result.cache_hits += 1; + } else { + result.recomputed += 1; + } + if w.skip_persist { + result.skipped_unchanged += 1; + } + result.success_count += 1; + result.total_flows += w.flow_count; + result.vulnerable_flows += w.vulnerable_count; + if let Some(record) = w.archive_record { + result.archive_records.push(record); + } + if let Some(analysis) = w.analysis { + if !w.skip_persist { + saves.push(analysis); + } } } } @@ -357,9 +448,11 @@ fn group_work_items_by_file(functions: &[Node]) -> Vec { fn process_file_group( group: &FileWorkGroup, ctx: &CfgWorkContext<'_>, -) -> Vec> { +) -> Vec> { let Some(source) = ctx.sources.get(&group.file_path) else { - return (0..group.func_indices.len()).map(|_| None).collect(); + return (0..group.func_indices.len()) + .map(|_| Err(CfgSkipReason::MissingSource)) + .collect(); }; let mut parsed: Option = None; let mut parse_attempted = false; @@ -381,6 +474,7 @@ fn process_file_group( ctx.enable_taint, ctx.dfg_loops, ) + .ok_or(CfgSkipReason::AnalysisError) }) .collect() } diff --git a/src/cli/discover_impl.rs b/src/cli/discover_impl.rs index de218dd9..5cc4368e 100644 --- a/src/cli/discover_impl.rs +++ b/src/cli/discover_impl.rs @@ -350,11 +350,39 @@ pub(crate) fn run_full_analysis( analysis_results.fill_community(&community_result); profile.community.secs = secs(community_start.elapsed()); if human_output { - info!( - "[✓] Detected {} communities (modularity: {:.2})", - community_result.communities.len(), - community_result.modularity - ); + let n_communities = community_result.communities.len(); + let n_nodes = petgraph_view.node_count().max(1); + let mut call_use_edges = 0usize; + let _ = petgraph_view.for_each_edge(|_, _, ty| { + if matches!( + ty, + rgctl_graph::schema::EdgeType::Calls | rgctl_graph::schema::EdgeType::Uses + ) { + call_use_edges += 1; + } + }); + // Sparse Calls/Uses relative to nodes ≈ flat call graph (communities ≈ isolates). + let flat_call_graph = call_use_edges == 0 || call_use_edges * 10 < n_nodes; + let degenerate = n_communities * 10 >= n_nodes * 9; + if flat_call_graph || degenerate { + warn!( + communities = n_communities, + nodes = n_nodes, + call_use_edges, + flat_call_graph, + degenerate, + "[!] Community partition may be meaningless (flat call graph or near 1:1 clustering)" + ); + info!( + "[!] Communities: {} (modularity: {:.2}) — warn: flat or degenerate clustering", + n_communities, community_result.modularity + ); + } else { + info!( + "[✓] Detected {} communities (modularity: {:.2})", + n_communities, community_result.modularity + ); + } } debug!("{}", mem_monitor.report()); @@ -676,8 +704,11 @@ pub(crate) fn run_full_analysis( println!(" CFG/PDG/Dominance: {} functions analyzed", success_count); if error_count > 0 { println!( - " Skipped: {} functions (unsupported language or parse error)", - error_count + " Skipped: {} functions (unsupported_language={}, missing_source={}, analysis_error={})", + error_count, + batch.skip_unsupported_language, + batch.skip_missing_source, + batch.skip_analysis_error ); } @@ -692,6 +723,15 @@ pub(crate) fn run_full_analysis( " No functions analyzed (CFG supported: {})", cfg_language_list() ); + if error_count > 0 { + println!( + " Skipped: {} functions (unsupported_language={}, missing_source={}, analysis_error={})", + error_count, + batch.skip_unsupported_language, + batch.skip_missing_source, + batch.skip_analysis_error + ); + } } if verbose { if batch.cache_hits > 0 || batch.recomputed > 0 || batch.skipped_unchanged > 0 { @@ -706,6 +746,17 @@ pub(crate) fn run_full_analysis( println!("{}", mem_monitor.report()); } } + if verbose { + for skip in &batch.skips { + info!( + target: "profile", + path = %skip.file_path, + symbol = %skip.symbol, + reason = skip.reason.as_str(), + "[profile] cfg skip" + ); + } + } } // Blast radius analysis with SCC + Dense Bitsets engine diff --git a/src/cli/inspect.rs b/src/cli/inspect.rs index a229ecfc..cdac2cd1 100644 --- a/src/cli/inspect.rs +++ b/src/cli/inspect.rs @@ -6,6 +6,7 @@ use super::inspect_output::{inspect_cfg_json, inspect_dom_json, inspect_pdg_json use super::markup::markup_context_unsupported; use crate::analysis::{DominatorTree, ProgramDependenceGraph, build_cfg_for_function}; use anyhow::Result; +use rgctl_graph::backend::GraphBackend; use std::path::Path; pub struct InspectArgs { @@ -20,7 +21,8 @@ pub fn run(ctx: &CliContext, args: InspectArgs) -> Result<()> { anyhow::bail!(msg); } let lang = language_from_path(Path::new(file)); - let mut cfg = build_cfg_for_function(&lang, &source, &node.name)?; + let display_name = node.name.as_str(); + let mut cfg = build_cfg_for_function(&lang, &source, display_name)?; let pdg = ProgramDependenceGraph::build(&cfg, source.as_bytes())?; let dom = DominatorTree::build(&cfg); @@ -31,7 +33,7 @@ pub fn run(ctx: &CliContext, args: InspectArgs) -> Result<()> { } match ctx.format { OutputFormat::Json => { - let response = inspect_cfg_json(&args.symbol, &cfg, prune); + let response = inspect_cfg_json(display_name, &cfg, prune); ctx.emit_json_value(&serde_json::to_value(&response)?)?; } OutputFormat::Mermaid => { @@ -43,7 +45,7 @@ pub fn run(ctx: &CliContext, args: InspectArgs) -> Result<()> { OutputFormat::Text => { println!( "CFG for {}: {} blocks, {} edges", - args.symbol, + display_name, cfg.blocks.len(), cfg.edges.len() ); @@ -60,12 +62,12 @@ pub fn run(ctx: &CliContext, args: InspectArgs) -> Result<()> { PdgEdgeLayer::Control => (0, pdg.control_deps.len()), }; if ctx.format == OutputFormat::Json { - let response = inspect_pdg_json(&args.symbol, &pdg, def_use, data, control); + let response = inspect_pdg_json(display_name, &pdg, def_use, data, control); ctx.emit_json_value(&serde_json::to_value(&response)?)?; } else { println!( "PDG for {}: {} nodes, {} data deps, {} control deps", - args.symbol, + display_name, pdg.nodes.len(), data, control @@ -74,12 +76,12 @@ pub fn run(ctx: &CliContext, args: InspectArgs) -> Result<()> { } InspectLayer::Dom { frontiers } => { if ctx.format == OutputFormat::Json { - let response = inspect_dom_json(&args.symbol, &cfg, &dom, frontiers); + let response = inspect_dom_json(display_name, &cfg, &dom, frontiers); ctx.emit_json_value(&serde_json::to_value(&response)?)?; } else if ctx.format == OutputFormat::Mermaid { ctx.emit(&dom_to_mermaid(&dom))?; } else { - println!("Dominators for {}: {} blocks", args.symbol, dom.idom.len()); + println!("Dominators for {}: {} blocks", display_name, dom.idom.len()); if frontiers { for (block, frontier) in &dom.frontiers { if !frontier.is_empty() { @@ -122,12 +124,42 @@ fn resolve_symbol_function( ctx: &CliContext, symbol: &str, ) -> Result<(rgctl_graph::schema::Node, String)> { + use rgctl_analysis::{candidates_from_backend, parse_fqn_symbol, resolve_symbol_uuid}; use rgctl_graph::schema::NodeType; use std::fs; + let parsed = parse_fqn_symbol(symbol, None, None); let graph = ctx.load_graph()?; let backend = graph.backend(); - let matches = backend.find_nodes_by_name(symbol)?; + + // Prefer shared FQN resolution (path::symbol / Class::method) used by blast-radius. + let candidates = candidates_from_backend(backend, &parsed.target_name)?; + if !candidates.is_empty() { + match resolve_symbol_uuid(&candidates, &parsed) { + Ok(id) => { + let node = backend + .get_node(id)? + .ok_or_else(|| anyhow::anyhow!("function symbol not found: {symbol}"))?; + let file = node + .file_path + .clone() + .ok_or_else(|| anyhow::anyhow!("function has no file path"))?; + let source = fs::read_to_string(&file)?; + return Ok((node, source)); + } + Err(rgctl_error::Error::AmbiguousSymbol { name, count }) => { + anyhow::bail!( + "Symbol '{name}' is ambiguous. Found {count} matches. \ + Refine with path syntax: rgctl inspect \"path/to/file.ts::{name}\" cfg" + ); + } + Err(rgctl_error::Error::NotFound(_)) => {} + Err(e) => return Err(e.into()), + } + } + + // Legacy fallback: bare name or suffix match when FQN filters yield nothing. + let matches = backend.find_nodes_by_name(&parsed.target_name)?; let node = matches .into_iter() .find(|n| n.node_type == NodeType::Function) @@ -136,7 +168,11 @@ fn resolve_symbol_function( .all_nodes() .ok()? .into_iter() - .find(|n| n.name == symbol || n.name.ends_with(symbol)) + .find(|n| { + n.node_type == NodeType::Function + && (n.name == parsed.target_name + || n.name.ends_with(&parsed.target_name)) + }) }) .ok_or_else(|| anyhow::anyhow!("function symbol not found: {symbol}"))?; let file = node diff --git a/src/cli/mod.rs b/src/cli/mod.rs index 985cba9d..78918201 100644 --- a/src/cli/mod.rs +++ b/src/cli/mod.rs @@ -46,10 +46,25 @@ use crate::analysis::{DEFAULT_CANDIDATE_POOL, DEFAULT_EMBEDDING_DIMENSIONS}; use args::{ ExportFormat, InspectLayer, PdgEdgeLayer, SkillHost, SliceDirection, SliceView, }; -use clap::{Parser, Subcommand}; +use clap::{ArgAction, Parser, Subcommand}; use context::CliContext; use std::time::{Duration, Instant}; +/// Merge repeated `-e` / `--exclude` values and comma-separated lists into one CSV. +fn join_exclude_patterns(patterns: &[String]) -> Option { + let parts: Vec<&str> = patterns + .iter() + .flat_map(|s| s.split(',')) + .map(str::trim) + .filter(|s| !s.is_empty()) + .collect(); + if parts.is_empty() { + None + } else { + Some(parts.join(",")) + } +} + #[derive(Parser)] #[command(name = "rgctl")] #[command(about = "A code knowledge graph built for LLM agents", version = BUILD_INFO)] @@ -85,8 +100,9 @@ pub enum Commands { #[arg(short = 'l', long = "languages")] languages: Option, - #[arg(short = 'e', long = "exclude")] - exclude: Option, + /// Path exclude glob (repeatable; comma-separated values also accepted) + #[arg(short = 'e', long = "exclude", action = ArgAction::Append)] + exclude: Vec, #[arg(short = 'v', long = "verbose")] verbose: bool, @@ -782,7 +798,7 @@ impl Cli { discover::DiscoverArgs { path, languages, - exclude, + exclude: join_exclude_patterns(&exclude), with_security, with_cfg, with_taint, diff --git a/tests/graph_correctness_lib.rs b/tests/graph_correctness_lib.rs index 020db26b..ba6f07c8 100644 --- a/tests/graph_correctness_lib.rs +++ b/tests/graph_correctness_lib.rs @@ -133,6 +133,9 @@ struct MatchSpec { name: String, #[serde(default)] class: Option, + /// Optional path / file suffix filter (`path/to/file.ts::name` for inspect/blast). + #[serde(default)] + file: Option, } #[derive(Debug, Default, Deserialize)] @@ -373,8 +376,14 @@ fn run_json(bin: &Path, cwd: &Path, args: &[&str]) -> Option { fn symbol_candidates(m: &MatchSpec) -> Vec { let mut out = Vec::new(); + if let Some(file) = &m.file { + out.push(format!("{file}::{}", m.name)); + } if let Some(cls) = &m.class { - out.push(format!("{}::{}", cls, m.name)); + // File-like class hints (Go) are handled via --file in run_blast; skip Class:: here. + if !(cls.ends_with(".go") || cls.contains('/')) { + out.push(format!("{cls}::{}", m.name)); + } } out.push(m.name.clone()); out @@ -393,6 +402,19 @@ fn run_blast(bin: &Path, cwd: &Path, m: &MatchSpec) -> Option { } } } + if let Some(file) = &m.file { + if let Some(v) = run_json( + bin, + cwd, + &["-f", "json", "blast-radius", &m.name, "--file", file], + ) { + return Some(v); + } + let fqn = format!("{file}::{}", m.name); + if let Some(v) = run_json(bin, cwd, &["-f", "json", "blast-radius", &fqn]) { + return Some(v); + } + } for sym in symbol_candidates(m) { if let Some(v) = run_json(bin, cwd, &["-f", "json", "blast-radius", &sym]) { return Some(v); @@ -619,7 +641,12 @@ fn check_symbol( if sev == "unsupported" { return vec![]; } - let m = entry.match_spec(); + let mut m = entry.match_spec(); + if m.file.is_none() { + if let Some(ident) = &entry.identity { + m.file = ident.file_suffix.clone(); + } + } if m.name.is_empty() { return vec![check( format!("symbol.{sid}"), @@ -1321,7 +1348,12 @@ fn check_invariants( let Some(entry) = facts.symbols.get(sid) else { continue; }; - let m = entry.match_spec(); + let mut m = entry.match_spec(); + if m.file.is_none() { + if let Some(ident) = &entry.identity { + m.file = ident.file_suffix.clone(); + } + } let cfg = run_inspect(bin, cwd, &m, "cfg"); let ok = cfg .as_ref() diff --git a/tests/javascript_langfeatures.rs b/tests/javascript_langfeatures.rs index 71cad054..07c64480 100644 --- a/tests/javascript_langfeatures.rs +++ b/tests/javascript_langfeatures.rs @@ -84,3 +84,104 @@ fn javascript_ecommerce_extends_nonzero() { let n = edge_count(&repo(), "EXTENDS"); assert!(n > 0, "expected Extends edges on ecommerce-javascript, got {n}"); } + +#[test] +fn javascript_named_arrows_present() { + ensure_discovered(); + let names = all_function_names(&repo()); + for expected in ["arrowAdd", "arrowHelper", "declaredAdd", "fetchAll"] { + assert!( + names.iter().any(|n| n == expected), + "missing Function `{expected}` in {names:?}" + ); + } +} + +#[test] +fn javascript_case_c1_no_functions() { + ensure_discovered(); + let names = functions_in_file(&repo(), "caseC1.js"); + assert!( + names.is_empty(), + "caseC1 (const-only) must emit 0 Function nodes, got {names:?}" + ); +} + +#[test] +fn javascript_case_declarations_have_no_anonymous_duplicates() { + ensure_discovered(); + assert_eq!( + functions_in_file(&repo(), "caseC2.js"), + vec!["gamma".to_string()] + ); + assert_eq!( + functions_in_file(&repo(), "caseA.js"), + vec!["alpha".to_string()] + ); + let a2 = functions_in_file(&repo(), "caseA2.js"); + assert_eq!(a2.len(), 2, "{a2:?}"); + assert!(a2.contains(&"alpha2".to_string()) && a2.contains(&"beta".to_string()), "{a2:?}"); + assert!(a2.iter().all(|n| !n.starts_with("anonymous")), "{a2:?}"); +} + +#[test] +fn javascript_case_b_named_arrows_and_no_decl_duplicates() { + ensure_discovered(); + let one = functions_in_file(&repo(), "caseB_one.js"); + for expected in ["dup", "uniqueOne", "dupDecl", "uniqueDecl"] { + assert!(one.iter().any(|n| n == expected), "missing {expected} in {one:?}"); + } + assert_eq!(one.len(), 4, "{one:?}"); + assert_eq!(functions_in_file(&repo(), "caseB_two.js").len(), 4); +} + +fn all_function_names(repo: &Path) -> Vec { + function_rows(repo) + .into_iter() + .map(|(_, name)| name) + .collect() +} + +fn functions_in_file(repo: &Path, file_suffix: &str) -> Vec { + let mut names: Vec = function_rows(repo) + .into_iter() + .filter(|(file, _)| file.replace('\\', "/").ends_with(file_suffix)) + .map(|(_, name)| name) + .collect(); + names.sort(); + names +} + +fn function_rows(repo: &Path) -> Vec<(String, String)> { + let v = gql(repo, "MATCH (n:Function) RETURN n LIMIT 10000"); + let mut out = Vec::new(); + if let Some(rows) = v.get("rows").and_then(|r| r.as_array()) { + for row in rows { + let cells = row.as_array().map(|a| a.as_slice()).unwrap_or(std::slice::from_ref(row)); + for cell in cells { + let name = cell + .get("node") + .and_then(|n| n.as_str()) + .or_else(|| cell.get("name").and_then(|n| n.as_str())) + .or_else(|| { + cell.get("n") + .and_then(|n| n.get("name")) + .and_then(|n| n.as_str()) + }); + let file = cell + .get("file") + .and_then(|f| f.as_str()) + .or_else(|| { + cell.get("n") + .and_then(|n| n.get("file")) + .and_then(|f| f.as_str()) + }) + .unwrap_or(""); + if let Some(name) = name { + out.push((file.to_string(), name.to_string())); + } + } + } + } + out +} diff --git a/tests/typescript_langfeatures.rs b/tests/typescript_langfeatures.rs index ed8b457c..32e3e69b 100644 --- a/tests/typescript_langfeatures.rs +++ b/tests/typescript_langfeatures.rs @@ -98,3 +98,110 @@ fn typescript_ecommerce_extends_nonzero() { let n = edge_count(&repo(), "EXTENDS"); assert!(n > 0, "expected Extends edges on ecommerce-typescript, got {n}"); } + +#[test] +fn typescript_named_arrows_present() { + ensure_discovered(); + let names = all_function_names(&repo()); + for expected in ["arrowAdd", "arrowHelper", "declaredAdd", "fetchAll"] { + assert!( + names.iter().any(|n| n == expected), + "missing Function `{expected}` in {names:?}" + ); + } +} + +#[test] +fn typescript_case_c1_no_functions() { + ensure_discovered(); + let names = functions_in_file(&repo(), "caseC1.ts"); + assert!( + names.is_empty(), + "caseC1 (const-only) must emit 0 Function nodes, got {names:?}" + ); +} + +#[test] +fn typescript_case_declarations_have_no_anonymous_duplicates() { + ensure_discovered(); + assert_eq!( + functions_in_file(&repo(), "caseC2.ts"), + vec!["gamma".to_string()], + "caseC2 must be only gamma" + ); + assert_eq!( + functions_in_file(&repo(), "caseA.ts"), + vec!["alpha".to_string()], + "caseA must be only alpha" + ); + let a2 = functions_in_file(&repo(), "caseA2.ts"); + assert_eq!(a2.len(), 2, "caseA2 must have exactly 2 functions: {a2:?}"); + assert!(a2.contains(&"alpha2".to_string()) && a2.contains(&"beta".to_string()), "{a2:?}"); + assert!( + a2.iter().all(|n| !n.starts_with("anonymous")), + "caseA2 must not emit anonymous duplicates: {a2:?}" + ); +} + +#[test] +fn typescript_case_b_named_arrows_and_no_decl_duplicates() { + ensure_discovered(); + let one = functions_in_file(&repo(), "caseB_one.ts"); + for expected in ["dup", "uniqueOne", "dupDecl", "uniqueDecl"] { + assert!(one.iter().any(|n| n == expected), "missing {expected} in {one:?}"); + } + assert_eq!(one.len(), 4, "caseB_one must not add anonymous keyword dupes: {one:?}"); + let two = functions_in_file(&repo(), "caseB_two.ts"); + assert_eq!(two.len(), 4, "caseB_two must not add anonymous keyword dupes: {two:?}"); +} + +fn all_function_names(repo: &Path) -> Vec { + function_rows(repo) + .into_iter() + .map(|(_, name)| name) + .collect() +} + +fn functions_in_file(repo: &Path, file_suffix: &str) -> Vec { + let mut names: Vec = function_rows(repo) + .into_iter() + .filter(|(file, _)| file.replace('\\', "/").ends_with(file_suffix)) + .map(|(_, name)| name) + .collect(); + names.sort(); + names +} + +fn function_rows(repo: &Path) -> Vec<(String, String)> { + let v = gql(repo, "MATCH (n:Function) RETURN n LIMIT 10000"); + let mut out = Vec::new(); + if let Some(rows) = v.get("rows").and_then(|r| r.as_array()) { + for row in rows { + let cells = row.as_array().map(|a| a.as_slice()).unwrap_or(std::slice::from_ref(row)); + for cell in cells { + let name = cell + .get("node") + .and_then(|n| n.as_str()) + .or_else(|| cell.get("name").and_then(|n| n.as_str())) + .or_else(|| { + cell.get("n") + .and_then(|n| n.get("name")) + .and_then(|n| n.as_str()) + }); + let file = cell + .get("file") + .and_then(|f| f.as_str()) + .or_else(|| { + cell.get("n") + .and_then(|n| n.get("file")) + .and_then(|f| f.as_str()) + }) + .unwrap_or(""); + if let Some(name) = name { + out.push((file.to_string(), name.to_string())); + } + } + } + } + out +}