diff --git a/crates/codegraph-cli/tests/explore_completeness_note.rs b/crates/codegraph-cli/tests/explore_completeness_note.rs new file mode 100644 index 0000000..4b27991 --- /dev/null +++ b/crates/codegraph-cli/tests/explore_completeness_note.rs @@ -0,0 +1,178 @@ +//! The completeness note at the end of a `codegraph explore` response claims +//! "complete" only for sections that are (upstream v1.6.1 #2077). +//! +//! On the tiers with a completeness signal (>= 500 indexed files) every +//! response ended with "Complete source for N files is included above — do NOT +//! re-read them", whatever the render had cut, and the same line said "Reserve +//! Read for a single specific line range", which explore output must never say. +//! The fixture has a function too long for any budget, so its section is +//! windowed, beside a small flow that renders whole. + +use std::path::{Path, PathBuf}; +use std::process::Command; + +fn bin() -> PathBuf { + PathBuf::from(env!("CARGO_BIN_EXE_codegraph")) +} + +struct TestDir { + path: PathBuf, +} + +impl TestDir { + fn new(label: &str) -> Self { + let path = std::env::temp_dir().join(format!( + "codegraph-cli-complete-{label}-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir_all(&path).unwrap(); + Self { path } + } +} + +impl Drop for TestDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} + +fn run_in(cwd: &Path, args: &[&str]) -> String { + let output = Command::new(bin()) + .args(args) + .current_dir(cwd) + .env("CODEGRAPH_NO_DAEMON", "1") + .output() + .expect("run codegraph binary"); + assert!( + output.status.success(), + "{args:?} failed: {}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +/// Anything that offers Read as a way forward. "treat it as already Read" is +/// the guarantee and "do NOT Read" a prohibition; neither offers it. +fn offers_read(text: &str) -> bool { + ["Reserve Read", "use Read", "Read for ", "fall back to Read"] + .iter() + .any(|offer| text.contains(offer)) +} + +/// What follows the last source fence: the epilogue. +fn epilogue_of(text: &str) -> &str { + text.rfind("```").map_or(text, |i| &text[i + 3..]) +} + +/// A project whose `runPipeline` is far past any tier's budget, so its section +/// is always windowed, beside `formatValue → padValue`, which renders whole. +/// `padding` small files lift it into the large tiers. +fn project(label: &str, padding: usize) -> (TestDir, PathBuf) { + let dir = TestDir::new(label); + let root = dir.path.join("app"); + let src = root.join("src"); + std::fs::create_dir_all(&src).unwrap(); + let mut body = vec![ + "import { stepNext } from './step';".to_string(), + String::new(), + "export function runPipeline(input: number): number {".to_string(), + " let acc = input;".to_string(), + ]; + for i in 0..900 { + if i == 450 { + body.push(" acc = stepNext(acc);".to_string()); + } + body.push(format!(" acc = acc + {i}; // pipeline stage {i}")); + } + body.extend([ + " const PIPELINE_TAIL_MARKER = acc;".to_string(), + " return PIPELINE_TAIL_MARKER;".to_string(), + "}".to_string(), + String::new(), + ]); + std::fs::write(src.join("pipeline.ts"), body.join("\n")).unwrap(); + std::fs::write( + src.join("step.ts"), + "import { finalizeStep } from './finalize';\n\nexport function stepNext(v: number): number {\n return finalizeStep(v * 2);\n}\n", + ) + .unwrap(); + std::fs::write( + src.join("finalize.ts"), + "export function finalizeStep(v: number): number {\n return v + 1;\n}\n", + ) + .unwrap(); + std::fs::write( + src.join("format.ts"), + "import { padValue } from './pad';\n\nexport function formatValue(v: number): string {\n return padValue(String(v));\n}\n", + ) + .unwrap(); + std::fs::write( + src.join("pad.ts"), + "export function padValue(s: string): string {\n return s.padStart(8, ' ');\n}\n", + ) + .unwrap(); + if padding > 0 { + let pad = root.join("pad"); + std::fs::create_dir_all(&pad).unwrap(); + for i in 0..padding { + std::fs::write( + pad.join(format!("pad{i}.ts")), + format!("export const pad{i} = {i};\n"), + ) + .unwrap(); + } + } + run_in(&dir.path, &["init", root.to_str().unwrap()]); + (dir, root) +} + +#[test] +fn large_tier_a_windowed_section_is_reported_trimmed_not_complete() { + let (_dir, root) = project("large-trim", 520); + let text = run_in(&root, &["explore", "runPipeline stepNext finalizeStep"]); + // The fixture does what it is for: the function is windowed. + assert!(text.contains("#### src/pipeline.ts "), "{text}"); + assert!(!text.contains("PIPELINE_TAIL_MARKER"), "{text}"); + + assert!(!text.contains("Complete source for"), "{text}"); + assert!(text.contains("Verbatim source for"), "{text}"); + assert!(text.contains("treat it as already Read"), "{text}"); + assert!( + text.contains("Trimmed for size: `pipeline.ts`") + || text.contains("Some sections were trimmed for size"), + "{text}" + ); + assert!(!offers_read(epilogue_of(&text)), "{}", epilogue_of(&text)); +} + +#[test] +fn large_tier_complete_sections_are_still_called_complete_without_a_read_escape() { + let (_dir, root) = project("large-complete", 520); + let text = run_in(&root, &["explore", "formatValue padValue"]); + assert!(text.contains("#### src/format.ts "), "{text}"); + assert!(text.contains("#### src/pad.ts "), "{text}"); + assert!( + text.contains("Complete source for 2 files is included above"), + "{text}" + ); + assert!(!text.contains("Verbatim source for"), "{text}"); + assert!(!offers_read(epilogue_of(&text)), "{}", epilogue_of(&text)); +} + +#[test] +fn small_tier_the_same_cut_gets_the_trimmed_note_and_a_complete_answer_none() { + let (_dir, root) = project("small", 0); + let trimmed = run_in(&root, &["explore", "runPipeline stepNext finalizeStep"]); + assert!(!trimmed.contains("PIPELINE_TAIL_MARKER"), "{trimmed}"); + assert!( + trimmed.contains("Some file sections were trimmed for size"), + "{trimmed}" + ); + let complete = run_in(&root, &["explore", "formatValue padValue"]); + assert!(!complete.contains("trimmed for size"), "{complete}"); +} diff --git a/crates/codegraph-cli/tests/explore_exact_targets.rs b/crates/codegraph-cli/tests/explore_exact_targets.rs new file mode 100644 index 0000000..9f94db2 --- /dev/null +++ b/crates/codegraph-cli/tests/explore_exact_targets.rs @@ -0,0 +1,351 @@ +//! EXACT targets in `codegraph explore` (upstream v1.6.1 #2063): a qualified +//! name (`SQLCompiler.as_sql`) or a line anchor (`compiler.py:776`, +//! `compiler.py lines 900-1003`). +//! +//! Upstream's django gap: `SQLCompiler.as_sql pre_sql_setup get_select` +//! returned the two unqualified methods and not `as_sql` — qualified, the one +//! the agent singled out — and the follow-ups that named its lines pinned the +//! file but dropped the numbers, so every run ended in a Read. The fixture is +//! upstream's in miniature (Python, as django), sized to this port's per-file +//! budget: `get_select` and `get_qualify_sql` above a larger `as_sql`, three +//! subclasses overriding `as_sql`, and a second file whose base method is too +//! big for any budget. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +fn bin() -> PathBuf { + PathBuf::from(env!("CARGO_BIN_EXE_codegraph")) +} + +struct TestDir { + path: PathBuf, +} + +impl TestDir { + fn new(label: &str) -> Self { + let path = std::env::temp_dir().join(format!( + "codegraph-cli-exact-{label}-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir_all(&path).unwrap(); + Self { path } + } +} + +impl Drop for TestDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} + +fn run_in(cwd: &Path, args: &[&str]) -> String { + let output = Command::new(bin()) + .args(args) + .current_dir(cwd) + .env("CODEGRAPH_NO_DAEMON", "1") + .output() + .expect("run codegraph binary"); + assert!( + output.status.success(), + "{args:?} failed: {}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +/// `n` filler statements, each a distinct line so nothing dedups or folds. +fn filler(tag: &str, n: usize) -> String { + (0..n) + .map(|i| { + format!( + " {tag}_{i} = self.query.alias_refcount.get(\"{tag}_{i}\", 0) + len(self.query.select)" + ) + }) + .collect::>() + .join("\n") +} + +/// Unrelated methods, so compiler.py is too big to ship whole: a whole file +/// would answer every question by accident and hide the gap. +fn helpers(n: usize) -> String { + (0..n) + .map(|i| { + format!( + "\n def helper_{i}(self, value):\n \"\"\"Unrelated helper {i}.\"\"\"\n first = self.query.alias_map.get(value)\n second = self.query.alias_refcount.get(value, 0)\n third = self.query.external_aliases.get(value, False)\n return first, second, third" + ) + }) + .collect::>() + .join("\n") +} + +fn compiler_source() -> String { + format!( + r#"class SQLCompiler: + def pre_sql_setup(self, with_col_aliases=False): + # PRE_SQL_SETUP_BODY + self.setup_query(with_col_aliases=with_col_aliases) + order_by = self.get_order_by() + return order_by + + def setup_query(self, with_col_aliases=False): + self.select = self.get_select(with_col_aliases=with_col_aliases) + return self.select + + def get_order_by(self): + return [] + + def get_select(self, with_col_aliases=False): + # GET_SELECT_BODY +{sel} + return [] + + def get_qualify_sql(self): + # GET_QUALIFY_BODY +{qual} + inner = self.get_select() + return inner + + def as_sql(self, with_limits=True, with_col_aliases=False): + # AS_SQL_HEAD + order_by = self.pre_sql_setup(with_col_aliases=with_col_aliases) +{head} + result = self.get_qualify_sql() +{tail} + return result # AS_SQL_TAIL_MARKER + + def execute_sql(self): + sql = self.as_sql() + return sql + + +{helpers} + +def render_sql(compiler): + return SQLCompiler.as_sql(compiler) + + +class SQLInsertCompiler(SQLCompiler): + def as_sql(self): + # INSERT_AS_SQL_BODY + return super().as_sql() + + def execute_sql(self): + sql = self.as_sql() + return sql + + +class SQLUpdateCompiler(SQLCompiler): + def as_sql(self): + # UPDATE_AS_SQL_BODY + return super().as_sql() + + def pre_sql_setup(self): + # UPDATE_PRE_SQL_SETUP_BODY + return super().pre_sql_setup() + + +class SQLDeleteCompiler(SQLCompiler): + def as_sql(self): + # DELETE_AS_SQL_BODY + return super().as_sql() +"#, + sel = filler("sel", 20), + qual = filler("qual", 20), + head = filler("head", 20), + tail = filler("tail", 21), + helpers = helpers(40), + ) +} + +/// A base method too big for any budget, whose call into `stage_query` sits +/// 300 lines into the body: far from the head and from every other definition, +/// so only a window on that call reaches it. +fn planner_source() -> String { + format!( + r#"class Planner: + def plan(self, query): + # PLAN_HEAD +{pre} + staged = self.stage_query(query) # PLAN_CALLS_STAGE +{post} + return staged + + def stage_query(self, query): + return self.finalize(query) + + def finalize(self, query): + # FINALIZE_BODY + return query + + def explain_plan(self, query): + # EXPLAIN_PLAN_BODY + return str(query) + + +class HashPlanner(Planner): + def plan(self, query): + return super().plan(query) + + +class MergePlanner(Planner): + def plan(self, query): + return super().plan(query) +"#, + pre = filler("plan", 300), + post = filler("post", 120), + ) +} + +const LOOKUPS: &str = r#"class Exact: + def as_sql(self, compiler, connection): + return "%s = %s" + + +class IExact: + def as_sql(self, compiler, connection): + return "UPPER(%s) = UPPER(%s)" +"#; + +struct Fixture { + _dir: TestDir, + project: PathBuf, + compiler: String, + planner: String, +} + +impl Fixture { + fn new(label: &str) -> Self { + let dir = TestDir::new(label); + let project = dir.path.join("orm"); + std::fs::create_dir_all(&project).unwrap(); + let compiler = compiler_source(); + let planner = planner_source(); + std::fs::write(project.join("compiler.py"), &compiler).unwrap(); + std::fs::write(project.join("planner.py"), &planner).unwrap(); + std::fs::write(project.join("lookups.py"), LOOKUPS).unwrap(); + run_in(&dir.path, &["init", project.to_str().unwrap()]); + Self { + _dir: dir, + project, + compiler, + planner, + } + } + + fn explore(&self, query: &str) -> String { + run_in(&self.project, &["explore", query]) + } +} + +/// 1-based line of the first line containing `needle`. +fn line_of(source: &str, needle: &str) -> usize { + source + .lines() + .position(|line| line.contains(needle)) + .unwrap_or_else(|| panic!("{needle} not in fixture")) + + 1 +} + +/// The `#### ` section, header through the line before the next one. +fn section_for<'a>(text: &'a str, file: &str) -> &'a str { + let header = format!("#### {file} "); + let Some(start) = text.find(&header) else { + return ""; + }; + let rest = &text[start + header.len()..]; + let end = rest + .find("\n#### ") + .map_or(text.len(), |i| start + header.len() + i); + &text[start..end] +} + +/// Line numbers rendered in `file`'s section. +fn rendered_lines(text: &str, file: &str) -> BTreeSet { + section_for(text, file) + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(number, _)| number.parse().ok()) + .collect() +} + +fn assert_rendered(text: &str, file: &str, lines: std::ops::RangeInclusive) { + let rendered = rendered_lines(text, file); + let missing: Vec = lines.filter(|line| !rendered.contains(line)).collect(); + assert!( + missing.is_empty(), + "{file} lines {missing:?} not rendered in:\n{text}" + ); +} + +#[test] +fn a_qualified_name_returns_the_methods_whole_body() { + let fx = Fixture::new("qualified"); + let def = line_of(&fx.compiler, "def as_sql(self, with_limits"); + let tail = line_of(&fx.compiler, "AS_SQL_TAIL_MARKER"); + let text = fx.explore("SQLCompiler.as_sql pre_sql_setup get_select"); + assert_rendered(&text, "compiler.py", def..=tail); + assert!( + section_for(&text, "compiler.py").contains("PRE_SQL_SETUP_BODY"), + "a named method that still fits keeps its body:\n{text}" + ); +} + +#[test] +fn the_blast_radius_leads_with_the_qualified_method() { + let fx = Fixture::new("blast"); + let def = line_of(&fx.compiler, "def as_sql(self, with_limits"); + let text = fx.explore("SQLCompiler.as_sql pre_sql_setup get_select"); + let first = text + .lines() + .skip_while(|line| !line.starts_with("### Blast radius")) + .find(|line| line.starts_with("- `")) + .unwrap_or_else(|| panic!("no blast radius entry in:\n{text}")); + assert!( + first.starts_with(&format!("- `as_sql` (compiler.py:{def})")), + "{first}" + ); +} + +#[test] +fn a_line_anchor_returns_the_method_enclosing_that_line() { + let fx = Fixture::new("line"); + let def = line_of(&fx.compiler, "def as_sql(self, with_limits"); + let tail = line_of(&fx.compiler, "AS_SQL_TAIL_MARKER"); + // No symbol named: the line alone has to say which method. + let text = fx.explore(&format!("compiler.py:{} full body", def + 20)); + assert_rendered(&text, "compiler.py", def..=tail); +} + +#[test] +fn a_line_range_returns_exactly_that_span() { + let fx = Fixture::new("range"); + // The tail of `as_sql`: the span a windowed render elides and the agent + // then asks for by number. + let start = line_of(&fx.compiler, "result = self.get_qualify_sql()") + 1; + let end = line_of(&fx.compiler, "AS_SQL_TAIL_MARKER"); + let text = fx.explore(&format!("compiler.py lines {start}-{end} tail")); + assert_rendered(&text, "compiler.py", start..=end); +} + +#[test] +fn an_oversize_exact_body_keeps_its_head_and_its_call_into_a_named_symbol() { + let fx = Fixture::new("oversize"); + let call = line_of(&fx.planner, "PLAN_CALLS_STAGE"); + let text = fx.explore("Planner.plan stage_query explain_plan"); + let section = section_for(&text, "planner.py"); + assert!(section.contains("PLAN_HEAD"), "{text}"); + let rendered = rendered_lines(&text, "planner.py"); + assert!( + rendered.contains(&call), + "the call into stage_query (line {call}) is not rendered:\n{text}" + ); + // Windowed, not dumped: most of the 425-line body is elided. + assert!(rendered.len() < 300, "{} lines rendered", rendered.len()); +} diff --git a/crates/codegraph-cli/tests/explore_named_file_budget.rs b/crates/codegraph-cli/tests/explore_named_file_budget.rs new file mode 100644 index 0000000..2af59f3 --- /dev/null +++ b/crates/codegraph-cli/tests/explore_named_file_budget.rs @@ -0,0 +1,241 @@ +//! A file the query NAMED may use the budget the rest of the response left +//! unspent (upstream v1.6.1 #2068), and only that budget. +//! +//! Upstream's express case: "response.js res.send res.json res.render …" +//! admits one file, and the per-file valve cut the named `send` body at 55 of +//! 97 lines in a response far under its budget. This port's valve is the +//! per-file ceiling, `1.5 × max_chars_per_file`. +//! +//! 1. **Far from spent.** One file, five named functions whose bodies together +//! overrun the per-file ceiling but not the response budget: every body comes +//! back. +//! 2. **Shared.** The named file again, plus a lower file that renders whole: +//! the named file still grows past its ceiling, out of what the lower file +//! left, and the lower file keeps its whole section. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +fn bin() -> PathBuf { + PathBuf::from(env!("CARGO_BIN_EXE_codegraph")) +} + +struct TestDir { + path: PathBuf, +} + +impl TestDir { + fn new(label: &str) -> Self { + let path = std::env::temp_dir().join(format!( + "codegraph-cli-named-{label}-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir_all(&path).unwrap(); + Self { path } + } +} + +impl Drop for TestDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} + +fn run_in(cwd: &Path, args: &[&str]) -> String { + let output = Command::new(bin()) + .args(args) + .current_dir(cwd) + .env("CODEGRAPH_NO_DAEMON", "1") + .output() + .expect("run codegraph binary"); + assert!( + output.status.success(), + "{args:?} failed: {}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +const NAMED: [&str; 5] = [ + "sendBody", + "sendJson", + "renderView", + "redirectTo", + "sendFileStream", +]; +const RESPONSE: &str = "lib/response.ts"; + +/// Five independent named functions, then unrelated header helpers, so the +/// file is far too big to come back whole. +fn response_source() -> String { + let mut lines = vec![ + "export interface Res { headers: Record; body: string[]; status: number }" + .to_string(), + String::new(), + ]; + for name in NAMED { + lines.push(format!( + "export function {name}(res: Res, payload: unknown): Res {{" + )); + // 28 statements rather than upstream's 33: this port numbers every line + // and pays its section frame inside the same 13K budget, so five + // 36-line bodies no longer fit it at all. + for i in 0..28 { + lines.push(format!( + " res.body.push(String(payload ?? '').slice({i}, {}) + '{name}:{i}');", + i + 7 + )); + } + lines.extend([" return res;".to_string(), "}".to_string(), String::new()]); + } + for i in 0..60 { + lines.push(format!( + "export function headerSlot{i}(res: Res, value: string): Res {{" + )); + for k in 0..6 { + lines.push(format!( + " res.headers['x-slot-{i}-{k}'] = value.slice({k}, {});", + k + 3 + )); + } + lines.extend([" return res;".to_string(), "}".to_string(), String::new()]); + } + lines.join("\n") +} + +/// A small file that calls one of the named functions, so it ranks into the +/// response and renders whole. +fn handler_source() -> String { + let mut lines = vec![ + "import { sendBody, Res } from '../lib/response';".to_string(), + String::new(), + "export function handleUpload(res: Res, chunks: string[]): Res {".to_string(), + ]; + for i in 0..12 { + lines.push(format!(" res = sendBody(res, chunks[{i}]);")); + } + lines.extend([" return res;".to_string(), "}".to_string()]); + lines.join("\n") + "\n" +} + +fn indexed(label: &str, files: &[(&str, String)]) -> (TestDir, PathBuf) { + let dir = TestDir::new(label); + let project = dir.path.join("app"); + for (rel, body) in files { + let path = project.join(rel); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + std::fs::write(path, body).unwrap(); + } + run_in(&dir.path, &["init", project.to_str().unwrap()]); + (dir, project) +} + +/// Line numbers rendered in `file`'s section. +fn rendered_lines(text: &str, file: &str) -> BTreeSet { + let header = format!("#### {file} "); + let Some(start) = text.find(&header) else { + return BTreeSet::new(); + }; + let rest = &text[start + header.len()..]; + let section = &rest[..rest.find("\n#### ").unwrap_or(rest.len())]; + section + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(number, _)| number.parse().ok()) + .collect() +} + +/// 1-based `(first, last)` line of each named function in `source`. +fn body_spans(source: &str) -> Vec<(&'static str, usize, usize)> { + let lines: Vec<&str> = source.lines().collect(); + NAMED + .iter() + .map(|name| { + let start = lines + .iter() + .position(|line| line.starts_with(&format!("export function {name}("))) + .unwrap() + + 1; + let end = start + lines[start..].iter().position(|line| *line == "}").unwrap() + 1; + (*name, start, end) + }) + .collect() +} + +fn incomplete_bodies(text: &str, source: &str) -> Vec<&'static str> { + let rendered = rendered_lines(text, RESPONSE); + body_spans(source) + .into_iter() + .filter(|(_, start, end)| (*start..=*end).any(|line| !rendered.contains(&line))) + .map(|(name, _, _)| name) + .collect() +} + +#[test] +fn a_named_file_far_from_its_budget_returns_every_named_body() { + let source = response_source(); + let (_dir, project) = indexed("far", &[(RESPONSE, source.clone())]); + let text = run_in(&project, &["explore", &NAMED.join(" ")]); + // Fixture shape: the bodies overrun the 5,700-char per-file ceiling. + let bodies: usize = body_spans(&source) + .iter() + .map(|(_, start, end)| { + source.lines().collect::>()[start - 1..*end] + .join("\n") + .len() + }) + .sum(); + assert!(bodies > 5700, "fixture too small: {bodies}"); + assert_eq!( + incomplete_bodies(&text, &source), + Vec::<&str>::new(), + "{text}" + ); + assert!( + text.len() <= 13000, + "response {} past its budget", + text.len() + ); +} + +#[test] +fn a_named_file_takes_only_what_the_other_files_left() { + let source = response_source(); + let handler = handler_source(); + let (_dir, project) = indexed( + "shared", + &[ + (RESPONSE, source.clone()), + ("routes/upload.ts", handler.clone()), + ], + ); + let text = run_in( + &project, + &["explore", &format!("{} handleUpload", NAMED.join(" "))], + ); + let lower = rendered_lines(&text, "routes/upload.ts"); + let want: BTreeSet = (1..=handler.lines().count()).collect(); + assert_eq!( + lower, want, + "the lower file must keep its whole section:\n{text}" + ); + let named_source: usize = text + .split("#### ") + .find(|section| section.starts_with(&format!("{RESPONSE} "))) + .map_or(0, str::len); + assert!( + named_source > 5700, + "the named file did not grow past its per-file ceiling ({named_source}):\n{text}" + ); + assert!( + text.len() <= 13000, + "response {} past its budget", + text.len() + ); +} diff --git a/crates/codegraph-cli/tests/explore_pinned_cap_named.rs b/crates/codegraph-cli/tests/explore_pinned_cap_named.rs new file mode 100644 index 0000000..671551f --- /dev/null +++ b/crates/codegraph-cli/tests/explore_pinned_cap_named.rs @@ -0,0 +1,293 @@ +//! A symbol the query names survives a pinned file's node cap (upstream v1.6.1 +//! #2064, a guard for #2062). +//! +//! A file named by path is pinned: its first 300 symbols by start line enter the +//! gather. Past that cap the file splits in two: a head cluster of one named +//! method merged with ~300 pinned neighbours and, one cluster down, the named +//! symbols the cap cut off. The head ranked first on density and shrank against +//! its own members only, so pin filler spent the file's room and the named +//! cluster rendered nothing. The cross-cluster hold-back of #2062 reaches this +//! shape without knowing about pinning; this fixture guards the pinned entry +//! path, since a hold-back re-scoped away from it would reopen the bug here and +//! nowhere else. +//! +//! In this port the shape does not go red without #2062 either: the head +//! cluster carries no edge-line members, so it never out-densifies the named +//! cluster, which ranks first. The guard holds the property, not the ranking. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +const FILE: &str = "src/rpc/protocol.ts"; + +fn bin() -> PathBuf { + PathBuf::from(env!("CARGO_BIN_EXE_codegraph")) +} + +struct TestDir { + path: PathBuf, +} + +impl TestDir { + fn new(label: &str) -> Self { + let path = std::env::temp_dir().join(format!( + "codegraph-cli-pincap-{label}-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir_all(&path).unwrap(); + Self { path } + } +} + +impl Drop for TestDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} + +fn run_in(cwd: &Path, args: &[&str]) -> String { + let output = Command::new(bin()) + .args(args) + .current_dir(cwd) + .env("CODEGRAPH_NO_DAEMON", "1") + .output() + .expect("run codegraph binary"); + assert!( + output.status.success(), + "{args:?} failed: {}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +/// `receiveOne` → `dispatchRequest` / `MessageCodec.serializeAck` at the top, +/// 300 unrelated `trackMetric*` methods, then the `MessageKind` enum and a +/// `MessageCodec` whose five serializers are followed by 150 `encodeField*` +/// members. +fn protocol_source() -> String { + let mut lines: Vec = [ + "export class Protocol {", + " private handlers = new Map unknown>();", + " private outbox: Uint8Array[] = [];", + "", + " receiveOne(raw: Uint8Array): void {", + " const kind = raw[0] as MessageKind;", + " if (kind === MessageKind.Request) {", + " this.dispatchRequest(raw);", + " this.outbox.push(MessageCodec.serializeAck(raw[1] ?? 0));", + " } else if (kind === MessageKind.Cancel) {", + " this.outbox.push(MessageCodec.serializeCancel(raw[1] ?? 0));", + " }", + " }", + "", + " private dispatchRequest(raw: Uint8Array): void {", + " const id = raw[1] ?? 0;", + " const method = String(raw[2] ?? '');", + " const handler = this.handlers.get(method);", + " if (!handler) {", + " this.outbox.push(MessageCodec.serializeReplyErr(id, new Error(`no handler: ${method}`)));", + " return;", + " }", + " const started = Date.now();", + " let result: unknown;", + " try {", + " result = handler(raw.subarray(3));", + " } catch (err) {", + " this.outbox.push(MessageCodec.serializeReplyErr(id, err as Error));", + " return;", + " }", + " if (Date.now() - started > 1000) {", + " this.outbox.push(MessageCodec.serializeAck(id));", + " }", + " if (this.outbox.length > 64) {", + " this.outbox.splice(0, this.outbox.length - 64);", + " }", + " this.outbox.push(MessageCodec.serializeReply(id, result ?? null));", + " }", + ] + .map(str::to_string) + .to_vec(); + for i in 0..300 { + lines.extend([ + String::new(), + format!(" trackMetric{i}(value: number): number {{"), + format!(" return value * {} + this.outbox.length;", i + 1), + " }".to_string(), + ]); + } + lines.extend( + [ + "}", + "", + "export const enum MessageKind {", + " Request = 1,", + " Reply = 2,", + " ReplyErr = 3,", + " Cancel = 4,", + " Ack = 5,", + "}", + "", + "export class MessageCodec {", + " static serializeRequest(id: number, method: string): Uint8Array {", + " const body = new TextEncoder().encode(method);", + " const out = new Uint8Array(body.length + 2);", + " out[0] = MessageKind.Request;", + " out[1] = id & 0xff;", + " out.set(body, 2);", + " return out;", + " }", + "", + " static serializeAck(id: number): Uint8Array {", + " return Uint8Array.of(MessageKind.Ack, id & 0xff);", + " }", + "", + " static serializeCancel(id: number): Uint8Array {", + " return Uint8Array.of(MessageKind.Cancel, id & 0xff);", + " }", + "", + " static serializeReply(id: number, value: unknown): Uint8Array {", + " const body = new TextEncoder().encode(JSON.stringify(value ?? null));", + " const out = new Uint8Array(body.length + 2);", + " out[0] = MessageKind.Reply;", + " out[1] = id & 0xff;", + " out.set(body, 2);", + " return out;", + " }", + "", + " static serializeReplyErr(id: number, err: Error): Uint8Array {", + " const body = new TextEncoder().encode(err.message);", + " const out = new Uint8Array(body.length + 2);", + " out[0] = MessageKind.ReplyErr;", + " out[1] = id & 0xff;", + " out.set(body, 2);", + " return out;", + " }", + ] + .map(str::to_string), + ); + for i in 0..150 { + lines.extend([ + String::new(), + format!(" static encodeField{i}(id: number): Uint8Array {{"), + format!( + " return Uint8Array.of({}, id & 0xff, {});", + i % 250, + (i * 7) % 250 + ), + " }".to_string(), + ]); + } + lines.push("}".to_string()); + lines.join("\n") + "\n" +} + +struct Fixture { + _dir: TestDir, + project: PathBuf, + source: String, +} + +impl Fixture { + fn new(label: &str) -> Self { + let dir = TestDir::new(label); + let project = dir.path.join("rpc"); + let source = protocol_source(); + let path = project.join(FILE); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + std::fs::write(&path, &source).unwrap(); + run_in(&dir.path, &["init", project.to_str().unwrap()]); + Self { + _dir: dir, + project, + source, + } + } + + fn explore(&self, query: &str) -> String { + run_in(&self.project, &["explore", query]) + } + + /// 1-based `(first, last)` line of the method `name`. + fn span_of(&self, name: &str) -> (usize, usize) { + let lines: Vec<&str> = self.source.lines().collect(); + let start = lines + .iter() + .position(|line| { + let trimmed = line.trim_start(); + [ + format!("{name}("), + format!("private {name}("), + format!("static {name}("), + ] + .iter() + .any(|opening| trimmed.starts_with(opening)) + }) + .unwrap_or_else(|| panic!("{name} not in fixture")); + let end = start + + lines[start..] + .iter() + .position(|line| *line == " }") + .unwrap(); + (start + 1, end + 1) + } + + /// Names whose whole definition did NOT reach the response. + fn incomplete_bodies<'a>(&self, text: &str, names: &[&'a str]) -> Vec<&'a str> { + let header = format!("#### {FILE} "); + let rendered: BTreeSet = text + .find(&header) + .map(|start| { + let rest = &text[start + header.len()..]; + rest[..rest.find("\n#### ").unwrap_or(rest.len())] + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(number, _)| number.parse().ok()) + .collect() + }) + .unwrap_or_default(); + names + .iter() + .copied() + .filter(|name| { + let (start, end) = self.span_of(name); + (start..=end).any(|line| !rendered.contains(&line)) + }) + .collect() + } +} + +#[test] +fn the_named_symbols_render_whole_when_the_file_is_not_pinned() { + // The parity the pinned queries below are held to. + let fx = Fixture::new("unpinned"); + let text = fx.explore("receiveOne serializeAck"); + assert_eq!( + fx.incomplete_bodies(&text, &["receiveOne", "serializeAck"]), + Vec::<&str>::new(), + "{text}" + ); +} + +#[test] +fn a_pinned_file_returns_a_named_symbol_past_the_node_cap() { + let fx = Fixture::new("pinned"); + let text = fx.explore("protocol.ts receiveOne serializeAck"); + assert!(text.contains("1 file pinned from the query."), "{text}"); + assert_eq!( + fx.incomplete_bodies(&text, &["receiveOne", "serializeAck"]), + Vec::<&str>::new(), + "{text}" + ); + let lone = fx.explore("protocol.ts serializeAck"); + assert_eq!( + fx.incomplete_bodies(&lone, &["serializeAck"]), + Vec::<&str>::new(), + "{lone}" + ); +} diff --git a/crates/codegraph-cli/tests/explore_same_basename_pin.rs b/crates/codegraph-cli/tests/explore_same_basename_pin.rs new file mode 100644 index 0000000..e62647f --- /dev/null +++ b/crates/codegraph-cli/tests/explore_same_basename_pin.rs @@ -0,0 +1,336 @@ +//! A bare basename two directories share pins only the file defining what the +//! query names (upstream v1.6.1 #2071). +//! +//! vscode has two `editorOptions.ts`: the editor's option registry, holding +//! `clampedInt`, `cursorStyleToString` and `cursorStyleFromString`, and a small +//! workbench helper holding none of them. `editorOptions.ts clampedInt …` +//! pinned both, and the two pins split the pinned room, so naming the file made +//! the answer worse. The fixture is that shape: the registry is ~60 option +//! classes with the named functions in the middle, the helper a handful of +//! unrelated functions. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::Command; + +const REGISTRY: &str = "src/editor/common/config/editorOptions.ts"; +const HELPER: &str = "src/workbench/common/editor/editorOptions.ts"; +const NAMED: [&str; 3] = ["clampedInt", "cursorStyleToString", "cursorStyleFromString"]; + +fn bin() -> PathBuf { + PathBuf::from(env!("CARGO_BIN_EXE_codegraph")) +} + +struct TestDir { + path: PathBuf, +} + +impl TestDir { + fn new(label: &str) -> Self { + let path = std::env::temp_dir().join(format!( + "codegraph-cli-basename-{label}-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir_all(&path).unwrap(); + Self { path } + } +} + +impl Drop for TestDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} + +fn run_in(cwd: &Path, args: &[&str]) -> String { + let output = Command::new(bin()) + .args(args) + .current_dir(cwd) + .env("CODEGRAPH_NO_DAEMON", "1") + .output() + .expect("run codegraph binary"); + assert!( + output.status.success(), + "{args:?} failed: {}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + String::from_utf8_lossy(&output.stdout).into_owned() +} + +fn option_class(lines: &mut Vec, i: usize) { + lines.extend([ + String::new(), + format!("export class EditorOption{i} extends BaseEditorOption {{"), + format!(" private readonly allowed{i} = ['alpha{i}', 'beta{i}', 'gamma{i}'];"), + String::new(), + " constructor() {".to_string(), + format!(" super({i}, 'option{i}', 'alpha{i}');"), + " }".to_string(), + String::new(), + " public validate(input: unknown): string {".to_string(), + " if (typeof input !== 'string') {".to_string(), + " return this.defaultValue;".to_string(), + " }".to_string(), + format!(" return this.allowed{i}.includes(input) ? input : this.defaultValue;"), + " }".to_string(), + "}".to_string(), + ]); +} + +fn registry_source() -> String { + let mut lines: Vec = [ + "export interface IConfigurationPropertySchema {", + " type?: string;", + " default?: unknown;", + " minimum?: number;", + " maximum?: number;", + "}", + "", + "export abstract class BaseEditorOption {", + " constructor(public readonly id: K, public readonly name: string, public readonly defaultValue: V) {}", + " public abstract validate(input: unknown): V;", + "}", + ] + .map(str::to_string) + .to_vec(); + for i in 0..30 { + option_class(&mut lines, i); + } + lines.extend( + [ + "", + "export function clampedInt(value: unknown, defaultValue: T, minimum: number, maximum: number): number | T {", + " if (typeof value === 'undefined') {", + " return defaultValue;", + " }", + " let r = parseInt(String(value), 10);", + " if (isNaN(r)) {", + " return defaultValue;", + " }", + " r = Math.max(minimum, r);", + " r = Math.min(maximum, r);", + " return r | 0;", + "}", + "", + "export class EditorIntOption extends BaseEditorOption {", + " public static clampedInt(value: unknown, defaultValue: T, minimum: number, maximum: number): number | T {", + " return clampedInt(value, defaultValue, minimum, maximum);", + " }", + "", + " constructor(id: K, name: string, defaultValue: number, public readonly minimum: number, public readonly maximum: number, schema?: IConfigurationPropertySchema) {", + " if (typeof schema !== 'undefined') {", + " schema.type = 'integer';", + " schema.default = defaultValue;", + " schema.minimum = minimum;", + " schema.maximum = maximum;", + " }", + " super(id, name, defaultValue);", + " }", + "", + " public validate(input: unknown): number {", + " return EditorIntOption.clampedInt(input, this.defaultValue, this.minimum, this.maximum);", + " }", + "}", + "", + "export const enum TextEditorCursorStyle {", + " Line = 1,", + " Block = 2,", + " Underline = 3,", + " LineThin = 4,", + " BlockOutline = 5,", + " UnderlineThin = 6,", + "}", + "", + "export function cursorStyleToString(cursorStyle: TextEditorCursorStyle): string {", + " switch (cursorStyle) {", + " case TextEditorCursorStyle.Line: return 'line';", + " case TextEditorCursorStyle.Block: return 'block';", + " case TextEditorCursorStyle.Underline: return 'underline';", + " case TextEditorCursorStyle.LineThin: return 'line-thin';", + " case TextEditorCursorStyle.BlockOutline: return 'block-outline';", + " case TextEditorCursorStyle.UnderlineThin: return 'underline-thin';", + " }", + "}", + "", + "export function cursorStyleFromString(cursorStyle: string): TextEditorCursorStyle {", + " switch (cursorStyle) {", + " case 'line': return TextEditorCursorStyle.Line;", + " case 'block': return TextEditorCursorStyle.Block;", + " case 'underline': return TextEditorCursorStyle.Underline;", + " case 'line-thin': return TextEditorCursorStyle.LineThin;", + " case 'block-outline': return TextEditorCursorStyle.BlockOutline;", + " case 'underline-thin': return TextEditorCursorStyle.UnderlineThin;", + " }", + " return TextEditorCursorStyle.Line;", + "}", + ] + .map(str::to_string), + ); + for i in 30..60 { + option_class(&mut lines, i); + } + lines.join("\n") + "\n" +} + +fn helper_source() -> String { + let mut lines: Vec = [ + "export interface ITextEditorViewState { scrollTop: number; cursor: number }", + "", + "export function applyTextEditorOptions(options: { selection?: number; viewState?: ITextEditorViewState }, editor: { reveal(line: number): void; restore(s: ITextEditorViewState): void }): boolean {", + " if (options.viewState) {", + " editor.restore(massageEditorViewState(options.viewState));", + " return true;", + " }", + " if (typeof options.selection === \"number\") {", + " editor.reveal(options.selection);", + " return true;", + " }", + " return false;", + "}", + "", + "function massageEditorViewState(state: ITextEditorViewState): ITextEditorViewState {", + " return { scrollTop: Math.max(0, state.scrollTop), cursor: Math.max(0, state.cursor) };", + "}", + ] + .map(str::to_string) + .to_vec(); + for i in 0..12 { + lines.extend([ + String::new(), + format!( + "export function restoreEditorGroup{i}(state: ITextEditorViewState): number {{" + ), + format!( + " const offset = state.scrollTop * {} + state.cursor;", + i + 2 + ), + format!( + " return offset > {} ? offset - {i} : offset + {i};", + 100 * (i + 1) + ), + "}".to_string(), + ]); + } + lines.join("\n") + "\n" +} + +struct Fixture { + _dir: TestDir, + project: PathBuf, + registry: String, +} + +impl Fixture { + fn new(label: &str) -> Self { + let dir = TestDir::new(label); + let project = dir.path.join("editor"); + let registry = registry_source(); + for (rel, body) in [(REGISTRY, registry.clone()), (HELPER, helper_source())] { + let path = project.join(rel); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + std::fs::write(path, body).unwrap(); + } + run_in(&dir.path, &["init", project.to_str().unwrap()]); + Self { + _dir: dir, + project, + registry, + } + } + + fn explore(&self, query: &str) -> String { + run_in(&self.project, &["explore", query]) + } + + /// How many of the registry's named definitions the response sent whole: + /// the two `clampedInt`s (function and static method) and the two + /// `cursorStyle*` functions. + fn complete_named_bodies(&self, text: &str) -> usize { + let header = format!("#### {REGISTRY} "); + let rendered: BTreeSet = text + .find(&header) + .map(|start| { + let rest = &text[start + header.len()..]; + rest[..rest.find("\n#### ").unwrap_or(rest.len())] + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(number, _)| number.parse().ok()) + .collect() + }) + .unwrap_or_default(); + let lines: Vec<&str> = self.registry.lines().collect(); + let mut complete = 0; + for (index, line) in lines.iter().enumerate() { + let opens_named = NAMED.iter().any(|name| { + line.starts_with(&format!("export function {name}")) + || line + .trim_start() + .starts_with(&format!("public static {name}")) + }); + if !opens_named { + continue; + } + let indent = line.len() - line.trim_start().len(); + let close = format!("{}}}", " ".repeat(indent)); + let end = index + + lines[index..] + .iter() + .position(|candidate| *candidate == close) + .unwrap(); + if (index + 1..=end + 1).all(|n| rendered.contains(&n)) { + complete += 1; + } + } + complete + } +} + +#[test] +fn pins_only_the_file_defining_the_named_symbols_and_names_the_one_set_aside() { + let fx = Fixture::new("narrow"); + let text = fx.explore(&format!("editorOptions.ts {}", NAMED.join(" "))); + assert!(text.contains("1 file pinned from the query."), "{text}"); + assert!( + text.contains(&format!( + "Not pinned: `{HELPER}`, which defines none of the named symbols." + )), + "{text}" + ); + assert!(!text.contains(&format!("#### {HELPER} ")), "{text}"); +} + +/// Upstream's two pins split a reserved share; this port funds files in order, +/// so the room was never split and this holds with or without narrowing. It +/// guards that naming the file never costs the named file its bodies. +#[test] +fn the_named_file_keeps_the_room_it_gets_without_the_path() { + let fx = Fixture::new("room"); + let pinned = fx.explore(&format!("editorOptions.ts {}", NAMED.join(" "))); + let unpinned = fx.explore(&NAMED.join(" ")); + assert!( + fx.complete_named_bodies(&pinned) >= fx.complete_named_bodies(&unpinned), + "pinned:\n{pinned}\nunpinned:\n{unpinned}" + ); + assert_eq!(fx.complete_named_bodies(&pinned), 4, "{pinned}"); +} + +#[test] +fn pins_both_when_the_query_names_no_symbol_to_choose_between_them() { + let fx = Fixture::new("both"); + let text = fx.explore("editorOptions.ts"); + assert!(text.contains("2 files pinned from the query."), "{text}"); + assert!(!text.contains("Not pinned:"), "{text}"); +} + +#[test] +fn a_path_with_its_directory_pins_that_file_alone_with_no_note() { + let fx = Fixture::new("dir"); + let text = fx.explore(&format!("{HELPER} {}", NAMED.join(" "))); + assert!(text.contains("1 file pinned from the query."), "{text}"); + assert!(!text.contains("Not pinned:"), "{text}"); +} diff --git a/crates/codegraph-mcp/src/engine.rs b/crates/codegraph-mcp/src/engine.rs index 4cd2af1..9093d62 100644 --- a/crates/codegraph-mcp/src/engine.rs +++ b/crates/codegraph-mcp/src/engine.rs @@ -10,7 +10,7 @@ use std::cell::RefCell; use std::collections::{BTreeSet, HashMap, HashSet}; use std::fs; use std::path::{Path, PathBuf}; -use std::sync::Arc; +use std::sync::{Arc, LazyLock, OnceLock}; use std::time::{Duration, Instant, UNIX_EPOCH}; use codegraph_core::config::Config; @@ -28,7 +28,9 @@ use serde_json::Value; use crate::dynamic_boundaries::scan_dynamic_dispatch; use crate::explore_budget::{ExploreOutputBudget, get_explore_output_budget}; use crate::protocol::ToolResult; -use crate::query_paths::{extract_query_paths_with_file_probe, query_might_contain_paths}; +use crate::query_paths::{ + QueryLineAnchor, extract_query_paths_with_probes, query_might_contain_paths, +}; /// Default caller/callee recursion depth for callers/callees tools. The upstream /// `getCallers`/`getCallees` default to `maxDepth: 1` (`traversal.ts` callers @@ -58,6 +60,22 @@ const POINTER_MAX_FILES: usize = 10; /// caller then drops the file whole rather than emitting a stub. const MIN_WINDOW_LINES: usize = 12; +/// Importance of an EXACT target (upstream #2063): a callable the query +/// singled out with no ambiguity left — a qualified name resolving to at most +/// three callables, or the callable enclosing a single-line anchor — or a line +/// span the query anchored. Above a protected focus (11) and a root (10). +const EXACT_IMPORTANCE: u32 = 12; + +/// Lines either side of a single-line anchor that no callable encloses. +const ANCHOR_LINE_CONTEXT: usize = 15; + +/// Most lines an oversize exact body is windowed on beyond its head: the +/// anchored line inside it and its calls into the question's other symbols. +const MAX_BODY_FOCUS_LINES: usize = 6; + +/// Query tokens consulted for qualified exact targets — upstream's seeder cap. +const MAX_EXACT_SYMBOL_TOKENS: usize = 16; + /// Minimum source budget protected for every path explicitly named in an /// explore query. Pinned files are ordered first and reserve this much for one /// another before incidental files can spend the remaining envelope. @@ -1226,20 +1244,29 @@ impl CodeGraphEngine { .into_iter() .map(|file| file.path) .collect::>(); - extract_query_paths_with_file_probe( + extract_query_paths_with_probes( &query, &indexed_paths, max_files.min(8), - &|relative| project_regular_file_exists(&self.project_root, relative), + Some(&|relative| project_regular_file_exists(&self.project_root, relative)), + Some(&|symbol| self.files_defining_symbol(symbol)), ) } else { - extract_query_paths_with_file_probe(&query, &[], max_files.min(8), &|relative| { - project_regular_file_exists(&self.project_root, relative) - }) + extract_query_paths_with_probes( + &query, + &[], + max_files.min(8), + Some(&|relative| project_regular_file_exists(&self.project_root, relative)), + None, + ) }; let match_query = path_extraction.stripped_query.trim().to_string(); - let subgraph = - self.find_relevant_context_with_pins(&match_query, &path_extraction.pinned_files)?; + let exact = self.exact_targets(&match_query, &path_extraction.line_anchors); + let subgraph = self.find_relevant_context_with_pins( + &match_query, + &path_extraction.pinned_files, + &exact, + )?; let unresolved_note = (!path_extraction.unresolved_path_spans.is_empty()).then(|| { path_extraction .unresolved_path_spans @@ -1253,8 +1280,9 @@ impl CodeGraphEngine { .as_ref() .map(|spans| format!(" (no indexed file uniquely matches {spans})")) .unwrap_or_default(); + let explanation = self.explore_miss_explanation(&match_query); return Ok(ToolResult::text(format!( - "No relevant code found for \"{query}\"{miss_note}" + "No relevant code found for \"{query}\"{miss_note}{explanation}" ))); } @@ -1282,6 +1310,34 @@ impl CodeGraphEngine { if let Some(spans) = &unresolved_note { lines.push(format!("No indexed file uniquely matches {spans}.")); } + // A same-named file a span did not pin is named, so an agent that did + // mean it sees where it went instead of a silent drop (#2071). + let set_aside_note = path_extraction + .set_aside_matches + .iter() + .map(|entry| { + let which = if entry.files.len() <= 2 { + entry + .files + .iter() + .map(|file| format!("`{file}`")) + .collect::>() + .join(", ") + } else { + format!("{} other `{}` files", entry.files.len(), entry.span) + }; + let verb = if entry.files.len() == 1 { + "defines" + } else { + "define" + }; + format!("Not pinned: {which}, which {verb} none of the named symbols.") + }) + .collect::>() + .join(" "); + if !set_aside_note.is_empty() { + lines.push(set_aside_note); + } lines.push(String::new()); if let Some(blast) = self.build_blast_radius(&subgraph)? { @@ -1330,10 +1386,16 @@ impl CodeGraphEngine { // list (`tools.ts:2483-2905`). let mut total_chars: usize = lines.join("\n").len(); let mut files_included = 0usize; - let mut any_file_trimmed = false; + let mut budget_dropped = false; let mut excluded_files: Vec<&String> = Vec::new(); let mut rendered_sources: Vec<(String, usize)> = Vec::new(); + // Every admitted section: `(line index, path, what it elided)`. + let mut sections: Vec<(usize, &String, Vec)> = Vec::new(); + // Named files whose per-file ceiling clipped them, with what a second + // render needs: `(line index, path, focuses, exact focus)`. + let mut named_clipped: Vec<(usize, &String, Vec, Vec)> = Vec::new(); let precise_tokens = precise_query_tokens(&match_query); + let question_ids = subgraph.question_ids(&precise_tokens); let indexed_file_count = self.store.counts().ok().map(|c| c.file_count); let effective_budget = budget .max_output_chars @@ -1359,6 +1421,12 @@ impl CodeGraphEngine { let lang = subgraph.file_language(file_path); let focuses = subgraph.file_focus_lines(file_path, &precise_tokens); + let exact_focus = self.exact_body_focus( + &subgraph, + file_path, + &path_extraction.line_anchors, + &question_ids, + ); let available = effective_budget .saturating_sub(total_chars) .saturating_sub(1); @@ -1375,6 +1443,8 @@ impl CodeGraphEngine { budget: &budget, funded_headroom, focuses: &focuses, + exact_focus: &exact_focus, + lift_file_cap: false, drifted: possibly_drifted, }; let rendered = self.render_explore_file(&subgraph, file_path, &file_lines, &lang, &ctx); @@ -1387,24 +1457,101 @@ impl CodeGraphEngine { // expression appears at the accumulator below, so the admission test // and the running total cannot disagree. if total_chars + rendered.section.len() + 1 > effective_budget { - any_file_trimmed = true; + budget_dropped = true; excluded_files.push(file_path); continue; } - if rendered.section.contains("... (gap) ...") - || rendered.section.contains("more (signatures elided)") - { - any_file_trimmed = true; - } let section_len = rendered.section.len(); if rendered.source_emitted { rendered_sources.push(((*file_path).clone(), lines.len())); } + let named = subgraph.is_pinned(file_path) + || subgraph.holds_exact(file_path) + || !focuses.is_empty(); + if rendered.clipped && named { + named_clipped.push((lines.len(), file_path, focuses, exact_focus)); + } + sections.push((lines.len(), file_path, rendered.elided)); lines.push(rendered.section); total_chars += section_len + 1; files_included += 1; } + // A file the query NAMED — by path, or by a symbol it defines — that its + // per-file ceiling clipped is rendered again into what the response left + // unspent (upstream #2068), in file order. Only that: every other file + // keeps exactly the section it was given, and the pointer list below + // keeps the room it would have had. Where nothing is spare a named file + // renders exactly as before. + let pointer_candidates: Vec = + if budget.include_additional_files && !excluded_files.is_empty() { + excluded_files + .iter() + .map(|file_path| { + format!("- {file_path}: {}", subgraph.file_node_locations(file_path)) + }) + .collect() + } else { + Vec::new() + }; + let pointer_reserve: usize = fit_pointer_lines( + &pointer_candidates, + effective_budget.saturating_sub(total_chars), + ) + .0 + .iter() + .map(|line| line.len() + 1) + .sum(); + let mut spare = effective_budget + .saturating_sub(total_chars) + .saturating_sub(pointer_reserve); + for (line_index, file_path, focuses, exact_focus) in &named_clipped { + if spare == 0 { + break; + } + let source = self.project_source(file_path); + let Some(content) = source.content.clone() else { + continue; + }; + let file_lines: Vec<&str> = content.split('\n').collect(); + let lang = subgraph.file_language(file_path); + let given = lines[*line_index].len(); + let ctx = RenderCtx { + budget: &budget, + funded_headroom: given + spare, + focuses, + exact_focus, + lift_file_cap: true, + drifted: source.is_possibly_drifted(), + }; + let grown = self.render_explore_file(&subgraph, file_path, &file_lines, &lang, &ctx); + let len = grown.section.len(); + if !grown.source_emitted || len <= given || len > given + spare { + continue; + } + spare -= len - given; + total_chars += len - given; + lines[*line_index] = grown.section; + if let Some(entry) = sections.iter_mut().find(|entry| entry.0 == *line_index) { + entry.2 = grown.elided; + } + } + // Sections missing some of what they set out to deliver, in render + // order, measured from the ranges actually sent (#2077): "complete" is + // claimed only for sections that are. + let trimmed_shown: Vec<(String, Vec)> = sections + .iter() + .filter(|(_, _, elided)| !elided.is_empty()) + .map(|(_, path, elided)| ((*path).clone(), elided.clone())) + .collect(); + let any_file_trimmed = budget_dropped + || !trimmed_shown.is_empty() + || sections.iter().any(|(index, _, _)| { + lines[*index].contains("... (gap) ...") + || lines[*index].contains("more (signatures elided)") + }); + let mut epilogue_room = effective_budget.saturating_sub(total_chars); + // "Additional relevant files (not shown)" — the excluded set, so the // agent can request specifics. Gated by the budget (`tools.ts:2910-2927`). if budget.include_additional_files && !excluded_files.is_empty() { @@ -1414,14 +1561,8 @@ impl CodeGraphEngine { // not use, so however long the pointer list is it cannot push the // response past its budget. The frame and the tail are already paid // for by the reserve and so are not charged here. - let candidates: Vec = excluded_files - .iter() - .map(|file_path| { - format!("- {file_path}: {}", subgraph.file_node_locations(file_path)) - }) - .collect(); - let (pointer_lines, unlisted) = - fit_pointer_lines(&candidates, effective_budget.saturating_sub(total_chars)); + let (pointer_lines, unlisted) = fit_pointer_lines(&pointer_candidates, epilogue_room); + epilogue_room = epilogue_room.saturating_sub(joined_line_cost(&pointer_lines)); lines.extend(pointer_lines); // The TRUE unlisted count — those dropped by elasticity plus those // dropped by the file cap — so a file the response did not render is @@ -1442,9 +1583,27 @@ impl CodeGraphEngine { // question was still uncovered (#1504). The useful half — one more explore // beats falling back to Read — is kept, the phantom cap is dropped, and // the absence of a limit is stated outright rather than left to inference. + // + // The completeness note is the most specific candidate that fits: the + // reserve holds the least specific one, so a note naming the trimmed + // files and symbols is paid only from what the sections and the pointer + // list left, and never costs a pointer line (upstream #2077). + let completeness_note = if budget.include_completeness_signal { + let candidates = + completeness_notes(files_included, &trimmed_shown, &subgraph.file_order); + let room = completeness_note_floor(max_files) + epilogue_room; + candidates + .iter() + .find(|note| note.len() <= room) + .or(candidates.last()) + .cloned() + .unwrap_or_default() + } else { + String::new() + }; lines.extend(epilogue_note_lines( &budget, - files_included, + &completeness_note, any_file_trimmed, indexed_file_count, )); @@ -1454,11 +1613,139 @@ impl CodeGraphEngine { // (`tools.ts:2954-2975`). let output = lines.join("\n"); let hard_ceiling = ((budget.max_output_chars as f64 * 1.5).round() as usize).min(25000); - let (output, kept_prefix_len) = cut_at_section_boundary(&output, hard_ceiling); + let trimmed_paths: Vec<&str> = trimmed_shown + .iter() + .map(|(path, _)| path.as_str()) + .collect(); + let (output, kept_prefix_len) = + cut_at_section_boundary(&output, hard_ceiling, &trimmed_paths); self.mark_surviving_explore_citations(&lines, &rendered_sources, kept_prefix_len); Ok(ToolResult::text(output)) } + /// Why an explore came back empty (upstream #1904). Explore matches names + /// and indexed code words lexically, not by meaning, so the answer names + /// the checked words that matched nothing, those that matched but scored + /// out, and indexed names sharing a word to retry with — from two bounded + /// FTS queries and the query-time name segments (the golden-neutral + /// substitute for upstream's `name_segment_vocab`). + fn explore_miss_explanation(&self, match_query: &str) -> String { + let mut explanation = "\n\nExplore matches symbol/file names and indexed code words lexically, not by meaning.".to_string(); + if self + .store + .counts() + .is_ok_and(|counts| counts.node_count == 0) + { + explanation.push_str("\nThis project has nothing indexed."); + return explanation; + } + let miss = self.explore_miss_diagnostics(match_query); + explanation.push_str("\nChecked indexed names, signatures, docstrings (FTS prefixes) and live name segments; not all source text."); + if miss.limited { + explanation.push_str("\nWord check limited to 16 words of at most 64 characters."); + } + let unmatched = capped_word_list(&miss.unmatched, 250); + explanation.push_str(&format!( + "\nNo lexical matches for checked words: {}.", + if unmatched.is_empty() { + "(none)" + } else { + &unmatched + } + )); + if !miss.matched.is_empty() { + explanation.push_str(&format!( + "\nMatched indexed words: {}; these did not yield a relevant result after filtering/scoring.", + capped_word_list(&miss.matched, 200) + )); + } + if miss.candidates.is_empty() { + explanation.push_str("\nNo shared-word symbol candidates found; retry codegraph_explore with literal symbol/file names or code terms."); + } else { + explanation.push_str(&format!( + "\nCandidates to retry with codegraph_explore (shared words, not confirmed answers): {}", + capped_word_list(&miss.candidates, 350) + )); + } + explanation + } + + /// The lexical evidence behind [`Self::explore_miss_explanation`]: at most + /// 16 query words of at most 64 characters are checked against FTS + /// prefixes and live name segments; up to 12 retry candidates. + fn explore_miss_diagnostics(&self, query: &str) -> ExploreMissDiagnostics { + static WORD: OnceLock = OnceLock::new(); + let word_re = + WORD.get_or_init(|| regex::Regex::new(r"[\p{L}\p{N}]+").expect("explore miss word")); + let mut words: Vec = Vec::new(); + for word in word_re.find_iter(query) { + let word = word.as_str().to_lowercase(); + if !words.contains(&word) { + words.push(word); + } + } + let checked = words + .iter() + .filter(|word| word.chars().count() <= 64) + .take(16) + .cloned() + .collect::>(); + let limited = checked.len() != words.len(); + let mut diagnostics = ExploreMissDiagnostics { + limited, + ..ExploreMissDiagnostics::default() + }; + if checked.is_empty() { + return diagnostics; + } + let names = self + .store + .distinct_non_file_node_names() + .unwrap_or_default(); + let mut segment_hits: BTreeSet = BTreeSet::new(); + let mut candidates: Vec = Vec::new(); + for (name, _) in &names { + let segments = codegraph_graph::segments::split_identifier_segments(name); + let mut shares = false; + for segment in segments { + if checked.contains(&segment) { + segment_hits.insert(segment); + shares = true; + } + } + if shares && candidates.len() < 12 { + candidates.push(name.clone()); + } + } + for word in checked { + let matched = segment_hits.contains(&word) + || self.store.fts_any_column_has_prefix(&word).unwrap_or(false); + if matched { + diagnostics.matched.push(word); + } else { + diagnostics.unmatched.push(word); + } + } + let checked_words = diagnostics + .matched + .iter() + .chain(&diagnostics.unmatched) + .cloned() + .collect::>(); + for name in self + .store + .fts_name_prefix_names(&checked_words, 12) + .unwrap_or_default() + { + if !candidates.contains(&name) { + candidates.push(name); + } + } + candidates.truncate(12); + diagnostics.candidates = candidates; + diagnostics + } + /// Render one file's source section under the per-file budget. Small files /// come back WHOLE; larger ones are clustered by `gapThreshold` and capped /// at `maxCharsPerFile`, dropping whole low-importance clusters (never @@ -1490,6 +1777,8 @@ impl CodeGraphEngine { "#### {file_path} — {names} · ⚠ changed since last index sync; source below is the full current file\n\n```{lang}\n{numbered}\n```\n" ), source_emitted: true, + clipped: false, + elided: Vec::new(), }; } return ExploreFileRender { @@ -1497,6 +1786,9 @@ impl CodeGraphEngine { "#### {file_path} — ⚠ changed on disk after the last index sync — source omitted because indexed line ranges are unsafe and the current file exceeds the {FILE_MODE_MAX_LINES}-line whole-file cap. Read this file directly.\n" ), source_emitted: false, + clipped: false, + // Nothing it set out to deliver was sent (#2077). + elided: elided_wanted_spans(&subgraph.wanted_spans_in(file_path), &[]), }; } @@ -1521,20 +1813,26 @@ impl CodeGraphEngine { return ExploreFileRender { section, source_emitted: true, + clipped: false, + elided: Vec::new(), }; } - // (2) Unaffordable but owning NO focus ⇒ also today's behaviour: - // return whole and let the caller drop it whole, so "an incidental - // file that doesn't fit is DROPPED whole" stays literally true. - if ctx.focuses.is_empty() { + // (2) Unaffordable but owning NO focus and no exact target ⇒ also + // today's behaviour: return whole and let the caller drop it + // whole, so "an incidental file that doesn't fit is DROPPED + // whole" stays literally true. + if ctx.focuses.is_empty() && !subgraph.holds_exact(file_path) { return ExploreFileRender { section, source_emitted: true, + clipped: false, + elided: Vec::new(), }; } - // (3) Unaffordable AND owning a focus ⇒ fall through to the clustering - // path, which windows against `funded_headroom` and guarantees a - // window per focus. The ONLY new behaviour. + // (3) Unaffordable AND owning a focus or an exact target ⇒ fall + // through to the clustering path, which windows against + // `funded_headroom` and pays the named members first. The ONLY + // new behaviour. } // Cluster nearby symbol ranges; merge ranges within `gapThreshold` @@ -1563,7 +1861,9 @@ impl CodeGraphEngine { // matched on the STORED definition line: a focus whose line the // clamp below moves is one the range filters would drop anyway, // and §3.1.1 conditions the guarantee on surviving them. - let importance = if ctx.focuses.contains(&(n.start_line as usize)) { + let importance = if subgraph.is_exact(&n.id) { + EXACT_IMPORTANCE + } else if ctx.focuses.contains(&(n.start_line as usize)) { 11 } else if subgraph.roots.iter().any(|r| r == &n.id) { 10 @@ -1581,18 +1881,48 @@ impl CodeGraphEngine { start, end, label: format!("{}({})", n.name, n.kind.as_str()), + member: ElidedSymbol { + name: n.name.clone(), + kind: n.kind.as_str(), + start_line: start, + }, importance, + qualified_name: Some(n.qualified_name.clone()), } }) // Drop ranges whose start is past EOF (a fully-stale node). .filter(|r| r.start <= total_lines) .collect(); + // Line spans the query anchored in this file (`lines 900-1003`) are + // exact targets too, rendered as exactly the span asked for rather than + // as the symbols around it (#2063). + for (start, end) in subgraph.anchor_spans_in(file_path) { + if start > total_lines { + continue; + } + let end = end.min(total_lines); + let name = format!("lines {start}-{end}"); + ranges.push(ClusterRange { + start, + end, + label: format!("{name}(range)"), + member: ElidedSymbol { + name, + kind: "range", + start_line: start, + }, + importance: EXACT_IMPORTANCE, + qualified_name: None, + }); + } ranges.sort_by_key(|r| r.start); if ranges.is_empty() { return ExploreFileRender { section: String::new(), source_emitted: false, + clipped: false, + elided: Vec::new(), }; } @@ -1602,6 +1932,8 @@ impl CodeGraphEngine { if r.start <= current.end + budget.gap_threshold { current.end = current.end.max(r.end); current.symbols.push(r.label.clone()); + current.members.push(r.member.clone()); + current.spans.push((r.start, r.end, r.importance)); current.score += r.importance; current.max_importance = current.max_importance.max(r.importance); } else { @@ -1699,47 +2031,60 @@ impl CodeGraphEngine { // instead of being taken whole (which then costs the entire file) or // skipped whole. Returns the rendered body and whether it was windowed, // so the caller can add the `... (gap) ...` honesty marker. - let window_cluster = |c: &Cluster, room: usize, focuses: &[usize]| -> (String, bool) { - let (start_idx, end_idx) = span_of(c); - if range_cost(start_idx, end_idx) <= room { - return (render_range(start_idx, end_idx), false); - } - let head_room = if focuses.is_empty() { - room - } else { - room * 60 / 100 - }; - let mut windows: Vec<(usize, usize)> = vec![grow_head(start_idx, end_idx, head_room)]; - if !focuses.is_empty() { + let window_cluster = + |c: &Cluster, room: usize, focuses: &[usize]| -> (Vec<(usize, usize)>, bool) { + let (start_idx, end_idx) = span_of(c); + if range_cost(start_idx, end_idx) <= room { + return (vec![(start_idx, end_idx)], false); + } + // Upstream `windowToCeiling` (CG-38): the whole room goes to the + // head first, and 40% of it is held back for focus windows only + // when that head leaves a focus uncovered — then only the focuses + // the smaller head misses take a share of it. + let full = grow_head(start_idx, end_idx, room); + let covers = |w: (usize, usize), line: usize| line > w.0 && line <= w.1; + if focuses.iter().all(|&line| covers(full, line)) { + return (vec![full], full != (start_idx, end_idx)); + } + let head_room = room * 60 / 100; + let head = grow_head(start_idx, end_idx, head_room); + let pending: Vec = focuses + .iter() + .copied() + .filter(|&line| !covers(head, line)) + .collect(); + let mut windows: Vec<(usize, usize)> = vec![head]; // Even shares with carry-forward, never greedy: upstream measured // that a greedy split let an early focus eat the whole reserve and // re-lose the target. One integer division, so the same input // cannot drift by a char across platforms. let reserve = room - head_room; - let base = reserve / focuses.len(); - let mut carry = reserve - base * focuses.len(); - for &line in focuses { + let base = reserve / pending.len(); + let mut carry = reserve - base * pending.len(); + for &line in &pending { let allot = base + carry; let w = grow_around(line.saturating_sub(1), start_idx, end_idx, allot); carry = allot.saturating_sub(range_cost(w.0, w.1)); windows.push(w); } - } - windows.sort_unstable(); - let mut merged: Vec<(usize, usize)> = Vec::new(); - for w in windows { - match merged.last_mut() { - Some(last) if w.0 <= last.1 => last.1 = last.1.max(w.1), - _ => merged.push(w), + windows.sort_unstable(); + let mut merged: Vec<(usize, usize)> = Vec::new(); + for w in windows { + match merged.last_mut() { + Some(last) if w.0 <= last.1 => last.1 = last.1.max(w.1), + _ => merged.push(w), + } } - } - let covers_all = merged.len() == 1 && merged[0] == (start_idx, end_idx); - let body = merged + let covers_all = merged.len() == 1 && merged[0] == (start_idx, end_idx); + (merged, !covers_all) + }; + // What windows cost with bare gap markers, the price selection uses. + let bare_len = |windows: &[(usize, usize)]| -> usize { + windows .iter() - .map(|&(lo, hi)| render_range(lo, hi)) - .collect::>() - .join(GAP_MARKER); - (body, !covers_all) + .map(|&(lo, hi)| range_cost(lo, hi)) + .sum::() + + GAP_MARKER.len() * windows.len().saturating_sub(1) }; // Rank clusters: entry-point importance first, then density, then span @@ -1769,23 +2114,293 @@ impl CodeGraphEngine { + lang.len() + SECTION_FENCE_CHARS + SECTION_GAP_TAIL_CHARS; - let ceiling = ((budget.max_chars_per_file as f64 * 1.5).round() as usize) - .min(ctx.funded_headroom) - .saturating_sub(section_frame); + let file_cap = if ctx.lift_file_cap { + ctx.funded_headroom + } else { + ((budget.max_chars_per_file as f64 * 1.5).round() as usize).min(ctx.funded_headroom) + }; + let ceiling = file_cap.saturating_sub(section_frame); + // A member the query asked for — a root or a protected focus — rather + // than incidental context that merely sits within `gap_threshold` of one + // (#2062). + let is_protected = |importance: u32| importance >= 10; + // Member ranges re-merged in source order and padded into windows, so + // adjacent survivors read as one block. + let member_windows = |spans: &[(usize, usize)]| -> Vec<(usize, usize)> { + let mut sorted = spans.to_vec(); + sorted.sort_unstable(); + let mut merged: Vec<(usize, usize)> = Vec::new(); + for (start, end) in sorted { + match merged.last_mut() { + Some(last) if start <= last.1 + budget.gap_threshold => { + last.1 = last.1.max(end); + } + _ => merged.push((start, end)), + } + } + let mut windows: Vec<(usize, usize)> = Vec::new(); + for (start, end) in merged { + let hi = (end + CONTEXT_PADDING).min(total_lines); + let lo = start + .saturating_sub(1) + .saturating_sub(CONTEXT_PADDING) + .min(hi); + match windows.last_mut() { + Some(last) if lo <= last.1 => last.1 = last.1.max(hi), + _ => windows.push((lo, hi)), + } + } + windows + }; + // What a cluster's protected members cost on their own — the room a + // cluster ranked ABOVE it must leave for them. 0 without any. + let protected_core_cost = |c: &Cluster| -> usize { + let core = c + .spans + .iter() + .filter(|s| is_protected(s.2)) + .map(|s| (s.0, s.1)) + .collect::>(); + if core.is_empty() { + 0 + } else { + bare_len(&member_windows(&core)) + GAP_MARKER.len() + } + }; + // The cluster's protected members always, then its incidental members + // (most important, smallest, earliest first) while the windows still fit + // `cap`. `None` when every member fits, or when the protected core alone + // overruns `room` and the ordinary windowing has to cut it. + let shrink_to_members = + |c: &Cluster, room: usize, cap: usize| -> Option> { + if c.spans.len() < 2 { + return None; + } + let mut order = c.spans.clone(); + order.sort_by(|a, b| { + is_protected(b.2) + .cmp(&is_protected(a.2)) + .then(b.2.cmp(&a.2)) + .then((a.1 - a.0).cmp(&(b.1 - b.0))) + .then(a.0.cmp(&b.0)) + }); + let mut kept: Vec<(usize, usize)> = Vec::new(); + for span in &order { + let mut candidate = kept.clone(); + candidate.push((span.0, span.1)); + let cost = bare_len(&member_windows(&candidate)); + if kept.is_empty() || is_protected(span.2) { + if cost > room { + return None; + } + kept = candidate; + } else if cost <= cap.min(room) { + kept = candidate; + } + } + (kept.len() < c.spans.len()).then(|| member_windows(&kept)) + }; + // What the protected members of every cluster ranked below position `p` + // cost. Rank orders clusters, not members, and density counts every + // member, so a named function packed with helpers can outrank a named + // function alone and spend the file's room on the helpers (#2062). + let mut owed_from = vec![0usize; ranked.len() + 1]; + for p in (0..ranked.len()).rev() { + owed_from[p] = owed_from[p + 1] + protected_core_cost(&clusters[ranked[p]]); + } // Per-cluster render state lives in a Vec indexed parallel to `clusters`, // never in a HashSet that is then iterated: emission order must come from // the index, not from hash order, or the output stops being byte-stable. - let mut rendered_clusters: Vec> = vec![None; clusters.len()]; + let mut rendered_clusters: Vec>> = vec![None; clusters.len()]; let mut projected = 0usize; let mut chosen_count = 0usize; let mut any_cluster_windowed = false; - for &idx in &ranked { + // A cluster's exact members (#2063), re-merged and padded like any other. + let exact_spans = |c: &Cluster| -> Vec<(usize, usize)> { + c.spans + .iter() + .filter(|s| s.2 >= EXACT_IMPORTANCE) + .map(|s| (s.0, s.1)) + .collect() + }; + let exact_core_cost = |c: &Cluster| -> usize { + let exact = exact_spans(c); + if exact.is_empty() { + 0 + } else { + bare_len(&member_windows(&exact)) + } + }; + // Whether 1-based `line` falls inside one of the 0-based exclusive-end + // `windows`. + let covers = |windows: &[(usize, usize)], line: usize| -> bool { + windows.iter().any(|&(lo, hi)| line > lo && line <= hi) + }; + // What a window of an exact cluster's exact bodies must still reach when + // they do not fit whole: each exact member's first line, its body focus + // lines, and any named focus inside an exact body. Only lines INSIDE the + // exact bodies, as upstream windows only the parts it kept: a named + // member beside an oversize exact body is dropped and named in the + // header rather than splitting the room the body's own calls need. + let exact_focuses_in = |c: &Cluster| -> Vec { + let exact = exact_spans(c); + let inside = |line: usize| exact.iter().any(|&(lo, hi)| line >= lo && line <= hi); + let mut lines: Vec = exact.iter().map(|&(lo, _)| lo).collect(); + for (start, end, body) in ctx.exact_focus { + if *start >= c.start && (*end).min(total_lines) <= c.end { + lines.extend(body.iter().copied()); + } + } + lines.extend(focuses_in(c).into_iter().filter(|&line| inside(line))); + lines.retain(|&line| line >= 1 && line <= total_lines); + lines.sort_unstable(); + lines.dedup(); + lines + }; + // Upstream's exact branch (#2063). The exact body is kept whole when it + // fits `room`, and every other member joins while the rendered windows + // still fit: a protected one within `room`, an incidental one within + // `cap`. Priced on the rendered windows, so nothing above the exact body + // in source order can be paid first and cut its tail. When the exact + // body alone overruns `room` it is windowed from its own first line: + // the whole room goes to that head unless it would leave a focus line + // uncovered, and then 60% of it. Focus lines still uncovered get a + // window each from an even share of what is left, carried forward, and + // only when one of `MIN_WINDOW_LINES` fits its share. + let exact_render = |c: &Cluster, room: usize, cap: usize| -> (Vec<(usize, usize)>, bool) { + let (start_idx, end_idx) = span_of(c); + if range_cost(start_idx, end_idx) <= cap.min(room) { + return (vec![(start_idx, end_idx)], false); + } + let exact = exact_spans(c); + let core = member_windows(&exact); + let focuses = exact_focuses_in(c); + let mut windows = if bare_len(&core) <= room { + let mut order: Vec<(usize, usize, u32)> = c + .spans + .iter() + .filter(|s| s.2 < EXACT_IMPORTANCE) + .copied() + .collect(); + order.sort_by(|a, b| { + is_protected(b.2) + .cmp(&is_protected(a.2)) + .then(b.2.cmp(&a.2)) + .then((a.1 - a.0).cmp(&(b.1 - b.0))) + .then(a.0.cmp(&b.0)) + }); + let mut kept = exact; + for span in order { + let mut candidate = kept.clone(); + candidate.push((span.0, span.1)); + let limit = if is_protected(span.2) { + room + } else { + cap.min(room) + }; + if bare_len(&member_windows(&candidate)) <= limit { + kept = candidate; + } + } + member_windows(&kept) + } else { + let head_start = core[0].0; + let full = grow_head(head_start, end_idx, room); + let head_room = if focuses.iter().all(|&line| covers(&[full], line)) { + room + } else { + room * 60 / 100 + }; + vec![grow_head(head_start, end_idx, head_room)] + }; + let pending: Vec = focuses + .iter() + .copied() + .filter(|&line| !covers(&windows, line)) + .collect(); + let mut left = room.saturating_sub(bare_len(&windows)); + for (i, &line) in pending.iter().enumerate() { + if covers(&windows, line) { + continue; + } + let share = (left / (pending.len() - i)).saturating_sub(GAP_MARKER.len()); + if share == 0 { + continue; + } + let w = grow_around(line - 1, start_idx, end_idx, share); + if w.1 - w.0 < MIN_WINDOW_LINES.min(end_idx - start_idx) + || range_cost(w.0, w.1) > share + { + continue; + } + left = left.saturating_sub(range_cost(w.0, w.1) + GAP_MARKER.len()); + windows.push(w); + } + windows.sort_unstable(); + let mut merged: Vec<(usize, usize)> = Vec::new(); + for w in windows { + match merged.last_mut() { + Some(last) if w.0 <= last.1 => last.1 = last.1.max(w.1), + _ => merged.push(w), + } + } + let whole = merged.len() == 1 && merged[0] == (start_idx, end_idx); + (merged, !whole) + }; + // A cluster's incidental members may only use what is left once every + // lower-ranked cluster's protected members are paid for, and in a + // cluster holding a protected member they never take it past its room, + // so a window from the head no longer cuts a named body's tail. Its own + // protected members are never held back (#2062). + // An exact cluster renders by `exact_render`, with the same hold-back. + let guarded_render = + |c: &Cluster, room: usize, owed_below: usize| -> (Vec<(usize, usize)>, bool) { + if c.max_importance >= EXACT_IMPORTANCE { + let cap = room.saturating_sub(owed_below); + return exact_render(c, room, cap); + } + let has_protected = c.spans.iter().any(|s| is_protected(s.2)); + let cap = match (owed_below > 0, has_protected) { + (true, _) => Some(room.saturating_sub(owed_below)), + (false, true) => Some(room), + (false, false) => None, + }; + if let Some(cap) = cap { + let (start_idx, end_idx) = span_of(c); + if range_cost(start_idx, end_idx) > cap.min(room) + && let Some(windows) = shrink_to_members(c, room, cap) + { + return (windows, true); + } + } + window_cluster(c, room, &focuses_in(c)) + }; + // What the exact members of every exact cluster ranked below position + // `p` cost. Several exact targets can land in different clusters of one + // big file, and the one ranked first would spend the whole room on its + // neighbours; so an exact cluster holds that back, never cutting into its + // own exact members (#2063). + let mut owed_exact_from = vec![0usize; ranked.len() + 1]; + for p in (0..ranked.len()).rev() { + let cost = exact_core_cost(&clusters[ranked[p]]); + owed_exact_from[p] = + owed_exact_from[p + 1] + if cost > 0 { cost + GAP_MARKER.len() } else { 0 }; + } + let hold_back = |c: &Cluster, rank: usize, room: usize| -> usize { + let owed = owed_exact_from[rank + 1]; + if c.max_importance < EXACT_IMPORTANCE || owed == 0 { + return room; + } + room.min(exact_core_cost(c).max(room.saturating_sub(owed))) + }; + for (rank, &idx) in ranked.iter().enumerate() { + let owed_below = owed_from[rank + 1]; if chosen_count == 0 { - let (body, windowed) = - window_cluster(&clusters[idx], ceiling, &focuses_in(&clusters[idx])); - projected += body.len(); + let room = hold_back(&clusters[idx], rank, ceiling); + let (windows, windowed) = guarded_render(&clusters[idx], room, owed_below); + projected += bare_len(&windows); any_cluster_windowed |= windowed; - rendered_clusters[idx] = Some(body); + rendered_clusters[idx] = Some(windows); chosen_count += 1; continue; } @@ -1796,7 +2411,11 @@ impl CodeGraphEngine { // first cluster's body legitimately exceed `file_budget`, so a plain // subtraction wraps to a huge `room` and hands this cluster unlimited // budget, the exact opposite of the intent. - let room = ceiling.saturating_sub(projected + GAP_MARKER.len()); + let room = hold_back( + &clusters[idx], + rank, + ceiling.saturating_sub(projected + GAP_MARKER.len()), + ); // The floor's own cost IS `MIN_CHARS` — the smallest useful window — // rather than a new magic number. When even that cannot be paid for // the cluster is still skipped, so shrinking can never overspend into @@ -1806,34 +2425,81 @@ impl CodeGraphEngine { if room < range_cost(start_idx, floor_end) { continue; } - let (body, windowed) = - window_cluster(&clusters[idx], room, &focuses_in(&clusters[idx])); + let (windows, windowed) = guarded_render(&clusters[idx], room, owed_below); any_cluster_windowed |= windowed; - projected += body.len() + GAP_MARKER.len(); - rendered_clusters[idx] = Some(body); + projected += bare_len(&windows) + GAP_MARKER.len(); + rendered_clusters[idx] = Some(windows); chosen_count += 1; } - let mut file_section = String::new(); + // Emit chosen clusters in source order, each window a part named by its + // 1-based line span. + let mut parts: Vec<(usize, usize, String)> = Vec::new(); let mut symbols: Vec = Vec::new(); for (i, cluster) in clusters.iter().enumerate() { - let Some(body) = rendered_clusters[i].as_ref() else { + let Some(windows) = rendered_clusters[i].as_ref() else { continue; }; - if !file_section.is_empty() { - file_section.push_str(GAP_MARKER); - } - file_section.push_str(body); + parts.extend( + windows + .iter() + .map(|&(lo, hi)| (lo + 1, hi, render_range(lo, hi))), + ); symbols.extend(cluster.symbols.iter().cloned()); } - if chosen_count < clusters.len() || any_cluster_windowed { + // Gap names (#1711) are paid from what selection left of this file's + // budget, never from source: selection measured every gap as bare. + let bare_text = parts.iter().map(|(_, _, text)| text.len()).sum::() + + GAP_MARKER.len() * parts.len().saturating_sub(1); + let spare = ceiling.max(projected).saturating_sub(bare_text); + let file_index_nodes = self.store.nodes_by_file_path(file_path).unwrap_or_default(); + let mut file_section = + join_parts_with_named_gaps(file_path, &parts, &file_index_nodes, spare); + let clipped = chosen_count < clusters.len() || any_cluster_windowed; + if clipped { file_section.push_str("\n\n... (gap) ..."); } - let header = explore_file_header(file_path, &symbols, budget.max_symbols_in_file_header); + // Every member across ALL clusters is a candidate for the header bias, + // so a dropped cluster's symbols no longer vanish from it (#1711). + let mut elided: Vec = Vec::new(); + for member in clusters.iter().flat_map(|c| &c.members) { + let covered = parts + .iter() + .any(|(first, last, _)| member.start_line >= *first && member.start_line <= *last); + if !covered && !elided.iter().any(|e| e.name == member.name) { + elided.push(member.clone()); + } + } + elided.sort_by_key(|s| s.start_line); + let header = explore_file_header( + file_path, + &symbols, + &elided, + budget.max_symbols_in_file_header, + ); + // Every member of every cluster, chosen or not: a dropped cluster, a + // shrunk one and a windowed body all leave a member short (#2077). + let wanted: Vec = ranges + .iter() + .map(|r| WantedSpan { + name: r.member.name.clone(), + kind: r.member.kind, + start: r.start, + end: r.end, + importance: r.importance, + qualified_name: r.qualified_name.clone(), + }) + .collect(); + let delivered: Vec<(usize, usize)> = parts + .iter() + .map(|(first, last, _)| (*first, *last)) + .collect(); ExploreFileRender { section: format!("{header}\n\n```{lang}\n{file_section}\n```\n"), source_emitted: true, + clipped, + elided: elided_wanted_spans(&wanted, &delivered), } } @@ -1842,9 +2508,17 @@ impl CodeGraphEngine { const ROOT_CAP: usize = 5; const DIRECT_CALLER_FILE_CAP: usize = 4; let traverser = GraphTraverser::new(&self.store); - let roots: Vec<&Node> = subgraph - .roots - .iter() + // Exact targets lead (upstream #2063): the roots are whatever search + // ranked first for the bare name, so without this a query for + // `SQLCompiler.as_sql` headlined a same-named override. + let mut ids: Vec<&String> = Vec::new(); + for id in subgraph.exact_ids.iter().chain(&subgraph.roots) { + if !ids.contains(&id) { + ids.push(id); + } + } + let roots: Vec<&Node> = ids + .into_iter() .filter_map(|id| subgraph.node(id)) .filter(|n| is_meaningful_kind(n.kind)) .take(ROOT_CAP) @@ -2306,48 +2980,195 @@ impl CodeGraphEngine { Ok(results.into_iter().map(|r| r.node).collect()) } - /// Build the explore subgraph. Ports the deterministic spine of - /// `findRelevantContext` (`context/index.ts:900-940`): the entry points - /// (`roots`) are the FTS search results for the query (`searchLimit: 8`, - /// `tools.ts:1597-1602`), in search-rank order. We then pull each root's - /// callers/callees + `contains` children into the subgraph so the blast - /// radius, relationship map, and source section have content. The RWR - /// relevance re-ranking / file gating is NOT ported (see KNOWN_DIFFS.md). - #[cfg(test)] - fn find_relevant_context(&self, query: &str) -> anyhow::Result { - self.find_relevant_context_with_pins(query, &[]) + /// Which indexed files define a symbol spelled like this query token + /// (upstream `filesDefiningSymbol`, #2071): exact names only, and the shared + /// matcher for a qualified token, so it agrees with how explore resolves + /// named symbols. A node that names a thing without defining it in its file + /// (a file, import, export or parameter) does not count. A lookup error + /// answers "none", which leaves the span's pins as they were. + fn files_defining_symbol(&self, symbol: &str) -> Vec { + let qualified = symbol.contains(['.', '/']) || symbol.contains("::"); + let name = if qualified { + last_qualifier_part(symbol).unwrap_or(symbol) + } else { + symbol + }; + let Ok(nodes) = self.store.nodes_by_name(name) else { + return Vec::new(); + }; + nodes + .into_iter() + .filter(|n| !qualified || matches_symbol(n, symbol)) + .filter(|n| { + !matches!( + n.kind, + NodeKind::File | NodeKind::Import | NodeKind::Export | NodeKind::Parameter + ) + }) + .map(|n| n.file_path) + .collect() } - fn find_relevant_context_with_pins( - &self, - query: &str, - pinned_files: &[String], - ) -> anyhow::Result { - let mut sub = ExploreSubgraph::default(); - let traverser = GraphTraverser::new(&self.store); - - let mut seed_ids: Vec = Vec::new(); - let results = if query.trim().is_empty() { - Vec::new() - } else { - let seed_names = - get_segment_matches(&self.store, &extract_segment_search_words(query), 8) + /// The query's EXACT targets (upstream #2063). A multi-line anchor names a + /// span, not a symbol — often the tail of a body a previous response + /// windowed — so it is kept as that span; a single-line anchor resolves to + /// the innermost callable containing it. A qualified token is exact when + /// its non-test definitions number at most three, and then only its + /// callables are. Lookups that fail are skipped: exact targets must never + /// fail an explore call. + /// + /// Test paths are judged by the shared `is_test_file`, which differs from + /// upstream's seeder regex only at its margins (it also counts `e2e/` and + /// `_test.`, but not `fixtures/` or `mocks/`). + fn exact_targets(&self, match_query: &str, anchors: &[QueryLineAnchor]) -> ExactTargets { + let mut exact = ExactTargets::default(); + for anchor in anchors { + let Ok(file_nodes) = self.store.nodes_by_file_path(&anchor.file) else { + continue; + }; + if anchor.start == anchor.end { + let line = anchor.start as i64; + let enclosing = file_nodes .into_iter() - .map(|matched| matched.name) - .collect(); - search_nodes( - &self.store, - query, - &SearchOptions { - limit: Some(8), - seed_names, - deprioritize: Some(Arc::clone(&self.deprioritize)), - ..Default::default() - }, - &self.project_name_tokens(), - )? - }; - for r in results { + .filter(|n| { + is_exact_target_kind(n.kind) && n.start_line <= line && n.end_line >= line + }) + .min_by_key(|n| { + ( + n.end_line - n.start_line, + n.start_line, + n.start_column, + n.id.clone(), + ) + }); + if let Some(node) = enclosing { + exact.push_node(node); + continue; + } + } + let (start, end) = if anchor.start == anchor.end { + ( + anchor.start.saturating_sub(ANCHOR_LINE_CONTEXT).max(1), + anchor.start + ANCHOR_LINE_CONTEXT, + ) + } else { + (anchor.start, anchor.end) + }; + exact.spans.push((anchor.file.clone(), start, end)); + } + for token in exact_symbol_tokens(match_query) { + if !(token.contains('.') || token.contains("::")) { + continue; + } + let Ok(all) = self.find_all_symbols(&token) else { + continue; + }; + let candidates: Vec = all + .nodes + .into_iter() + .filter(|n| is_seedable_kind(n.kind) && !is_test_file(&n.file_path)) + .collect(); + if candidates.len() > 3 { + continue; + } + for node in candidates { + if is_exact_target_kind(node.kind) { + exact.push_node(node); + } + } + } + exact + } + + /// Per exact target in `file_path`, `(start, end, focus lines)`: the lines a + /// window of its body must reach when it does not fit whole — an anchored + /// line inside it, then every line where it uses another symbol the + /// question is about (upstream `bodyFocusLines`), capped. + fn exact_body_focus( + &self, + subgraph: &ExploreSubgraph, + file_path: &str, + anchors: &[QueryLineAnchor], + question_ids: &HashSet, + ) -> Vec { + let mut out = Vec::new(); + for id in &subgraph.exact_ids { + let Some(n) = subgraph.node(id) else { + continue; + }; + if n.file_path != file_path || n.start_line < 1 || n.end_line < n.start_line { + continue; + } + let (start, end) = (n.start_line as usize, n.end_line as usize); + let mut lines: Vec = anchors + .iter() + .filter(|a| a.file == n.file_path && a.start > start && a.start <= end) + .map(|a| a.start) + .collect(); + let mut uses: Vec = self + .store + .edges_by_source_kind(&n.id, None) + .unwrap_or_default() + .into_iter() + .filter(|e| e.target != n.id && question_ids.contains(&e.target)) + .filter_map(|e| e.line) + .filter(|&line| line > start as i64 && line <= end as i64) + .map(|line| line as usize) + .collect(); + uses.sort_unstable(); + lines.extend(uses); + let mut seen = HashSet::new(); + lines.retain(|line| seen.insert(*line)); + lines.truncate(MAX_BODY_FOCUS_LINES); + out.push((start, end, lines)); + } + out.sort_unstable(); + out + } + + /// Build the explore subgraph. Ports the deterministic spine of + /// `findRelevantContext` (`context/index.ts:900-940`): the entry points + /// (`roots`) are the FTS search results for the query (`searchLimit: 8`, + /// `tools.ts:1597-1602`), in search-rank order. We then pull each root's + /// callers/callees + `contains` children into the subgraph so the blast + /// radius, relationship map, and source section have content. The RWR + /// relevance re-ranking / file gating is NOT ported (see KNOWN_DIFFS.md). + #[cfg(test)] + fn find_relevant_context(&self, query: &str) -> anyhow::Result { + self.find_relevant_context_with_pins(query, &[], &ExactTargets::default()) + } + + fn find_relevant_context_with_pins( + &self, + query: &str, + pinned_files: &[String], + exact: &ExactTargets, + ) -> anyhow::Result { + let mut sub = ExploreSubgraph::default(); + let traverser = GraphTraverser::new(&self.store); + + let mut seed_ids: Vec = Vec::new(); + let results = if query.trim().is_empty() { + Vec::new() + } else { + let seed_names = + get_segment_matches(&self.store, &extract_segment_search_words(query), 8) + .into_iter() + .map(|matched| matched.name) + .collect(); + search_nodes( + &self.store, + query, + &SearchOptions { + limit: Some(8), + seed_names, + deprioritize: Some(Arc::clone(&self.deprioritize)), + ..Default::default() + }, + &self.project_name_tokens(), + )? + }; + for r in results { if sub.insert(r.node.clone()) { seed_ids.push(r.node.id.clone()); sub.roots.push(r.node.id.clone()); @@ -2371,6 +3192,12 @@ impl CodeGraphEngine { sub.insert(node); } } + for node in &exact.nodes { + sub.exact_ids.push(node.id.clone()); + sub.exact_files.insert(node.file_path.clone()); + sub.insert(node.clone()); + } + sub.anchor_spans = exact.spans.clone(); for root_id in seed_ids.clone() { for c in traverser.get_callers(&root_id, CALL_DEPTH)? { @@ -2919,6 +3746,13 @@ struct SourceProbe { struct ExploreFileRender { section: String, source_emitted: bool, + /// The per-file ceiling cut something cluster selection wanted: a cluster + /// was dropped or windowed. Only such a render can grow when a named file + /// is given the response's unspent budget (#2068). + clipped: bool, + /// The symbols this section set out to deliver and did not, measured from + /// the ranges it sent (#2077). Empty for a complete section. + elided: Vec, } impl SourceProbe { @@ -3034,6 +3868,15 @@ struct ExploreSubgraph { /// Project-configured ranking-only paths. They stay in the subgraph and /// source output, but lose same-tier ordering ties to first-party files. deprioritized_files: HashSet, + /// EXACT targets (upstream #2063), in discovery order: they lead the blast + /// radius and outrank every other member of their file's clusters. + exact_ids: Vec, + /// Files holding an exact target. Ranked with the rescued files, the + /// port's named-file tier, so the answer the agent singled out is funded + /// before incidental files. + exact_files: HashSet, + /// Anchored line spans to render as written, `(file, start, end)`. + anchor_spans: Vec<(String, usize, usize)>, } impl ExploreSubgraph { @@ -3050,6 +3893,66 @@ impl ExploreSubgraph { self.pinned_files.iter().any(|path| path == file_path) } + fn is_exact(&self, id: &str) -> bool { + self.exact_ids.iter().any(|exact| exact == id) + } + + /// Whether `file_path` holds an exact target or an anchored span. + fn holds_exact(&self, file_path: &str) -> bool { + self.exact_files.contains(file_path) + || self + .anchor_spans + .iter() + .any(|(file, _, _)| file == file_path) + } + + /// The nodes of `file_path` as the spans a section of it sets out to + /// deliver (#2077): no file node — no section delivers "the file" as a + /// symbol — and no import or export. + fn wanted_spans_in(&self, file_path: &str) -> Vec { + self.nodes + .iter() + .filter(|n| { + n.file_path == file_path + && !matches!(n.kind, NodeKind::File | NodeKind::Import | NodeKind::Export) + && n.start_line > 0 + }) + .map(|n| WantedSpan { + name: n.name.clone(), + kind: n.kind.as_str(), + start: n.start_line as usize, + end: n.end_line.max(n.start_line) as usize, + importance: if self.is_exact(&n.id) { + EXACT_IMPORTANCE + } else if self.roots.iter().any(|r| r == &n.id) { + 10 + } else { + 1 + }, + qualified_name: Some(n.qualified_name.clone()), + }) + .collect() + } + + fn anchor_spans_in(&self, file_path: &str) -> Vec<(usize, usize)> { + self.anchor_spans + .iter() + .filter(|(file, _, _)| file == file_path) + .map(|&(_, start, end)| (start, end)) + .collect() + } + + /// The symbols the question is about: its exact targets and the roots a + /// shape-precise query token names — the port's analog of upstream's + /// `questionIds` (exact ∪ named ∪ spine), since it has no flow spine. + fn question_ids(&self, precise: &[String]) -> HashSet { + let named = self.roots.iter().filter(|id| { + self.node(id) + .is_some_and(|n| precise.iter().any(|t| t.eq_ignore_ascii_case(&n.name))) + }); + self.exact_ids.iter().chain(named).cloned().collect() + } + fn insert(&mut self, node: Node) -> bool { if self.index.contains_key(&node.id) { return false; @@ -3113,8 +4016,12 @@ impl ExploreSubgraph { // A rescued change-surface file (#1064) is the lexically- // dissimilar answer — give it the TOP tier so it outranks // incidental roots that merely share query words and survives - // the output file budget. - let tier = if self.rescued_files.contains(fp.as_str()) { + // the output file budget. A file holding an exact target + // (#2063) is the answer by the same argument, and upstream sorts + // the two into one named-file tier. + let tier = if self.rescued_files.contains(fp.as_str()) + || self.exact_files.contains(fp.as_str()) + { 3 } else if root_files.contains(fp.as_str()) { 2 @@ -3278,6 +4185,10 @@ impl ExploreSubgraph { } } +/// An exact target's `(start, end, focus lines)`: where its body sits and the +/// lines a window of it must still reach (#2063). +type BodyFocus = (usize, usize, Vec); + /// Per-file render state for one iteration of `handle_explore`'s file loop. /// /// These cannot live in [`ExploreOutputBudget`]: that struct is `Copy` and built @@ -3292,8 +4203,18 @@ struct RenderCtx<'a> { funded_headroom: usize, /// Definition lines of this file's protected focuses (§3.1): roots whose name /// a shape-precise query token names. Every focus line is guaranteed to land - /// inside some emitted window. + /// inside some emitted window — except in a cluster holding an exact target + /// (#2063), which pays the exact body first: a focus outside it that no + /// longer fits whole is dropped and named in the file header instead, as + /// upstream ranks exact above named. focuses: &'a [usize], + /// This file's exact targets as `(start, end, body focus lines)`, from + /// [`CodeGraphEngine::exact_body_focus`]. + exact_focus: &'a [BodyFocus], + /// Bound the section by `funded_headroom` alone, not also by the per-file + /// ceiling: set only when a named file is re-rendered into the budget the + /// rest of the response left unspent (#2068). + lift_file_cap: bool, drifted: bool, } @@ -3308,12 +4229,37 @@ fn decimal_width(n: usize) -> usize { width } +/// What an explore query singled out with no ambiguity left (upstream #2063): +/// a qualified name that resolves to at most three callables +/// (`SQLCompiler.as_sql`, not the other `as_sql`s), the callable enclosing a +/// single-line anchor (`compiler.py:776`), or a line span the query anchored +/// that no callable answers (`compiler.py lines 900-1003`). +#[derive(Debug, Default)] +struct ExactTargets { + /// Exact callables in discovery order: anchors first, then qualified names. + nodes: Vec, + /// Anchored spans, `(file, start, end)`, 1-based and inclusive. A + /// single-line anchor no callable encloses becomes the lines around it. + spans: Vec<(String, usize, usize)>, +} + +impl ExactTargets { + fn push_node(&mut self, node: Node) { + if !self.nodes.iter().any(|known| known.id == node.id) { + self.nodes.push(node); + } + } +} + /// Cluster source range used when sizing a god-file (`tools.ts:2708-2718`). struct ClusterRange { start: usize, end: usize, label: String, + member: ElidedSymbol, importance: u32, + /// The node's qualified name; `None` for an anchored line span. + qualified_name: Option, } /// A merged run of adjacent ranges (`tools.ts:2744-2771`). @@ -3321,6 +4267,13 @@ struct Cluster { start: usize, end: usize, symbols: Vec, + /// Every member's name, kind and definition line, so the header can name + /// the ones a trim cut (#1711). + members: Vec, + /// Every member's `(start, end, importance)` source range, 1-based, so a + /// cluster can render its protected members without its incidental ones + /// (#2062). + spans: Vec<(usize, usize, u32)>, score: u32, max_importance: u32, } @@ -3331,12 +4284,138 @@ impl Cluster { start: r.start, end: r.end, symbols: vec![r.label.clone()], + members: vec![r.member.clone()], + spans: vec![(r.start, r.end, r.importance)], score: r.importance, max_importance: r.importance, } } } +/// What an empty explore's diagnostics found (upstream #1904). +#[derive(Debug, Default)] +struct ExploreMissDiagnostics { + matched: Vec, + unmatched: Vec, + candidates: Vec, + limited: bool, +} + +/// Backticked words joined by `, `, keeping whole words up to `cap` (UTF-16 +/// units, as upstream measures) and marking a cut with ` …`. +fn capped_word_list(words: &[String], cap: usize) -> String { + let list = |words: &[String]| { + words + .iter() + .map(|word| format!("`{word}`")) + .collect::>() + .join(", ") + }; + let mut kept = 0; + while kept < words.len() && list(&words[..=kept]).encode_utf16().count() <= cap { + kept += 1; + } + let mut out = list(&words[..kept]); + if kept < words.len() { + out.push_str(" …"); + } + out +} + +/// An indexed symbol a trim left out of a rendered file (#1711). +#[derive(Debug, Clone, PartialEq, Eq)] +struct ElidedSymbol { + name: String, + kind: &'static str, + start_line: usize, +} + +/// How many elided symbols a gap marker names (#1711): enough for a follow-up +/// explore to have a target, few enough that the marker never rivals the +/// source it points at. +const ELIDED_SYMBOL_CAP: usize = 6; + +const BARE_GAP_MARKER: &str = "\n\n... (gap) ...\n\n"; + +/// Indexed symbols whose definition starts strictly between two rendered spans +/// (1-based lines): the hole a trim dropped (#1711). +fn symbols_between_ranges(nodes: &[Node], from_end: usize, to_start: usize) -> Vec { + if to_start <= from_end + 1 { + return Vec::new(); + } + let mut out: Vec = Vec::new(); + for node in nodes { + if matches!(node.kind, NodeKind::Import | NodeKind::Export) { + continue; + } + let start = usize::try_from(node.start_line).unwrap_or(0); + if start <= from_end || start >= to_start || out.iter().any(|s| s.name == node.name) { + continue; + } + out.push(ElidedSymbol { + name: node.name.clone(), + kind: node.kind.as_str(), + start_line: start, + }); + } + out.sort_by_key(|s| s.start_line); + out +} + +/// The gap marker between two non-contiguous slices of one file. A bare +/// `... (gap) ...` said something was missing but not WHAT, while the trim +/// note asked for exact names it never gave (#1711); a hole holding indexed +/// symbols names them as `name (file:line)`. +fn format_gap_marker(file_path: &str, elided: &[ElidedSymbol]) -> String { + if elided.is_empty() { + return BARE_GAP_MARKER.to_string(); + } + let shown = elided + .iter() + .take(ELIDED_SYMBOL_CAP) + .map(|s| format!("{} ({file_path}:{})", s.name, s.start_line)) + .collect::>() + .join(", "); + let more = elided.len().saturating_sub(ELIDED_SYMBOL_CAP); + let more = if more > 0 { + format!(", +{more} more") + } else { + String::new() + }; + format!("\n\n... (gap: {shown}{more}) ...\n\n") +} + +/// Join rendered parts — `(first_line, last_line, text)`, 1-based — with gap +/// markers that name what the trim skipped. `spare` is what naming may cost +/// beyond bare markers: selection priced every join as bare, so a gap the +/// spare cannot cover stays bare rather than pushing source out (#2057). +fn join_parts_with_named_gaps( + file_path: &str, + parts: &[(usize, usize, String)], + nodes: &[Node], + mut spare: usize, +) -> String { + let Some((_, _, first)) = parts.first() else { + return String::new(); + }; + let mut out = first.clone(); + for pair in parts.windows(2) { + let named = format_gap_marker( + file_path, + &symbols_between_ranges(nodes, pair[0].1, pair[1].0), + ); + let extra = named.len() - BARE_GAP_MARKER.len(); + if extra <= spare { + out.push_str(&named); + spare -= extra; + } else { + out.push_str(BARE_GAP_MARKER); + } + out.push_str(&pair[1].2); + } + out +} + // === Free-function renderers (1:1 with upstream helpers) ==================== /// `formatSearchResults` (`tools.ts:3324-3338`). @@ -3969,6 +5048,61 @@ fn is_handler_method_name(name: &str) -> bool { ) } +/// Kinds an exact target can be (upstream `ANCHOR_CALLABLE_KINDS` / the +/// seeder's `CALLABLE`): the innermost one containing an anchored line is the +/// symbol the agent points at. A class is deliberately not one — it spans most +/// of its file, and "the whole class" is not what a line number asks for. +/// `constructor` has no Rust NodeKind (ctors are `method`). +fn is_exact_target_kind(kind: NodeKind) -> bool { + matches!( + kind, + NodeKind::Method | NodeKind::Function | NodeKind::Component + ) +} + +/// Upstream's seeder `SEEDABLE`: what a named token resolves to, counted when +/// deciding whether a qualified name is specific enough to be exact. +fn is_seedable_kind(kind: NodeKind) -> bool { + is_exact_target_kind(kind) || matches!(kind, NodeKind::Variable | NodeKind::Constant) +} + +/// File extensions the seeder strips from a token (upstream `FILE_EXT`), so +/// `compiler.py` reads as `compiler`. +static SEEDER_FILE_EXT: LazyLock = LazyLock::new(|| { + regex::Regex::new( + r"(?i)\.(?:java|kt|kts|ts|tsx|js|jsx|mjs|cjs|cs|py|go|rb|php|swift|rs|cpp|cc|cxx|c|h|hpp|scala|lua|dart|vue|svelte|astro|erl|hrl)$", + ) + .expect("seeder file-extension regex is valid") +}); + +/// A token shaped like a symbol name (upstream's seeder test), ASCII only as in +/// JavaScript's `\w`: `get_select`, `SQLCompiler.as_sql`, `Engine::ServeHTTP`. +static SEEDER_SYMBOL_TOKEN: LazyLock = LazyLock::new(|| { + regex::Regex::new(r"^[A-Za-z_$][A-Za-z0-9_$]*(?:(?:::|\.)[A-Za-z0-9_$]+)*$") + .expect("seeder symbol-token regex is valid") +}); + +/// The query's symbol-shaped tokens, split as upstream's named-symbol seeder +/// splits them, in order, deduped and capped. +fn exact_symbol_tokens(query: &str) -> Vec { + let mut out: Vec = Vec::new(); + for raw in query.split(|c: char| c.is_whitespace() || matches!(c, ',' | '(' | ')' | '[' | ']')) + { + let token = SEEDER_FILE_EXT.replace(raw, ""); + let token = token.trim(); + if token.len() < 3 || !SEEDER_SYMBOL_TOKEN.is_match(token) { + continue; + } + if !out.iter().any(|seen| seen == token) { + out.push(token.to_string()); + } + if out.len() >= MAX_EXACT_SYMBOL_TOKENS { + break; + } + } + out +} + fn is_container(kind: NodeKind) -> bool { matches!( kind, @@ -4119,9 +5253,15 @@ fn cap_header_names(names: &[String], cap: usize) -> String { format!("{shown}, +{} more", names.len() - cap) } -/// Build a clustered file's `#### path — symbols` header, ranking symbols by -/// frequency and capping at `cap` (`tools.ts:2859-2876`). -fn explore_file_header(file_path: &str, symbols: &[String], cap: usize) -> String { +/// A clustered file's header, preferring the symbols the trim elided so +/// `+N more` is less likely to hide the answer (#1711): elided first (in +/// source order), then frequency, then name. +fn explore_file_header( + file_path: &str, + symbols: &[String], + elided: &[ElidedSymbol], + cap: usize, +) -> String { let mut counts: Vec<(String, usize)> = Vec::new(); for s in symbols { if let Some(slot) = counts.iter_mut().find(|(name, _)| name == s) { @@ -4130,7 +5270,35 @@ fn explore_file_header(file_path: &str, symbols: &[String], cap: usize) -> Strin counts.push((s.clone(), 1)); } } - counts.sort_by_key(|b| std::cmp::Reverse(b.1)); + let elided_labels = elided + .iter() + .map(|s| format!("{}({})", s.name, s.kind)) + .collect::>(); + for label in &elided_labels { + if !counts.iter().any(|(name, _)| name == label) { + counts.push((label.clone(), 1)); + } + } + let rank_of = |label: &str| -> Option { + let name = label.split_once('(').map_or(label, |(name, _)| name); + elided_labels + .iter() + .position(|l| l == label) + .or_else(|| elided.iter().position(|s| s.name == name)) + }; + counts.sort_by(|(a, count_a), (b, count_b)| { + let (rank_a, rank_b) = (rank_of(a), rank_of(b)); + rank_b + .is_some() + .cmp(&rank_a.is_some()) + .then(count_b.cmp(count_a)) + .then( + rank_a + .unwrap_or(usize::MAX) + .cmp(&rank_b.unwrap_or(usize::MAX)), + ) + .then(a.cmp(b)) + }); let ranked: Vec = counts.into_iter().map(|(name, _)| name).collect(); format!("#### {file_path} — {}", cap_header_names(&ranked, cap)) } @@ -4243,14 +5411,245 @@ const COMPLETENESS_RULE: &str = "---"; /// The trim note emitted at tiers that gate the completeness signal OFF. It is /// only reached when a file was actually trimmed, but `any_file_trimmed` is /// loop-determined, so the pre-loop reserve has to assume it. -const TRIMMED_NOTE: &str = "> Some file sections were trimmed for size. For a specific symbol you still need, run another `codegraph_explore` (or `codegraph_node`) with its exact name — line-numbered source, cheaper and more complete than Read."; +const TRIMMED_NOTE: &str = "> Some file sections were trimmed for size. Elided symbols are named inside gap markers as `name (file:line)` and preferred in the file header — run another `codegraph_explore` (or `codegraph_node`) with those exact names for their source."; + +/// One symbol a file section set out to deliver: a cluster member, or a node +/// of a section that renders no source (#2077). Completeness is judged against +/// these, not against the file — explore never promises whole files, only the +/// symbols it selected for each one. +#[derive(Debug, Clone, PartialEq, Eq)] +struct WantedSpan { + name: String, + kind: &'static str, + start: usize, + end: usize, + importance: u32, + /// The indexed qualified name (`SQLCompiler::as_sql`), when the span is a node. + qualified_name: Option, +} + +/// The wanted spans a section did NOT deliver in full: some line of the span +/// is outside every range it sent (upstream `elidedWantedSpans`, #2077). +/// Derived from the emitted ranges rather than from a flag each trim site has +/// to remember to set, so a member shrink, a ceiling window and a dropped +/// cluster all show up. Most relevant first, then source order. +fn elided_wanted_spans(wanted: &[WantedSpan], delivered: &[(usize, usize)]) -> Vec { + let mut sorted = delivered.to_vec(); + sorted.sort_unstable(); + let mut merged: Vec<(usize, usize)> = Vec::new(); + for (start, end) in sorted { + match merged.last_mut() { + Some(last) if start <= last.1 + 1 => last.1 = last.1.max(end), + _ => merged.push((start, end)), + } + } + let mut out: Vec = wanted + .iter() + .filter(|w| { + w.start > 0 + && w.end >= w.start + && !merged + .iter() + .any(|&(start, end)| start <= w.start && w.end <= end) + }) + .cloned() + .collect(); + out.sort_by(|a, b| b.importance.cmp(&a.importance).then(a.start.cmp(&b.start))); + out +} + +/// The shortest trailing slice of each path that no other path ends with: +/// `extHostExtensionService.ts` alone when it is the only one, +/// `node/extHostExtensionService.ts` beside `common/extHostExtensionService.ts`. +fn shortest_unique_suffixes(paths: &[String]) -> HashMap { + let mut all: Vec<&String> = paths.iter().collect(); + all.sort_unstable(); + all.dedup(); + let mut out = HashMap::new(); + for path in &all { + let segments: Vec<&str> = path.split('/').collect(); + let mut n = 1; + while n < segments.len() { + let suffix = segments[segments.len() - n..].join("/"); + let taken = all.iter().any(|other| { + other != path && (**other == suffix || other.ends_with(&format!("/{suffix}"))) + }); + if !taken { + break; + } + n += 1; + } + out.insert((*path).clone(), segments[segments.len() - n..].join("/")); + } + out +} + +/// Trimmed files the completeness note names one by one; the rest are a count. +const TRIMMED_FILES_NAMED: usize = 3; +/// Elided symbols the completeness note names: the ones the query named. +const TRIMMED_SYMBOLS_NAMED: usize = 4; + +/// The name the completeness note offers for an elided symbol: `Owner.member` +/// for a method, so an overloaded name resolves to the definition that was cut +/// (django has 110 `as_sql`s). The bare name otherwise. +fn follow_up_name(span: &WantedSpan) -> String { + let owner = span + .qualified_name + .as_deref() + .filter(|_| span.kind == NodeKind::Method.as_str()) + .and_then(|qualified| { + let segments: Vec<&str> = qualified.split("::").collect(); + (segments.len() >= 2).then(|| segments[segments.len() - 2]) + }) + .filter(|owner| { + owner.starts_with(|c: char| c.is_ascii_alphabetic() || c == '_' || c == '$') + && owner + .chars() + .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '$') + }); + match owner { + Some(owner) => format!("{owner}.{}", span.name), + None => span.name.clone(), + } +} + +/// How the completeness note counts the files it vouches for: never "0 files" +/// beside a note about what was shown. +fn files_wording(files_included: usize) -> String { + match files_included { + 0 => "these files".to_string(), + 1 => "1 file".to_string(), + n => format!("{n} files"), + } +} + +fn complete_source_note(files: &str) -> String { + format!( + "> **Complete source for {files} is included above — do NOT re-read them.** If your question also needs files/symbols listed under \"Not shown above\" (or any area this call didn't cover), make ANOTHER codegraph_explore targeting those names — it returns the same source with line numbers and is cheaper and more complete than reading." + ) +} + +const TRIMMED_NOTE_WHAT: &str = "gap markers and file headers name what was elided"; +const TRIMMED_NOTE_TAIL: &str = "For those, or anything under \"Not shown above\", make ANOTHER codegraph_explore with those exact names instead of reading the files — it returns their source with line numbers."; -fn completeness_signal(files_included: usize) -> String { +fn trimmed_note_head(files: &str) -> String { + format!("> **Verbatim source for {files} is included above — treat it as already Read.**") +} + +/// The least specific trimmed note: it names nothing, so it is the floor the +/// epilogue reserve holds. +fn generic_trimmed_note(files: &str) -> String { format!( - "> **Complete source for {files_included} files is included above — do NOT re-read them.** If your question also needs files/symbols listed under \"Not shown above\" (or any area this call didn't cover), make ANOTHER codegraph_explore targeting those names — it returns the same source with line numbers and is cheaper and more complete than reading. Reserve Read for a single specific line range explore can't surface." + "{} Some sections were trimmed for size; {TRIMMED_NOTE_WHAT}. {TRIMMED_NOTE_TAIL}", + trimmed_note_head(files) ) } +/// The large tiers' completeness note, as candidates from most to least +/// specific (upstream `exploreCompletenessNotes`, #2077). "Complete" is claimed +/// only when no section elided anything it set out to deliver. Otherwise the +/// note keeps the guarantee that is still true (every block shown is +/// verbatim), names the trimmed files and the most relevant elided symbols as +/// room allows, and sends the agent to another codegraph_explore for them. It +/// never offers Read. +fn completeness_notes( + files_included: usize, + trimmed: &[(String, Vec)], + known_paths: &[String], +) -> Vec { + let files = files_wording(files_included); + if trimmed.is_empty() { + return vec![complete_source_note(&files)]; + } + let mut labeled: Vec = known_paths.to_vec(); + labeled.extend(trimmed.iter().map(|(path, _)| path.clone())); + let labels = shortest_unique_suffixes(&labeled); + let shown: Vec = trimmed + .iter() + .take(TRIMMED_FILES_NAMED) + .map(|(path, _)| format!("`{}`", labels[path])) + .collect(); + let more = trimmed.len() - shown.len(); + let mut trimmed_list = shown.join(", "); + if more > 0 { + trimmed_list.push_str(&format!(" +{more} more")); + } + let mut names: Vec = Vec::new(); + for (_, elided) in trimmed { + for span in elided { + if names.len() >= TRIMMED_SYMBOLS_NAMED { + break; + } + // A container elided by a trim is too big for one section by + // construction; its members are the useful names. + let container = matches!( + span.kind, + "file" + | "module" + | "namespace" + | "class" + | "struct" + | "union" + | "interface" + | "protocol" + | "trait" + ); + if span.importance < 10 || container { + continue; + } + let name = follow_up_name(span); + if !names.contains(&name) { + names.push(name); + } + } + } + let head = trimmed_note_head(&files); + let with_files = format!("{head} Trimmed for size: {trimmed_list}; {TRIMMED_NOTE_WHAT}"); + let mut candidates = Vec::new(); + if !names.is_empty() { + let named = names + .iter() + .map(|name| format!("`{name}`")) + .collect::>() + .join(", "); + candidates.push(format!("{with_files} (e.g. {named}). {TRIMMED_NOTE_TAIL}")); + } + candidates.push(format!("{with_files}. {TRIMMED_NOTE_TAIL}")); + candidates.push(generic_trimmed_note(&files)); + candidates +} + +/// The longest note the epilogue can fall back to with at most `max_files` +/// sections: the reserve holds it, so the note always fits, and a more +/// specific one is used only when what the sections and the pointer list left +/// pays for its detail. +fn completeness_note_floor(max_files: usize) -> usize { + [ + files_wording(0), + files_wording(1), + files_wording(max_files.max(2)), + ] + .iter() + .map(|files| { + complete_source_note(files) + .len() + .max(generic_trimmed_note(files).len()) + }) + .max() + .unwrap_or(0) +} + +/// The note closing a response the hard ceiling cut, in two wordings: the +/// trimmed one drops "complete" when a section that survives the cut was +/// trimmed, and is no longer, so the cut's room holds either. +fn truncation_note(trimmed: bool) -> &'static str { + if trimmed { + "\n\n... (output truncated to budget; the source above is verbatim — treat it as already Read. For names its gap markers list, or any area not covered, run another codegraph_explore — do NOT Read these files.)" + } else { + "\n\n... (output truncated to budget; the source above is complete and verbatim — treat it as already Read. For any area not covered, run another codegraph_explore with the specific names — do NOT Read these files.)" + } +} + fn explore_budget_note(file_count: i64) -> String { format!( "> **{file_count} files indexed.** Each call covers ~6 files; if your question spans more, make ANOTHER `codegraph_explore` targeting the uncovered area BEFORE falling back to Read — another explore is cheaper and more complete than reading those files. There is no call limit." @@ -4266,17 +5665,19 @@ fn pointer_unlisted_tail(unlisted: usize) -> String { /// The pre-loop reserve and the post-loop emitter both call THIS function, so the /// reserve is derived from the very strings that get emitted and cannot drift from /// them. The reserve passes the arguments that maximise the result, which is why -/// it is an upper bound rather than an estimate. +/// it is an upper bound rather than an estimate: the completeness note at its +/// floor length (`completeness_note_floor`), which every note the emitter picks +/// either meets or pays the difference for from the room the sections left. fn epilogue_note_lines( budget: &ExploreOutputBudget, - files_included: usize, + completeness_note: &str, any_file_trimmed: bool, file_count: Option, ) -> Vec { let mut lines = Vec::new(); if budget.include_completeness_signal { lines.push(COMPLETENESS_RULE.to_string()); - lines.push(completeness_signal(files_included)); + lines.push(completeness_note.to_string()); lines.push(String::new()); } else if any_file_trimmed { lines.push(TRIMMED_NOTE.to_string()); @@ -4301,8 +5702,8 @@ fn joined_line_cost(lines: &[String]) -> usize { /// file is admitted so the sections cannot spend what the epilogue will need. /// /// Every term is the length of a string the epilogue actually emits, evaluated at -/// its worst case: `max_files` maximises the completeness signal's digit width, -/// `any_file_trimmed = true` selects the larger of the two mutually exclusive +/// its worst case: the completeness note at the longest length it can fall back +/// to, `any_file_trimmed = true` selects the larger of the two mutually exclusive /// notes, and the unlisted tail is reserved at `file_order` length, which is /// >= any achievable count because every unlisted file came from `file_order`. fn fixed_epilogue_reserve( @@ -4320,7 +5721,8 @@ fn fixed_epilogue_reserve( String::new(), ]); } - reserve += joined_line_cost(&epilogue_note_lines(budget, max_files, true, file_count)); + let floor_note = "x".repeat(completeness_note_floor(max_files)); + reserve += joined_line_cost(&epilogue_note_lines(budget, &floor_note, true, file_count)); reserve } @@ -4350,7 +5752,11 @@ fn fit_pointer_lines(candidates: &[String], pointer_budget: usize) -> (Vec (String, usize) { +fn cut_at_section_boundary( + output: &str, + ceiling: usize, + trimmed_files: &[&str], +) -> (String, usize) { if output.len() <= ceiling { return (output.to_string(), output.len()); } @@ -4362,10 +5768,12 @@ fn cut_at_section_boundary(output: &str, ceiling: usize) -> (String, usize) { }; let kept_prefix_len = if boundary > 0 { boundary } else { cut.len() }; let safe = &output[..kept_prefix_len]; + // "Complete" only while no trimmed section survives the cut (#2077). + let trimmed = trimmed_files + .iter() + .any(|file| safe.contains(&format!("#### {file} "))); ( - format!( - "{safe}\n\n... (output truncated to budget; the source above is complete and verbatim — treat it as already Read. For any area not covered, run another codegraph_explore with the specific names — do NOT Read these files.)" - ), + format!("{safe}{}", truncation_note(trimmed)), kept_prefix_len, ) } @@ -4435,6 +5843,8 @@ mod tests { budget, funded_headroom: budget.max_output_chars.saturating_sub(1), focuses, + exact_focus: &[], + lift_file_cap: false, drifted: false, } } @@ -4515,6 +5925,674 @@ mod tests { assert!(out.contains("line 30"), "cluster source missing: {out:?}"); } + /// A gap between two rendered spans names the indexed symbols that start + /// inside it, from the full file index (#1711), and only while the spare + /// budget pays for the names; otherwise it stays bare (#2057). + #[test] + fn gap_markers_name_the_symbols_a_trim_skipped_within_the_spare_budget() { + let nodes = vec![ + node("alpha", "a.ts", 10, 30, NodeKind::Function), + node("helperMid", "a.ts", 100, 120, NodeKind::Function), + node("helperMid", "a.ts", 130, 140, NodeKind::Function), + node("loader", "a.ts", 150, 150, NodeKind::Import), + node("beta", "a.ts", 200, 220, NodeKind::Function), + ]; + assert_eq!( + symbols_between_ranges(&nodes, 33, 197) + .iter() + .map(|s| (s.name.as_str(), s.start_line)) + .collect::>(), + vec![("helperMid", 100)] + ); + let parts = vec![(7, 33, "head".to_string()), (197, 223, "tail".to_string())]; + assert_eq!( + join_parts_with_named_gaps("a.ts", &parts, &nodes, usize::MAX), + "head\n\n... (gap: helperMid (a.ts:100)) ...\n\ntail" + ); + assert_eq!( + join_parts_with_named_gaps("a.ts", &parts, &nodes, 3), + "head\n\n... (gap) ...\n\ntail", + "names the spare cannot pay for stay out" + ); + + let many: Vec = (0..8) + .map(|i| ElidedSymbol { + name: format!("s{i}"), + kind: "function", + start_line: 40 + i, + }) + .collect(); + assert!(format_gap_marker("a.ts", &many).contains("s5 (a.ts:45), +2 more")); + } + + /// An empty explore explains itself lexically (upstream #1904): which + /// checked words matched nothing, which matched but scored out, and the + /// indexed names sharing a word to retry with. + #[test] + fn empty_explore_explains_lexical_misses_and_offers_shared_word_candidates() { + let mut engine = test_engine(); + put_indexed_source( + &engine, + "src/a.ts", + "function explainHowThingsWork() {\n return 1;\n}\n", + Language::TypeScript, + 1, + ); + put_nodes( + &mut engine, + &[node_lang( + "explainHowThingsWork", + "explainHowThingsWork", + "src/a.ts", + 1, + 3, + NodeKind::Function, + Language::TypeScript, + )], + ); + let explore = |query: &str| { + text_of(&engine.execute("codegraph_explore", &serde_json::json!({ "query": query }))) + }; + + let text = explore("how do we stop users signing up too fast"); + assert!(text.contains("No relevant code found"), "{text}"); + assert!(text.contains("lexically, not by meaning"), "{text}"); + let line = |prefix: &str| { + text.lines() + .find(|line| line.starts_with(prefix)) + .unwrap_or_default() + .to_string() + }; + assert!(line("No lexical matches").contains("`signing`"), "{text}"); + assert!(line("Matched indexed words").contains("`how`"), "{text}"); + assert!( + line("Candidates to retry with codegraph_explore").contains("`explainHowThingsWork`"), + "{text}" + ); + + let text = explore("zebra quantum"); + assert!( + text.contains("No lexical matches for checked words: `zebra`, `quantum`."), + "{text}" + ); + assert!( + text.contains("No shared-word symbol candidates found"), + "{text}" + ); + } + + #[test] + fn empty_explore_on_an_empty_index_says_so() { + let engine = test_engine(); + let text = text_of(&engine.execute( + "codegraph_explore", + &serde_json::json!({ "query": "signup throttle" }), + )); + assert!(text.contains("This project has nothing indexed"), "{text}"); + } + + #[test] + fn capped_word_lists_keep_whole_words_and_mark_the_cut() { + let words = ["alpha", "beta", "gamma"].map(String::from); + assert_eq!(capped_word_list(&words, 100), "`alpha`, `beta`, `gamma`"); + assert_eq!(capped_word_list(&words, 15), "`alpha`, `beta` …"); + assert_eq!(capped_word_list(&words, 3), " …"); + } + + /// The header prefers the symbols the trim cut, in source order, so + /// `+N more` stops hiding the answer (#1711). + #[test] + fn clustered_file_header_prefers_elided_symbols() { + let symbols = vec![ + "alpha(function)".to_string(), + "alpha(function)".to_string(), + "zed(function)".to_string(), + ]; + let elided = vec![ + ElidedSymbol { + name: "syncStateNow".to_string(), + kind: "method", + start_line: 300, + }, + ElidedSymbol { + name: "beta".to_string(), + kind: "function", + start_line: 120, + }, + ]; + let mut elided_sorted = elided.clone(); + elided_sorted.sort_by_key(|s| s.start_line); + assert_eq!( + explore_file_header("f.ts", &symbols, &elided_sorted, 3), + "#### f.ts — beta(function), syncStateNow(method), alpha(function), +1 more" + ); + } + + /// A named function packed with incidental helpers can outrank, by density, + /// a cluster holding a named function alone. Its helpers may then only use + /// what is left once the lower cluster's named body is paid for, so both + /// named bodies come back whole (upstream #2062). + #[test] + fn incidental_members_leave_room_for_a_lower_ranked_named_body() { + let engine = test_engine(); + let file = "lib/response.ts"; + let owned: Vec = (1..=600) + .map(|i| format!(" const v{i} = step{i}(input); // line {i}")) + .collect(); + let file_lines: Vec<&str> = owned.iter().map(String::as_str).collect(); + let send_file = node("sendFile", file, 10, 30, NodeKind::Function); + let send_body = node("sendBody", file, 300, 336, NodeKind::Function); + let mut nodes = vec![send_file.clone(), send_body.clone()]; + for k in 0..40 { + let start = 32 + 3 * k; + nodes.push(node( + &format!("helper{k}"), + file, + start, + start + 1, + NodeKind::Function, + )); + } + let sg = subgraph_with(nodes, vec![send_file.id.clone(), send_body.id.clone()]); + let budget = crate::explore_budget::get_explore_output_budget(200); + let out = engine + .render_explore_file( + &sg, + file, + &file_lines, + "typescript", + &render_ctx(&budget, &[]), + ) + .section; + let emitted = out + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(n, _)| n.parse::().ok()) + .collect::>(); + for line in (10..=30).chain(300..=336) { + assert!( + emitted.contains(&line), + "named body line {line} missing: {emitted:?}" + ); + } + } + + /// CG-38, as upstream's `windowToCeiling` has it: the whole room goes to a + /// windowed cluster's head first, so a focus the head already covers costs + /// the head nothing — it used to hold 40% of the room back for a window on + /// a line the head showed anyway. + #[test] + fn a_focus_the_head_already_covers_costs_the_head_nothing() { + let engine = test_engine(); + let file = "god.ts"; + let owned: Vec = (1..=600) + .map(|i| format!(" // padding line {i} inside hugeHandler keeping the body enormous")) + .collect(); + let file_lines: Vec<&str> = owned.iter().map(String::as_str).collect(); + let huge = node("hugeHandler", file, 1, 600, NodeKind::Function); + let sg = subgraph_with(vec![huge.clone()], vec![huge.id.clone()]); + let budget = crate::explore_budget::get_explore_output_budget(200); + let render = |focuses: &[usize]| { + engine + .render_explore_file( + &sg, + file, + &file_lines, + "typescript", + &render_ctx(&budget, focuses), + ) + .section + }; + assert_eq!(render(&[1]), render(&[])); + } + + /// #2063: two exact targets in different clusters of one file. The one + /// ranked first sits among named roots that would otherwise fill the whole + /// room, so it holds back what the lower-ranked exact target still owes, and + /// both render whole. + #[test] + fn an_exact_cluster_holds_back_what_a_lower_ranked_exact_target_owes() { + let engine = test_engine(); + let file = "app/App.py"; + let owned: Vec = (1..=600) + .map(|i| format!(" v{i} = step{i}(state) # line {i}")) + .collect(); + let file_lines: Vec<&str> = owned.iter().map(String::as_str).collect(); + let trigger = node("triggerRender", file, 10, 40, NodeKind::Method); + let pointer = node("handlePointer", file, 500, 530, NodeKind::Method); + let mut nodes = vec![trigger.clone(), pointer.clone()]; + let mut roots = Vec::new(); + for k in 0..30 { + let start = 42 + 4 * k; + let root = node( + &format!("hook{k}"), + file, + start, + start + 1, + NodeKind::Method, + ); + roots.push(root.id.clone()); + nodes.push(root); + } + let mut sg = subgraph_with(nodes, roots); + sg.exact_ids = vec![trigger.id.clone(), pointer.id.clone()]; + sg.exact_files.insert(file.to_string()); + let budget = crate::explore_budget::get_explore_output_budget(200); + let out = engine + .render_explore_file(&sg, file, &file_lines, "python", &render_ctx(&budget, &[])) + .section; + let emitted = out + .lines() + .filter_map(|line| line.split_once('\t')) + .filter_map(|(n, _)| n.parse::().ok()) + .collect::>(); + for line in (10..=40).chain(500..=530) { + assert!( + emitted.contains(&line), + "exact body line {line} missing: {emitted:?}" + ); + } + } + + /// #2063: an anchor resolves to the innermost callable enclosing its line; + /// a range, or a line no callable encloses, stays a span. + #[test] + fn line_anchors_resolve_to_the_innermost_callable_or_stay_spans() { + let mut engine = test_engine(); + let file = "pkg/compiler.py"; + let class = node("SQLCompiler", file, 1, 100, NodeKind::Class); + let method = node("as_sql", file, 10, 60, NodeKind::Method); + let inner = node("render_part", file, 20, 30, NodeKind::Function); + put_nodes(&mut engine, &[class, method.clone(), inner.clone()]); + let anchor = |start: usize, end: usize| QueryLineAnchor { + file: file.to_string(), + start, + end, + }; + let exact = engine.exact_targets( + "", + &[ + anchor(25, 25), + anchor(50, 50), + anchor(150, 150), + anchor(5, 5), + anchor(40, 60), + ], + ); + let ids: Vec<&str> = exact.nodes.iter().map(|n| n.id.as_str()).collect(); + assert_eq!(ids, vec![inner.id.as_str(), method.id.as_str()]); + assert_eq!( + exact.spans, + vec![ + (file.to_string(), 135, 165), + (file.to_string(), 1, 20), + (file.to_string(), 40, 60), + ] + ); + } + + /// #2063: a qualified name is exact only when its non-test definitions + /// number at most three, and then only its callables are. + #[test] + fn a_qualified_name_is_exact_only_when_it_names_at_most_three_definitions() { + let mut engine = test_engine(); + let lang = Language::Python; + let plan = node_lang( + "plan", + "Planner::plan", + "planner.py", + 2, + 9, + NodeKind::Method, + lang, + ); + let plan_test = node_lang( + "plan", + "Planner::plan", + "tests/test_planner.py", + 2, + 4, + NodeKind::Method, + lang, + ); + let limit = node_lang( + "limit", + "Planner::limit", + "planner.py", + 11, + 11, + NodeKind::Variable, + lang, + ); + let as_sql: Vec = ["a", "b", "c", "d"] + .iter() + .map(|dir| { + node_lang( + "as_sql", + "Compiler::as_sql", + &format!("{dir}/compiler.py"), + 3, + 8, + NodeKind::Method, + lang, + ) + }) + .collect(); + let render: Vec = ["a", "b", "c"] + .iter() + .map(|dir| { + node_lang( + "render", + "Widget::render", + &format!("{dir}/widget.py"), + 3, + 8, + NodeKind::Method, + lang, + ) + }) + .collect(); + let mut all = vec![plan.clone(), plan_test, limit]; + all.extend(as_sql); + all.extend(render.clone()); + put_nodes(&mut engine, &all); + let exact = engine.exact_targets( + "Planner.plan Planner.limit Compiler.as_sql Widget.render plan", + &[], + ); + let ids: Vec<&str> = exact.nodes.iter().map(|n| n.id.as_str()).collect(); + assert_eq!(ids.first(), Some(&plan.id.as_str()), "{ids:?}"); + let mut rest = ids[1..].to_vec(); + rest.sort_unstable(); + let mut want: Vec<&str> = render.iter().map(|n| n.id.as_str()).collect(); + want.sort_unstable(); + assert_eq!( + rest, want, + "three definitions are still exact; four are not" + ); + assert!(exact.spans.is_empty()); + } + + #[test] + fn exact_symbol_tokens_split_like_the_seeder() { + assert_eq!( + exact_symbol_tokens( + "SQLCompiler.as_sql, get_select() compiler.py Engine::ServeHTTP of x é.a get_select" + ), + vec![ + "SQLCompiler.as_sql", + "get_select", + "compiler", + "Engine::ServeHTTP", + ] + ); + } + + fn wanted( + name: &str, + start: usize, + end: usize, + importance: u32, + kind: &'static str, + ) -> WantedSpan { + WantedSpan { + name: name.to_string(), + kind, + start, + end, + importance, + qualified_name: None, + } + } + + fn names_of(spans: &[WantedSpan]) -> Vec<&str> { + spans.iter().map(|s| s.name.as_str()).collect() + } + + /// Anything that offers Read as a way forward (#2077). "treat it as already + /// Read" is the guarantee and "do NOT Read" a prohibition; neither offers it. + fn offers_read(text: &str) -> bool { + ["Reserve Read", "use Read", "Read for ", "fall back to Read"] + .iter() + .any(|offer| text.contains(offer)) + } + + /// The note #2077 replaced, for the size bound. + const OLD_COMPLETENESS_NOTE: &str = "> **Complete source for 8 files is included above — do NOT re-read them.** If your question also needs files/symbols listed under \"Not shown above\" (or any area this call didn't cover), make ANOTHER codegraph_explore targeting those names — it returns the same source with line numbers and is cheaper and more complete than reading. Reserve Read for a single specific line range explore can't surface."; + + #[test] + fn elided_wanted_spans_are_judged_on_what_was_sent() { + let elided = elided_wanted_spans( + &[ + wanted("inside", 10, 20, 1, "method"), + wanted("straddles", 25, 60, 1, "method"), + wanted("absent", 90, 95, 1, "method"), + ], + &[(1, 40)], + ); + assert_eq!(names_of(&elided), vec!["straddles", "absent"]); + // Adjacent ranges are one continuous copy. + assert!( + elided_wanted_spans(&[wanted("whole", 5, 45, 1, "method")], &[(1, 30), (31, 50)]) + .is_empty() + ); + // Most relevant first, then source order; no usable range, no entry. + let ordered = elided_wanted_spans( + &[ + wanted("late", 80, 81, 10, "method"), + wanted("peripheral", 5, 6, 1, "method"), + wanted("early", 40, 41, 10, "method"), + wanted("zero", 0, 0, 10, "method"), + wanted("inverted", 9, 3, 10, "method"), + ], + &[], + ); + assert_eq!(names_of(&ordered), vec!["early", "late", "peripheral"]); + } + + #[test] + fn shortest_unique_suffixes_name_a_file_apart_from_its_namesakes() { + let paths: Vec = [ + "src/vs/workbench/api/common/extHostExtensionService.ts", + "src/vs/workbench/api/node/extHostExtensionService.ts", + "src/vs/workbench/services/extensions/common/rpcProtocol.ts", + ] + .map(str::to_string) + .to_vec(); + let labels = shortest_unique_suffixes(&paths); + assert_eq!(labels[&paths[0]], "common/extHostExtensionService.ts"); + assert_eq!(labels[&paths[1]], "node/extHostExtensionService.ts"); + assert_eq!(labels[&paths[2]], "rpcProtocol.ts"); + // One path a suffix of another: the whole path. + let nested: Vec = ["a/b.ts", "x/a/b.ts"].map(str::to_string).to_vec(); + let labels = shortest_unique_suffixes(&nested); + assert_eq!(labels["a/b.ts"], "a/b.ts"); + assert_eq!(labels["x/a/b.ts"], "x/a/b.ts"); + } + + #[test] + fn completeness_is_claimed_only_when_nothing_was_trimmed() { + let notes = completeness_notes(4, &[], &["a.ts".to_string(), "b.ts".to_string()]); + assert_eq!(notes.len(), 1); + assert!( + notes[0].contains("Complete source for 4 files"), + "{}", + notes[0] + ); + assert!(!offers_read(¬es[0]), "{}", notes[0]); + + let file = "src/vs/workbench/services/extensions/common/rpcProtocol.ts".to_string(); + let trimmed = vec![( + file.clone(), + vec![ + wanted("_receiveOneMessage", 280, 357, 11, "method"), + // A container is no follow-up target. + wanted("RPCProtocol", 200, 900, 10, "class"), + wanted("serializeRequest", 700, 720, 10, "method"), + wanted("helperNobodyAskedFor", 10, 12, 1, "method"), + ], + )]; + let notes = completeness_notes(3, &trimmed, &[file, "src/a.ts".to_string()]); + for note in ¬es { + assert!(!note.contains("Complete source"), "{note}"); + assert!(note.contains("Verbatim source for 3 files"), "{note}"); + assert!(note.contains("treat it as already Read"), "{note}"); + assert!(note.contains("codegraph_explore"), "{note}"); + assert!(!offers_read(note), "{note}"); + } + assert!(notes[0].contains("`rpcProtocol.ts`"), "{}", notes[0]); + assert!( + notes[0].contains("`_receiveOneMessage`, `serializeRequest`"), + "{}", + notes[0] + ); + assert!(!notes[0].contains("`RPCProtocol`"), "{}", notes[0]); + assert!(!notes[0].contains("helperNobodyAskedFor"), "{}", notes[0]); + } + + #[test] + fn completeness_notes_count_files_in_words_and_label_namesakes_apart() { + let one = vec![("a.ts".to_string(), vec![wanted("f", 1, 9, 10, "function")])]; + let known = ["a.ts".to_string()]; + assert!( + completeness_notes(1, &[], &known)[0] + .contains("Complete source for 1 file is included") + ); + assert!( + completeness_notes(1, &one, &known)[0] + .contains("Verbatim source for 1 file is included") + ); + for note in completeness_notes(0, &one, &known) { + assert!(!note.contains(" 0 file"), "{note}"); + assert!(note.contains("Verbatim source for these files"), "{note}"); + } + let trimmed = vec![( + "src/common/rpcProtocol.ts".to_string(), + vec![wanted("f", 1, 9, 10, "function")], + )]; + let notes = completeness_notes( + 1, + &trimmed, + &[ + "src/common/rpcProtocol.ts".to_string(), + "src/node/rpcProtocol.ts".to_string(), + ], + ); + assert!(notes[0].contains("`common/rpcProtocol.ts`"), "{}", notes[0]); + } + + #[test] + fn completeness_notes_offer_an_elided_method_as_owner_dot_member() { + let q = |mut span: WantedSpan, qualified: &str| { + span.qualified_name = Some(qualified.to_string()); + span + }; + let trimmed = vec![( + "django/db/models/sql/compiler.py".to_string(), + vec![ + q( + wanted("as_sql", 776, 1003, 12, "method"), + "SQLCompiler::as_sql", + ), + q( + wanted("PLUGIN_ID", 5, 5, 10, "method"), + "org.lamport.tla::HelpActivator::PLUGIN_ID", + ), + // A local function keeps its bare name. + q(wanted("inner", 40, 44, 10, "function"), "Outer::run::inner"), + // A path is no owner. + q(wanted("odd", 50, 52, 10, "method"), "src/a.py::odd"), + ], + )]; + let notes = completeness_notes( + 1, + &trimmed, + &["django/db/models/sql/compiler.py".to_string()], + ); + assert!( + notes[0] + .contains("(e.g. `SQLCompiler.as_sql`, `HelpActivator.PLUGIN_ID`, `inner`, `odd`)"), + "{}", + notes[0] + ); + } + + #[test] + fn completeness_notes_run_from_most_to_least_specific_within_the_old_bound() { + let trimmed: Vec<(String, Vec)> = ["a", "b", "c", "d", "e"] + .iter() + .map(|n| { + ( + format!("packages/{n}/src/deeply/nested/{n}Service.ts"), + vec![wanted(&format!("{n}Handler"), 10, 90, 10, "function")], + ) + }) + .collect(); + let known: Vec = trimmed.iter().map(|(path, _)| path.clone()).collect(); + let notes = completeness_notes(8, &trimmed, &known); + assert_eq!(notes.len(), 3); + for pair in notes.windows(2) { + assert!(pair[1].len() < pair[0].len(), "{pair:?}"); + } + assert!( + notes[1].contains("`aService.ts`, `bService.ts`, `cService.ts` +2 more"), + "{}", + notes[1] + ); + assert!(!notes[2].contains("Service.ts"), "{}", notes[2]); + assert!(notes[2].len() < OLD_COMPLETENESS_NOTE.len()); + // The reserve covers the longest note the epilogue can fall back to. + assert!(completeness_note_floor(8) >= notes[2].len()); + assert!(completeness_note_floor(8) >= completeness_notes(8, &[], &known)[0].len()); + assert!(completeness_note_floor(8) >= completeness_notes(0, &trimmed, &known)[2].len()); + } + + #[test] + fn the_truncation_note_drops_complete_for_a_trimmed_response() { + let (complete, trimmed) = (truncation_note(false), truncation_note(true)); + assert!(trimmed.len() <= complete.len()); + assert!(complete.contains("complete and verbatim")); + assert!(!trimmed.contains("complete"), "{trimmed}"); + assert!(trimmed.contains("treat it as already Read")); + for note in [complete, trimmed] { + assert!(note.contains("codegraph_explore")); + assert!(!offers_read(note), "{note}"); + } + } + + /// End to end: two rendered clusters with an index-only symbol between + /// them get a named gap. + #[test] + fn render_explore_file_names_an_index_only_symbol_in_a_gap() { + let mut engine = test_engine(); + let file = "big.ts"; + let owned: Vec = (1..=400).map(|i| format!("line {i}")).collect(); + let file_lines: Vec<&str> = owned.iter().map(String::as_str).collect(); + let alpha = node("alpha", file, 10, 30, NodeKind::Function); + let beta = node("beta", file, 200, 220, NodeKind::Function); + let mid = node("helperMid", file, 100, 110, NodeKind::Function); + put_nodes(&mut engine, &[alpha.clone(), mid, beta.clone()]); + let sg = subgraph_with( + vec![alpha.clone(), beta.clone()], + vec![alpha.id.clone(), beta.id.clone()], + ); + let budget = crate::explore_budget::get_explore_output_budget(200); + let out = engine + .render_explore_file( + &sg, + file, + &file_lines, + "typescript", + &render_ctx(&budget, &[]), + ) + .section; + assert!(out.contains("line 10") && out.contains("line 200"), "{out}"); + assert!( + out.contains("... (gap: helperMid (big.ts:100)) ..."), + "{out}" + ); + } + /// Regression: a small HEALTHY file returns WHOLE, byte-for-byte, exactly as /// before the clamp change. #[test] @@ -4599,6 +6677,8 @@ mod tests { budget: &budget, funded_headroom: whole.len() - 1, focuses: &[], + exact_focus: &[], + lift_file_cap: false, drifted: false, }; let unaffordable_no_focus = engine @@ -5738,7 +7818,7 @@ mod tests { format!("#### {file} — truncated\n\n```rust\nfn truncated() {{}}\n```\n"), ]; let output = lines.join("\n"); - let (output, kept_prefix_len) = cut_at_section_boundary(&output, 180); + let (output, kept_prefix_len) = cut_at_section_boundary(&output, 180, &[]); engine.mark_surviving_explore_citations(&lines, &[(file.to_string(), 1)], kept_prefix_len); let result = engine.with_staleness_banner(ToolResult::text(output)); @@ -6009,7 +8089,10 @@ mod tests { "an admission-dropped file must be named in the pointer block: {text}" ); assert!( - text.contains(&format!("Complete source for {} files", headers.len())), + text.contains(&format!( + "Complete source for {} is included", + files_wording(headers.len()) + )), "the epilogue must survive with a count matching the rendered sections: {text}" ); assert!( @@ -6868,7 +8951,7 @@ mod tests { s.push_str(&"a".repeat(100)); s.push_str("\n#### file.rs — X\n"); s.push_str(&"b".repeat(400)); - let (out, _) = cut_at_section_boundary(&s, 200); + let (out, _) = cut_at_section_boundary(&s, 200, &[]); assert!(out.contains("output truncated to budget"), "got: {out}"); } @@ -7592,7 +9675,7 @@ mod tests { "a(function)".to_string(), "b(function)".to_string(), ]; - let h = explore_file_header("f.rs", &syms, 5); + let h = explore_file_header("f.rs", &syms, &[], 5); assert!(h.starts_with("#### f.rs — "), "got: {h}"); } diff --git a/crates/codegraph-mcp/src/instructions.rs b/crates/codegraph-mcp/src/instructions.rs index 670568e..c6c9a6f 100644 --- a/crates/codegraph-mcp/src/instructions.rs +++ b/crates/codegraph-mcp/src/instructions.rs @@ -110,6 +110,7 @@ typically one to a few calls; a grep/read exploration is dozens. - If a tool reports the project isn't initialized, `.codegraph/` doesn't exist yet — offer to run `codegraph init .` to build the index. - Index lags file writes by ~1 second. - Cross-file resolution is best-effort name matching; ambiguous calls may return multiple candidates. +- Explore matches names and indexed code words lexically, not by meaning; an empty result reports word matches and may suggest indexed candidate names to retry with `codegraph_explore`. - No live correctness validation — that's still the TypeScript compiler / test suite / linter's job. Codegraph supplements those with structural context they don't have. "#; diff --git a/crates/codegraph-mcp/src/query_paths.rs b/crates/codegraph-mcp/src/query_paths.rs index 941cce9..930b672 100644 --- a/crates/codegraph-mcp/src/query_paths.rs +++ b/crates/codegraph-mcp/src/query_paths.rs @@ -20,9 +20,39 @@ static DOTTED_BASENAME: LazyLock = LazyLock::new(|| { static KEBAB_BASENAME: LazyLock = LazyLock::new(|| { Regex::new(r"^[A-Za-z0-9]+(?:-[A-Za-z0-9]+)+$").expect("kebab basename regex is valid") }); +/// Line references that ride along in agent-written paths: `foo.ts:123`, +/// `foo.ts:12-40`, `foo.ts#L88`, `foo.ts#L88-L120`. ASCII digits only, as in +/// upstream's JavaScript `\d`. static LINE_REFERENCE: LazyLock = LazyLock::new(|| { - Regex::new(r"(?::\d+(?:-\d+)?|#L\d+(?:-L?\d+)?)$").expect("line-reference regex is valid") + Regex::new(r"(?::([0-9]+)(?:-([0-9]+))?|#L([0-9]+)(?:-L?([0-9]+))?)$") + .expect("line-reference regex is valid") }); +/// A standalone line-number token: `900`, `900-1003`, `900–1003`, `900..1003`, +/// `L900`, `L900-L1003`. The `L` form is a line number on its own; a bare number +/// only counts after a `line`/`lines` word or directly after a path. +static LINE_NUMBER_TOKEN: LazyLock = LazyLock::new(|| { + Regex::new(r"^(L?)([0-9]+)(?:(?:-|\x{2013}|\.\.)L?([0-9]+))?$") + .expect("line-number regex is valid") +}); +static LINE_WORD: LazyLock = + LazyLock::new(|| Regex::new(r"(?i)^lines?$").expect("line-word regex is valid")); +/// `lines 900 to 1003`: the connective between two bare numbers. +static RANGE_CONNECTIVE: LazyLock = LazyLock::new(|| { + Regex::new(r"(?i)^(?:to|through|thru)$").expect("range-connective regex is valid") +}); +/// Nothing real is this long; a larger number is a port, an id or a typo. +const MAX_LINE_NUMBER: u64 = 1_000_000; +/// A query token that names a symbol, in the shape explore's named-symbol +/// seeder reads: `clampedInt`, `SQLCompiler.as_sql`, `Engine::ServeHTTP`. +/// ASCII only, as JavaScript's `\w`. +static SYMBOL_TOKEN: LazyLock = LazyLock::new(|| { + Regex::new(r"^[A-Za-z_$][A-Za-z0-9_$]*(?:(?:::|\.)[A-Za-z0-9_$]+)*$") + .expect("symbol-token regex is valid") +}); +/// Symbol tokens consulted per query — the seeder's cap. +const MAX_SYMBOL_TOKENS: usize = 16; +/// Files one span may pin before it is ambiguous. +const MAX_MATCHES_PER_SPAN: usize = 3; static LAST_EXTENSION: LazyLock = LazyLock::new(|| { Regex::new(r"\.[A-Za-z][A-Za-z0-9]{0,7}$").expect("last-extension regex is valid") }); @@ -32,11 +62,41 @@ pub(crate) struct QueryPathExtraction { pub stripped_query: String, pub pinned_files: Vec, pub unresolved_path_spans: Vec, + /// Line spans the query anchored to a pinned file, 1-based and inclusive: + /// `compiler.py:776`, `foo.ts:12-40`, `foo.ts#L88-L120`, or prose next to a + /// path (`compiler.py lines 900-1003`). An agent writes these when it wants + /// THOSE lines, so a pin that drops them answers a different question. + /// Only a span whose path resolved to exactly one file is kept: a line + /// number means nothing across two candidate files (upstream #2063). + pub line_anchors: Vec, + /// Files a span matched but did NOT pin: the span named several files, some + /// of which define a symbol the query also names, and these define none + /// (upstream #2071). Surfaced so a set-aside file is visible, not silently + /// dropped. + pub set_aside_matches: Vec, +} + +/// The index lookup path extraction is given: which indexed files define a +/// symbol spelled like this query token. +pub(crate) type SymbolFiles<'a> = &'a dyn Fn(&str) -> Vec; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct QuerySetAsideMatch { + /// The span as the query wrote it, normalized (`editorOptions.ts`). + pub span: String, + pub files: Vec, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct QueryLineAnchor { + pub file: String, + pub start: usize, + pub end: usize, } pub(crate) fn query_might_contain_paths(query: &str) -> bool { query.split_whitespace().any(|token| { - let stripped = strip_wrapping(token); + let (stripped, _) = strip_wrapping(token); stripped.contains(['/', '\\']) || DOTTED_BASENAME.is_match(&stripped) || KEBAB_BASENAME.is_match(&stripped) @@ -49,33 +109,61 @@ pub(crate) fn extract_query_paths( indexed_paths: &[String], max_pins: usize, ) -> QueryPathExtraction { - extract_query_paths_inner(query, indexed_paths, max_pins, None) + extract_query_paths_with_probes(query, indexed_paths, max_pins, None, None) } +#[cfg(test)] pub(crate) fn extract_query_paths_with_file_probe( query: &str, indexed_paths: &[String], max_pins: usize, exists_on_disk: &dyn Fn(&str) -> bool, ) -> QueryPathExtraction { - extract_query_paths_inner(query, indexed_paths, max_pins, Some(exists_on_disk)) + extract_query_paths_with_probes(query, indexed_paths, max_pins, Some(exists_on_disk), None) } -fn extract_query_paths_inner( +/// Resolve the query's paths. Both probes are injected so this module stays +/// free of the filesystem and the index, and neither may fail: +/// +/// - `exists_on_disk` tells a dotless path the index does not hold +/// (`scripts/deploy`) from slashed prose (`and/or`). +/// - `symbol_files` answers which indexed files define a symbol spelled like a +/// query token (`clampedInt`, `SQLCompiler.as_sql`). It is consulted only when +/// a span matches several files: the matches defining a symbol the query also +/// names are pinned and the rest set aside (upstream #2071). +pub(crate) fn extract_query_paths_with_probes( query: &str, indexed_paths: &[String], max_pins: usize, exists_on_disk: Option<&dyn Fn(&str) -> bool>, + symbol_files: Option>, ) -> QueryPathExtraction { let passthrough = || QueryPathExtraction { stripped_query: query.to_string(), pinned_files: Vec::new(), unresolved_path_spans: Vec::new(), + line_anchors: Vec::new(), + set_aside_matches: Vec::new(), }; if query.trim().is_empty() || (indexed_paths.is_empty() && exists_on_disk.is_none()) { return passthrough(); } let max_pins = max_pins.clamp(1, MAX_PINS); + let mut narrowing = SymbolNarrowing { + symbol_files, + query, + indexed_paths, + tokens: None, + definers_of: BTreeMap::new(), + }; + let mut set_aside: Vec = Vec::new(); + // A span is collected past the ambiguity budget only while narrowing is + // possible, and still reports as ambiguous if it does not narrow into it. + let collect_limit = if symbol_files.is_some() { + usize::MAX + } else { + MAX_MATCHES_PER_SPAN + }; let lower_to_original = indexed_paths .iter() .map(|path| (path.to_lowercase(), path.clone())) @@ -88,13 +176,17 @@ fn extract_query_paths_inner( let mut pinned = Vec::new(); let mut pinned_seen = BTreeSet::new(); let mut unresolved = Vec::new(); + let mut anchors: Vec = Vec::new(); + // Token index → the ONE file it pinned, in insertion order, for binding + // prose line ranges to their nearest path. + let mut single_file_at: Vec<(usize, String)> = Vec::new(); let mut candidates_examined = 0usize; for (index, token) in tokens.iter().enumerate() { if pinned.len() >= max_pins || candidates_examined >= MAX_CANDIDATE_SPANS { break; } - let stripped = strip_wrapping(token); + let (stripped, lines) = strip_wrapping(token); if stripped.chars().count() < 4 { continue; } @@ -107,18 +199,37 @@ fn extract_query_paths_inner( continue; } candidates_examined += 1; - let resolved = resolve_span(&normalized.to_lowercase(), &lower_to_original, 3); - if !resolved.matches.is_empty() { + let resolved = resolve_span( + &normalized.to_lowercase(), + &lower_to_original, + collect_limit, + ); + let collected = !resolved.matches.is_empty(); + let pinnable = if collected { + narrowing.pinnable(&normalized, resolved.matches, &mut set_aside) + } else { + None + }; + let ambiguous = resolved.ambiguous || (collected && pinnable.is_none()); + if let Some(matches) = pinnable { consumed.insert(index); - for path in resolved.matches { + for path in &matches { if pinned.len() >= max_pins { break; } if pinned_seen.insert(path.clone()) { - pinned.push(path); + pinned.push(path.clone()); } } - } else if resolved.ambiguous + record_single_file( + index, + &matches, + lines, + &pinned_seen, + &mut single_file_at, + &mut anchors, + ); + } else if ambiguous || is_clearly_path_shaped(&normalized) || (normalized.contains('/') && exists_on_disk.is_some_and(|probe| probe(&normalized))) { @@ -139,19 +250,19 @@ fn extract_query_paths_inner( if consumed.contains(&index) { continue; } - let stripped = strip_wrapping(token); + let (stripped, lines) = strip_wrapping(token); if stripped.chars().count() < 4 || !KEBAB_BASENAME.is_match(&stripped) { continue; } candidates_examined += 1; - let Some(matches) = stems.get(&stripped.to_lowercase()) else { + let Some(matches) = stems + .get(&stripped.to_lowercase()) + .and_then(|matches| narrowing.pinnable(&stripped, matches.clone(), &mut set_aside)) + else { continue; }; - if matches.len() > 3 { - continue; - } consumed.insert(index); - for path in matches { + for path in &matches { if pinned.len() >= max_pins { break; } @@ -159,11 +270,81 @@ fn extract_query_paths_inner( pinned.push(path.clone()); } } + record_single_file( + index, + &matches, + lines, + &pinned_seen, + &mut single_file_at, + &mut anchors, + ); + } + + // Third pass: prose line ranges (`lines 900-1003`, `line 42`, `L88-L120`, + // `lines 900 to 1003`) bound to the NEAREST single-file path token. Left + // in the query the numbers match nothing and `lines` feeds FTS a word + // every file contains; a range with no path to bind to says nothing about + // which file, so it is left alone. + if !single_file_at.is_empty() { + let nearest_file = |index: usize| -> String { + let mut best = &single_file_at[0]; + for entry in &single_file_at { + if entry.0.abs_diff(index) < best.0.abs_diff(index) { + best = entry; + } + } + best.1.clone() + }; + for index in 0..tokens.len() { + if consumed.contains(&index) { + continue; + } + let token = strip_line_token_punctuation(&tokens[index]); + let after_word = index > 0 + && !consumed.contains(&(index - 1)) + && LINE_WORD.is_match(strip_line_token_punctuation(&tokens[index - 1])); + let after_path = index > 0 && single_file_at.iter().any(|(at, _)| *at == index - 1); + let Some(number) = LINE_NUMBER_TOKEN.captures(token) else { + continue; + }; + // A bare number is a line number only in a line context; `L900` is + // one on its own. + if number[1].is_empty() && !after_word && !after_path { + continue; + } + let mut span = line_span(&number[2], number.get(3).map(|m| m.as_str())); + let mut used = vec![index]; + if span.is_some() + && number.get(3).is_none() + && index + 2 < tokens.len() + && RANGE_CONNECTIVE.is_match(strip_line_token_punctuation(&tokens[index + 1])) + && let Some(tail) = + LINE_NUMBER_TOKEN.captures(strip_line_token_punctuation(&tokens[index + 2])) + && tail.get(3).is_none() + { + span = line_span(&number[2], Some(&tail[2])); + used.extend([index + 1, index + 2]); + } + let Some((start, end)) = span else { + continue; + }; + anchors.push(QueryLineAnchor { + file: nearest_file(index), + start, + end, + }); + consumed.extend(used); + if after_word { + consumed.insert(index - 1); + } + } } if consumed.is_empty() { return passthrough(); } + let mut seen_anchor = BTreeSet::new(); + anchors.retain(|a| seen_anchor.insert((a.file.clone(), a.start, a.end))); QueryPathExtraction { stripped_query: tokens .into_iter() @@ -173,7 +354,202 @@ fn extract_query_paths_inner( .join(" "), pinned_files: pinned, unresolved_path_spans: unresolved, + line_anchors: anchors, + // A file another span pinned outright (the agent also wrote its full + // path) was not set aside after all. + set_aside_matches: set_aside + .into_iter() + .map(|entry| QuerySetAsideMatch { + span: entry.span, + files: entry + .files + .into_iter() + .filter(|file| !pinned_seen.contains(file)) + .collect(), + }) + .filter(|entry| !entry.files.is_empty()) + .collect(), + } +} + +/// Narrowing a span's matches by the symbols the query names (upstream #2071). +/// +/// A bare basename two directories share (`editorOptions.ts`) used to pin every +/// match, and the matches split the pinned reservation. When the query names +/// symbols, the file it means is the one defining them: the matches defining at +/// least one are kept and the rest set aside. When no match defines one, or +/// every match does, the symbols pick nothing out and the span resolves as +/// before. The same test lets a basename shared past the ambiguity budget +/// (django's `models.py`) resolve after all, to the one defining the named +/// class. +struct SymbolNarrowing<'a> { + symbol_files: Option>, + query: &'a str, + indexed_paths: &'a [String], + /// The query's precise symbol tokens, computed on first use. + tokens: Option>, + definers_of: BTreeMap>, +} + +impl SymbolNarrowing<'_> { + /// The matches defining a named symbol, or `None` when narrowing picks + /// nothing out. + fn narrow(&mut self, matches: &[String]) -> Option> { + let symbol_files = self.symbol_files?; + if matches.len() < 2 { + return None; + } + let tokens = self.tokens.get_or_insert_with(|| { + let basenames: BTreeSet = self + .indexed_paths + .iter() + .map(|path| { + path.rsplit(['/', '\\']) + .next() + .unwrap_or(path) + .to_lowercase() + }) + .collect(); + query_symbol_tokens(self.query, &basenames) + }); + let candidates: BTreeSet<&String> = matches.iter().collect(); + let mut definers: BTreeSet = BTreeSet::new(); + for token in tokens.iter() { + let files = self + .definers_of + .entry(token.clone()) + .or_insert_with(|| symbol_files(token).into_iter().collect()); + definers.extend( + files + .iter() + .filter(|file| candidates.contains(file)) + .cloned(), + ); + } + if definers.is_empty() || definers.len() == candidates.len() { + return None; + } + Some( + matches + .iter() + .filter(|file| definers.contains(*file)) + .cloned() + .collect(), + ) + } + + /// The matches a span may pin, recording what it set aside; `None` when + /// they stay over the ambiguity budget. + fn pinnable( + &mut self, + span: &str, + matches: Vec, + set_aside: &mut Vec, + ) -> Option> { + if let Some(narrowed) = self.narrow(&matches) + && narrowed.len() <= MAX_MATCHES_PER_SPAN + { + set_aside.push(QuerySetAsideMatch { + span: span.to_string(), + files: matches + .into_iter() + .filter(|file| !narrowed.contains(file)) + .collect(), + }); + return Some(narrowed); + } + (matches.len() <= MAX_MATCHES_PER_SPAN).then_some(matches) + } +} + +/// The seeder's NL-stopword test: camelCase, PascalCase, snake_case, `$` and +/// qualified tokens are unmistakably code. A bare lowercase word (`options`, +/// `render`) is also English, and a file defining one proves nothing about +/// which same-named file the agent meant. +fn is_precise_symbol_token(token: &str) -> bool { + token.contains(['.', '_', '$']) + || token.contains("::") + || token.starts_with(|c: char| c.is_ascii_uppercase()) + || token + .as_bytes() + .windows(2) + .any(|w| w[0].is_ascii_lowercase() && w[1].is_ascii_uppercase()) +} + +/// The precise symbol tokens a query names, excluding the files it names +/// (`editorOptions.ts` is identifier-shaped too). Split on the same brackets +/// the seeder splits on, so `clampedInt()` still counts. +fn query_symbol_tokens(query: &str, indexed_basenames: &BTreeSet) -> Vec { + let mut out: Vec = Vec::new(); + for raw in query.split_whitespace() { + for part in raw.split([',', '(', ')', '[', ']', '{', '}']) { + let token = part + .trim_start_matches(['\'', '"', '`', '<']) + .trim_end_matches(['\'', '"', '`', '>', '.', ',', ';', ':', '!', '?']); + if token.len() < 3 + || !SYMBOL_TOKEN.is_match(token) + || !is_precise_symbol_token(token) + || indexed_basenames.contains(&token.to_lowercase()) + { + continue; + } + if !out.iter().any(|seen| seen == token) { + out.push(token.to_string()); + } + if out.len() >= MAX_SYMBOL_TOKENS { + return out; + } + } + } + out +} + +/// Remember a token that resolved to exactly one pinned file, and the line +/// reference it carried, if any. +fn record_single_file( + index: usize, + matches: &[String], + lines: Option<(usize, usize)>, + pinned_seen: &BTreeSet, + single_file_at: &mut Vec<(usize, String)>, + anchors: &mut Vec, +) { + let [file] = matches else { + return; + }; + if !pinned_seen.contains(file) { + return; + } + single_file_at.push((index, file.clone())); + if let Some((start, end)) = lines { + anchors.push(QueryLineAnchor { + file: file.clone(), + start, + end, + }); + } +} + +/// A 1-based inclusive span, in order, or `None` for a zero, an overflow, or a +/// number no file has. +fn line_span(start: &str, end: Option<&str>) -> Option<(usize, usize)> { + let start = start.parse::().ok()?; + let end = match end { + Some(end) => end.parse::().ok()?, + None => start, + }; + if start < 1 || end < 1 || start > MAX_LINE_NUMBER || end > MAX_LINE_NUMBER { + return None; } + let (start, end) = (start as usize, end as usize); + Some((start.min(end), start.max(end))) +} + +/// Prose punctuation a line-number token can carry: `lines 900-1003,`, `(L88)`. +fn strip_line_token_punctuation(token: &str) -> &str { + token + .trim_start_matches(['(', '\'', '"', '`', '[']) + .trim_end_matches([')', '\'', '"', '`', ']', '.', ',', ';', ':', '!', '?']) } struct SpanResolution { @@ -244,7 +620,10 @@ fn build_basename_stems(indexed_paths: &[String]) -> BTreeMap String { +/// Strip prose punctuation around a token without eating punctuation that is +/// part of the path, then split off a trailing line reference: the file is what +/// resolves, the lines are what the agent wants from it. +fn strip_wrapping(token: &str) -> (String, Option<(usize, usize)>) { let mut value = token.to_string(); while let Some(first) = value.chars().next() { let strip = matches!(first, '\'' | '"' | '`' | '<') @@ -266,7 +645,17 @@ fn strip_wrapping(token: &str) -> String { } value.pop(); } - LINE_REFERENCE.replace(&value, "").to_string() + let Some(reference) = LINE_REFERENCE.captures(&value) else { + return (value, None); + }; + let lines = match (reference.get(1), reference.get(3)) { + (Some(start), _) => line_span(start.as_str(), reference.get(2).map(|m| m.as_str())), + (None, Some(start)) => line_span(start.as_str(), reference.get(4).map(|m| m.as_str())), + (None, None) => None, + }; + let path_end = reference.get(0).map_or(value.len(), |m| m.start()); + value.truncate(path_end); + (value, lines) } fn normalize_span(span: &str) -> String { @@ -455,4 +844,311 @@ mod tests { assert_eq!(out.pinned_files, vec!["scripts/pre-commit"]); assert!(out.unresolved_path_spans.is_empty()); } + + fn anchor(file: &str, start: usize, end: usize) -> QueryLineAnchor { + QueryLineAnchor { + file: file.to_string(), + start, + end, + } + } + + const CHAT: &str = "src/lib/chat-manager.ts"; + + #[test] + fn keeps_a_line_suffix_as_an_anchor_on_the_pinned_file() { + let paths = index(); + let anchors = |query: &str| extract_query_paths(query, &paths, 8).line_anchors; + assert_eq!( + anchors(&format!("body of {CHAT}:776")), + vec![anchor(CHAT, 776, 776)] + ); + assert_eq!( + anchors(&format!("see {CHAT}:12-40")), + vec![anchor(CHAT, 12, 40)] + ); + assert_eq!( + anchors("regression at src/lib/task-runner-manager.ts#L88-L120"), + vec![anchor("src/lib/task-runner-manager.ts", 88, 120)] + ); + } + + #[test] + fn binds_a_prose_line_range_to_the_adjacent_path_and_strips_it() { + let paths = index(); + let out = extract_query_paths(&format!("{CHAT} lines 900-1003 flushQueue tail"), &paths, 8); + assert_eq!(out.line_anchors, vec![anchor(CHAT, 900, 1003)]); + // `lines` would feed FTS a word every file holds; the numbers match nothing. + assert_eq!(out.stripped_query, "flushQueue tail"); + } + + #[test] + fn accepts_the_other_line_range_spellings() { + let paths = index(); + let anchors = |query: &str| extract_query_paths(query, &paths, 8).line_anchors; + for query in [ + format!("lines 900 to 1003 of {CHAT}"), + format!("L900-L1003 in {CHAT}"), + format!("{CHAT} 900-1003"), + format!("{CHAT} (lines 1003-900)"), + format!("{CHAT} lines 900\u{2013}1003"), + format!("{CHAT} lines 900..1003"), + format!("{CHAT} lines 900 through 1003"), + format!("{CHAT} lines 900 thru L1003."), + ] { + assert_eq!(anchors(&query), vec![anchor(CHAT, 900, 1003)], "{query}"); + } + assert_eq!( + anchors(&format!("{CHAT} line 42")), + vec![anchor(CHAT, 42, 42)] + ); + } + + #[test] + fn binds_each_range_to_its_nearest_path() { + let paths = index(); + let out = extract_query_paths( + &format!("{CHAT} lines 10-20 and src/lib/task-runner-manager.ts lines 30-40"), + &paths, + 8, + ); + assert_eq!( + out.line_anchors, + vec![ + anchor(CHAT, 10, 20), + anchor("src/lib/task-runner-manager.ts", 30, 40) + ] + ); + assert_eq!(out.stripped_query, "and"); + } + + #[test] + fn leaves_line_numbers_alone_without_one_resolved_file() { + let paths = index(); + let no_path = extract_query_paths("flushQueue lines 900-1003", &paths, 8); + assert!(no_path.line_anchors.is_empty()); + assert_eq!(no_path.stripped_query, "flushQueue lines 900-1003"); + // A bare number with no line context is not a line number either. + let prose = extract_query_paths(&format!("{CHAT} retries 3 times"), &paths, 8); + assert!(prose.line_anchors.is_empty()); + assert_eq!(prose.stripped_query, "retries 3 times"); + // `generic-modal.tsx` pins two files; a line number means nothing across both. + let two_files = extract_query_paths("generic-modal.tsx:40", &paths, 8); + assert_eq!(two_files.pinned_files.len(), 2); + assert!(two_files.line_anchors.is_empty()); + } + + #[test] + fn rejects_line_numbers_no_file_has_and_dedupes_repeats() { + let paths = index(); + let anchors = |query: &str| extract_query_paths(query, &paths, 8).line_anchors; + assert!(anchors(&format!("{CHAT}:0")).is_empty()); + assert!(anchors(&format!("{CHAT} line 1000001")).is_empty()); + assert!(anchors(&format!("{CHAT} line 99999999999999999999999")).is_empty()); + assert_eq!( + anchors(&format!("{CHAT}:7 and {CHAT} line 7")), + vec![anchor(CHAT, 7, 7)] + ); + } + + const REGISTRY: &str = "src/editor/config/editorOptions.ts"; + const HELPER: &str = "src/workbench/editor/editorOptions.ts"; + + fn shared_index() -> Vec { + let mut paths = index(); + paths.extend([REGISTRY.to_string(), HELPER.to_string()]); + paths + } + + /// Path extraction with a `symbol_files` lookup over `defs`, recording every + /// symbol it is asked about. + fn narrowed(query: &str, defs: &[(&str, &[&str])]) -> (QueryPathExtraction, Vec) { + let asked = std::cell::RefCell::new(Vec::new()); + let lookup = |symbol: &str| -> Vec { + asked.borrow_mut().push(symbol.to_string()); + defs.iter() + .find(|(name, _)| *name == symbol) + .map(|(_, files)| files.iter().map(|f| f.to_string()).collect()) + .unwrap_or_default() + }; + let out = extract_query_paths_with_probes(query, &shared_index(), 8, None, Some(&lookup)); + (out, asked.into_inner()) + } + + fn set_aside(span: &str, files: &[&str]) -> QuerySetAsideMatch { + QuerySetAsideMatch { + span: span.to_string(), + files: files.iter().map(|f| f.to_string()).collect(), + } + } + + #[test] + fn pins_only_the_match_defining_a_named_symbol_and_reports_the_rest() { + let (out, asked) = narrowed( + "editorOptions.ts clampedInt cursorStyleToString", + &[ + ("clampedInt", &[REGISTRY]), + ("cursorStyleToString", &[REGISTRY]), + ], + ); + assert_eq!(out.pinned_files, vec![REGISTRY]); + assert_eq!( + out.set_aside_matches, + vec![set_aside("editorOptions.ts", &[HELPER])] + ); + assert_eq!(out.stripped_query, "clampedInt cursorStyleToString"); + // The span itself is a file name, not a symbol to look up. + assert_eq!(asked, vec!["clampedInt", "cursorStyleToString"]); + } + + #[test] + fn narrows_a_shared_kebab_stem_the_same_way() { + let (out, _) = narrowed( + "generic-modal GenericModalFooter", + &[("GenericModalFooter", &["src/y/generic-modal.tsx"])], + ); + assert_eq!(out.pinned_files, vec!["src/y/generic-modal.tsx"]); + assert_eq!( + out.set_aside_matches, + vec![set_aside("generic-modal", &["src/x/generic-modal.tsx"])] + ); + } + + #[test] + fn pins_every_match_when_no_match_or_every_match_defines_a_named_symbol() { + let (none, asked) = narrowed( + "editorOptions.ts clampedInt", + &[("clampedInt", &["src/lib/chat-manager.ts"])], + ); + assert_eq!(none.pinned_files, vec![REGISTRY, HELPER]); + assert!(none.set_aside_matches.is_empty()); + // Consulted and found nothing — not skipped. + assert_eq!(asked, vec!["clampedInt"]); + + let (both, _) = narrowed( + "editorOptions.ts EditorOptions", + &[("EditorOptions", &[HELPER, REGISTRY])], + ); + assert_eq!(both.pinned_files, vec![REGISTRY, HELPER]); + assert!(both.set_aside_matches.is_empty()); + } + + #[test] + fn a_bare_english_word_does_not_pick_a_file() { + // `options` and `close` are words as much as names: a file defining one + // says nothing about which same-named file the agent meant. + let defs: &[(&str, &[&str])] = &[ + ("options", &[HELPER]), + ("close", &["src/x/generic-modal.tsx"]), + ]; + let (words, asked) = narrowed("editorOptions.ts options", defs); + assert_eq!(words.pinned_files, vec![REGISTRY, HELPER]); + assert!(asked.is_empty()); + let (kebab, asked) = narrowed("generic-modal close behavior", defs); + assert_eq!( + kebab.pinned_files, + vec!["src/x/generic-modal.tsx", "src/y/generic-modal.tsx"] + ); + assert!(asked.is_empty()); + } + + #[test] + fn looks_a_qualified_token_up_as_written_and_reads_a_call_as_its_name() { + let (out, asked) = narrowed( + "editorOptions.ts EditorIntOption.clampedInt `cursorStyleFromString()`", + &[ + ("EditorIntOption.clampedInt", &[REGISTRY]), + ("cursorStyleFromString", &[REGISTRY]), + ], + ); + assert_eq!(out.pinned_files, vec![REGISTRY]); + assert_eq!( + asked, + vec!["EditorIntOption.clampedInt", "cursorStyleFromString"] + ); + } + + #[test] + fn resolves_a_basename_shared_past_the_ambiguity_budget_when_it_narrows() { + let (out, _) = narrowed( + "user-profile UserProfileCard", + &[("UserProfileCard", &["src/c/user-profile.tsx"])], + ); + assert_eq!(out.pinned_files, vec!["src/c/user-profile.tsx"]); + assert_eq!( + out.set_aside_matches, + vec![set_aside( + "user-profile", + &[ + "src/a/user-profile.tsx", + "src/b/user-profile.tsx", + "src/d/user-profile.tsx" + ] + )] + ); + + let (dotted, _) = narrowed( + "+page.svelte ChatScroller", + &[( + "ChatScroller", + &["src/routes/(protected)/chat-window/+page.svelte"], + )], + ); + assert_eq!( + dotted.pinned_files, + vec!["src/routes/(protected)/chat-window/+page.svelte"] + ); + assert!(dotted.unresolved_path_spans.is_empty()); + } + + #[test] + fn keeps_an_over_budget_span_ambiguous_when_the_symbols_do_not_narrow_it() { + let defs: &[(&str, &[&str])] = &[( + "PageHeader", + &[ + "src/routes/m/projects/[id]/runs/[runId]/+page.svelte", + "src/routes/m/projects/[id]/chat/[scope]/+page.svelte", + "src/routes/m/projects/[id]/+page.svelte", + "src/routes/(protected)/chat-window/+page.svelte", + ], + )]; + let (out, _) = narrowed("why do all +page.svelte PageHeader files flash", defs); + assert!(out.pinned_files.is_empty()); + assert_eq!(out.unresolved_path_spans, vec!["+page.svelte"]); + assert!(out.set_aside_matches.is_empty()); + + let (kebab, _) = narrowed("refactor the user-profile rendering", defs); + assert!(kebab.pinned_files.is_empty()); + assert_eq!(kebab.stripped_query, "refactor the user-profile rendering"); + } + + #[test] + fn binds_a_line_anchor_once_the_span_narrows_to_one_file() { + let (out, _) = narrowed( + "editorOptions.ts:1291 clampedInt", + &[("clampedInt", &[REGISTRY])], + ); + assert_eq!(out.pinned_files, vec![REGISTRY]); + assert_eq!(out.line_anchors, vec![anchor(REGISTRY, 1291, 1291)]); + } + + #[test] + fn does_not_report_a_file_the_query_also_names_by_its_full_path() { + let (out, _) = narrowed( + &format!("editorOptions.ts clampedInt {HELPER}"), + &[("clampedInt", &[REGISTRY])], + ); + assert_eq!(out.pinned_files, vec![REGISTRY, HELPER]); + assert!(out.set_aside_matches.is_empty()); + } + + #[test] + fn never_looks_anything_up_for_a_span_that_matches_one_file() { + let (out, asked) = narrowed( + "chat-manager.ts ChatManager", + &[("ChatManager", &["src/lib/chat-manager.ts"])], + ); + assert_eq!(out.pinned_files, vec!["src/lib/chat-manager.ts"]); + assert!(asked.is_empty()); + } } diff --git a/crates/codegraph-mcp/tests/golden_mcp.rs b/crates/codegraph-mcp/tests/golden_mcp.rs index 716d886..6e565de 100644 --- a/crates/codegraph-mcp/tests/golden_mcp.rs +++ b/crates/codegraph-mcp/tests/golden_mcp.rs @@ -1390,12 +1390,26 @@ fn explore_output_respects_its_stated_budget() { text.contains("Complete source for 3 files"), "the completeness signal must be present with a TRUE count:\n{text}" ); - // The exclusion must still be STATED even when no pointer line fits: five - // candidates are excluded here and the remaining budget cannot hold a - // 166-byte line, so the tail is the only honest way to say so. - assert!( - text.lines().any(|l| l == "- ... and 5 more files"), - "the unlisted count must be stated:\n{text}" + // The exclusion must still be STATED however few pointer lines fit: five + // candidates are excluded here, so the lines listed and the tail's count add + // up to five. The leftover held no 166-byte line until #2077 shortened the + // completeness note, and holds one since; the count is what must not drift. + let listed = text + .lines() + .filter(|l| l.starts_with("- src/ledger")) + .count(); + let counted = text + .lines() + .find_map(|l| { + l.strip_prefix("- ... and ") + .and_then(|rest| rest.strip_suffix(" more files")) + .and_then(|n| n.parse::().ok()) + }) + .unwrap_or(0); + assert_eq!( + listed + counted, + 5, + "every excluded file must be listed or counted:\n{text}" ); } diff --git a/crates/codegraph-store/src/queries.rs b/crates/codegraph-store/src/queries.rs index 3363184..52b7366 100644 --- a/crates/codegraph-store/src/queries.rs +++ b/crates/codegraph-store/src/queries.rs @@ -571,6 +571,49 @@ impl Store { rows.collect() } + /// Whether any node's name, qualified name, signature or docstring holds an + /// FTS token starting with `word` — the posting-list half of upstream's + /// `getExploreMissDiagnostics` (#1904). `word` must be letters/digits only. + pub fn fts_any_column_has_prefix(&self, word: &str) -> rusqlite::Result { + self.conn.query_row( + "SELECT EXISTS (SELECT 1 FROM nodes_fts WHERE nodes_fts MATCH ?1)", + params![format!( + "{{name qualified_name signature docstring}} : \"{word}\"*" + )], + |row| row.get::<_, bool>(0), + ) + } + + /// Up to `limit` names of non-file, non-import nodes whose NAME holds an FTS + /// token starting with one of `words` (letters/digits only), in index order + /// (#1904's retry candidates). + pub fn fts_name_prefix_names( + &self, + words: &[String], + limit: usize, + ) -> rusqlite::Result> { + if words.is_empty() { + return Ok(Vec::new()); + } + let pattern = format!( + "name : ({})", + words + .iter() + .map(|word| format!("\"{word}\"*")) + .collect::>() + .join(" OR ") + ); + let mut stmt = self.conn.prepare( + "SELECT n.name FROM nodes_fts JOIN nodes n ON n.rowid = nodes_fts.rowid \ + WHERE nodes_fts MATCH ?1 AND n.kind NOT IN ('file', 'import') \ + ORDER BY nodes_fts.rowid LIMIT ?2", + )?; + let rows = stmt.query_map(params![pattern, limit as i64], |row| { + row.get::<_, String>(0) + })?; + rows.collect() + } + /// Ports `getAllNodeNames` from `upstream db/queries.ts:1655-1661`. /// `SELECT DISTINCT name FROM nodes` — the candidate name set for fuzzy fallback. pub fn all_node_names(&self) -> rusqlite::Result> { diff --git a/docs/mcp.md b/docs/mcp.md index 9683731..ea15669 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -343,10 +343,15 @@ plus the call/impact graph around them. Prefer it over individual `callers`/ `callees` chains when surveying an unfamiliar area. Explicit source paths in the query are resolved before fuzzy search and pinned -to the front of the result. Quoted/backticked paths, `./`, Windows separators, -`:123`, and `#L12` are normalized; exact matches precede segment-aligned suffix +to the front of the result. Quoted/backticked paths, `./`, and Windows +separators are normalized; exact matches precede segment-aligned suffix matches. Extensionless kebab basenames are accepted only when they resolve to at -most three indexed files, so ordinary hyphenated prose remains prose. The +most three indexed files, so ordinary hyphenated prose remains prose. When a +path or basename matches several files and the query also names symbols in a +code shape (camelCase, PascalCase, snake_case, `$` or qualified), only the +matches defining one of them are pinned, and the summary names the files set +aside; a basename shared by more than three files resolves the same way when +the named symbols narrow it to three or fewer. The resolver examines at most eight path spans, drops at most eight leading segments, pins at most eight files (also bounded by `maxFiles`), and reports at most four unresolved explicit paths. Resolved or clearly missing explicit paths @@ -354,6 +359,28 @@ are removed from the normal query, preventing route parameters and basenames from becoming noisy symbol seeds. Pinned files survive low-score filtering and receive a protected source budget. +Line references on a path that resolved to exactly one file are kept as +anchors: `compiler.py:776`, `foo.ts:12-40`, `foo.ts#L88-L120`, and prose ranges +bound to the nearest such path (`compiler.py lines 900-1003`, +`L900-L1003 in compiler.py`, `lines 900 to 1003`). A bare number counts only +after `line`/`lines` or directly after the path, and numbers above 1,000,000 +are ignored. A single-line anchor selects the innermost method, function, or +component enclosing it; a range, or a line no callable encloses (with 15 lines +either side), renders as that span. Those, and a qualified name +(`SQLCompiler.as_sql`, `Engine::ServeHTTP`) with at most three non-test +definitions, are **exact targets**: they lead the blast radius, their files rank +ahead of incidental files, and their clusters render first. An exact body is +returned whole when it fits the file's budget, and otherwise from its own head +plus windows on the anchored line and on its calls into the other symbols the +query names. A named neighbour that no longer fits beside it is listed in the +file header instead. + +Each file's source is held to a per-file ceiling of 1.5 × the tier's per-file +budget. A file the query names (by path, by an exact target, or by a symbol it +defines) that this ceiling clipped is rendered again into whatever budget the +rest of the response left unspent, so a question about one file can use the +whole response while every other file keeps exactly the section it was given. + Explore also resolves prose and camelCase query segments against indexed symbol names, then merges the resulting callable, Variable, and Constant names as dampened exact seeds. This recovers names such as `feedAtBottom` from "feed diff --git a/docs/upstream-sync/V1_6_1_AUDIT_2026-09-30.md b/docs/upstream-sync/V1_6_1_AUDIT_2026-09-30.md index 798d931..fb871cc 100644 --- a/docs/upstream-sync/V1_6_1_AUDIT_2026-09-30.md +++ b/docs/upstream-sync/V1_6_1_AUDIT_2026-09-30.md @@ -79,29 +79,44 @@ KEEP-RUST divergences taken in this PR: ## MCP, CLI, installer, lifecycle and explore -| Issue / PR | Upstream | Behavior | Disposition | State | -| ------------------------------------------------------------------------------------ | ---------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| #1832 | `cba44c9` | only a prompt that is entirely a task-notification envelope is skipped | CORRECT | landed in #275 | -| #1967 | `94ae04d` | dynamic-import boundaries: template or concatenated specifiers are runtime imports | PORT | landed in #279 | -| #1964 | `a9c8808` | a failed sync keeps the pending paths and owed full scan and retries | PORT | landed in #279, with the index-lease timeout classified as contention | -| #1959 | `2268a09` | restart a degraded watcher, RECOVERING/DISABLED banners, refuse drifted answers while degraded | PORT | landed in #279 as an in-loop recovery (the watcher keeps collecting events and retries a full reconcile every 30 s), with the #876 banner never ported before | -| #1654 | `4fd3e6d` | `como`/`wie` leave the unconditional structural keyword list | PORT | landed in #279 | -| #1870 | `98718cf` | installer `detect()` is read-only; backups only when a config is rewritten | PORT | landed in #279 | -| #1839 (CLI) | `fa25883` | callers/callees JSON carries each node's `relationships` | PORT | landed in #278, with callers/callees following `instantiates` edges (upstream #774/#804, never ported before) | -| #1895 | `ee97a57` | MCP root discovery ignores a `codegraph.db` without a schema | PORT (CLI ALREADY-HAVE) | landed in #279 | -| #995 | `c13bd08` | WSL `/mnt/` defaults to `.codegraph-wsl` and maps SQLITE_IOERR to guidance | PORT | landed in #279 | -| #2053 / #1773 | `0a9f3de`, `c405ac5` | bounded retry when Windows sharing violations block `daemon.pid` replacement | PORT | landed in #279 | -| #1294 | `d3671fa` | `install.sh` under MINGW/MSYS/CYGWIN points at the PowerShell installer | PORT | landed in #279 | -| #1904, PR #2062, PR #2063, PR #2068, PR #2071, PR #2077 | `9ad6ee9`, `adad840`, `47840a6`, `724b5de`, `290e03f`, `003305f` | explore: empty-result lexical diagnostics, cluster budget order, line anchors as exact targets, single-file budget, same-basename pins, honest completion notes | PORT | planned | -| #1834, #1831, #1830, #1968, #1372, #1361, #1835, #1963 / #1356, #1325, #1902 | various | already implemented by the port | ALREADY-HAVE | — | -| #770 | `c897daf` | watch symlinked directories | DEFER until the scan's symlink policy is settled (the scan does not follow symlinked directories) | — | -| telemetry, npm/fnm, Node timers, `FILE_NOTIFY_CHANGE_LAST_ACCESS`, test-only commits | various | no Rust analogue | N/A | — | +| Issue / PR | Upstream | Behavior | Disposition | State | +| ------------------------------------------------------------------------------------ | -------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| #1832 | `cba44c9` | only a prompt that is entirely a task-notification envelope is skipped | CORRECT | landed in #275 | +| #1967 | `94ae04d` | dynamic-import boundaries: template or concatenated specifiers are runtime imports | PORT | landed in #279 | +| #1964 | `a9c8808` | a failed sync keeps the pending paths and owed full scan and retries | PORT | landed in #279, with the index-lease timeout classified as contention | +| #1959 | `2268a09` | restart a degraded watcher, RECOVERING/DISABLED banners, refuse drifted answers while degraded | PORT | landed in #279 as an in-loop recovery (the watcher keeps collecting events and retries a full reconcile every 30 s), with the #876 banner never ported before | +| #1654 | `4fd3e6d` | `como`/`wie` leave the unconditional structural keyword list | PORT | landed in #279 | +| #1870 | `98718cf` | installer `detect()` is read-only; backups only when a config is rewritten | PORT | landed in #279 | +| #1839 (CLI) | `fa25883` | callers/callees JSON carries each node's `relationships` | PORT | landed in #278, with callers/callees following `instantiates` edges (upstream #774/#804, never ported before) | +| #1895 | `ee97a57` | MCP root discovery ignores a `codegraph.db` without a schema | PORT (CLI ALREADY-HAVE) | landed in #279 | +| #995 | `c13bd08` | WSL `/mnt/` defaults to `.codegraph-wsl` and maps SQLITE_IOERR to guidance | PORT | landed in #279 | +| #2053 / #1773 | `0a9f3de`, `c405ac5` | bounded retry when Windows sharing violations block `daemon.pid` replacement | PORT | landed in #279 | +| #1294 | `d3671fa` | `install.sh` under MINGW/MSYS/CYGWIN points at the PowerShell installer | PORT | landed in #279 | +| #1711 | `d983f73` | a file trim names the symbols it elided, in gap markers and the file header | PORT | landed in #281 | +| PR #2057 | `66aebd1` | three explore regressions: gap names never push source out; interface members are no prose corroboration; dispatch notes need the supertype to declare the member | PORT (gap names), N/A (the other two) | gap names landed with #1711 in #281; the port has neither the prose-corroboration seeding nor dispatch-site notes | +| #1904 | `9ad6ee9` | an empty explore explains lexical matching, lists unmatched words and suggests indexed names | PORT | landed in #281, with the query-time name segments standing in for `name_segment_vocab` | +| PR #2062 | `adad840` | a cluster's incidental members wait until lower-ranked named members are paid | PORT | landed in #281 | +| PR #2064 | `7506dbf` | test: named symbols in a pinned file past the node cap | PORT (test only) | landed in #281; holds without #2062 here, since the port's head cluster has no edge-line members to out-densify the named one | +| PR #2063 | `47840a6` | line anchors (`:776`, `lines 900-1003`) and exact targets (qualified names, anchored callables, anchored spans) | PORT | landed in #281; the per-symbol focused view and the Flow spine it also tiers have no counterpart (F10) | +| PR #2068 | `724b5de` | a named file may use the budget the response left unspent | PORT (merged-spine half N/A) | landed in #281 as a second render pass into the unspent budget; the port has no flow spine | +| PR #2071 | `290e03f` | a basename several directories share pins only the matches defining the named symbols | PORT | landed in #281 | +| PR #2077 | `003305f` | "complete source" only for complete sections; trimmed files and elided symbols named; no Read offer | PORT | landed in #281; the lost-pointer and cut notes have no counterpart (the port reserves its epilogue up front) | +| CG-38 (v1.6.0) | `89c53dd` | a windowed cluster's head gets the whole room unless a focus is left uncovered | CORRECT | landed in #281; the CG-38 port held 40% back whenever a focus existed | +| #1834, #1831, #1830, #1968, #1372, #1361, #1835, #1963 / #1356, #1325, #1902 | various | already implemented by the port | ALREADY-HAVE | — | +| #770 | `c897daf` | watch symlinked directories | DEFER until the scan's symlink policy is settled (the scan does not follow symlinked directories) | — | +| telemetry, npm/fnm, Node timers, `FILE_NOTIFY_CHANGE_LAST_ACCESS`, test-only commits | various | no Rust analogue | N/A | — | A Rust-only defect found by this pass: the watcher treats only `locked`/`busy` errors as lock contention, but an index-lease timeout reports `timed out acquiring permanent index lock`, so a long foreground `index` makes the watcher drop the batch and count a failure. It landed with #1964 in #279. +Two more turned up while porting the explore rows. The CG-38 port windowed a +cluster with 60% of its room on the head whenever it held a focus, even a focus +the head already showed, where upstream gives the head the whole room first; +this was never recorded as a divergence and is corrected in #281. And the +completeness count read "1 files"; #2077 gives it upstream's wording. + ## UI family (re-triage) The browser viewer (`ui/**`, `src/ui-server/**`, `codegraph ui`) is an explicit diff --git a/reference/golden/mcp/initialize.json b/reference/golden/mcp/initialize.json index 4e6d64d..237a23b 100644 --- a/reference/golden/mcp/initialize.json +++ b/reference/golden/mcp/initialize.json @@ -24,7 +24,7 @@ "name": "codegraph", "version": "0.12.0" }, - "instructions": "# Codegraph — code intelligence over an indexed knowledge graph\n\nCodegraph is a SQLite knowledge graph of every symbol, edge, and file in\nthe workspace — pre-computed structure you would otherwise re-derive by\nreading files (cached intelligence: thousands of parse/trace decisions you\ndon't pay to re-reason each run). Reads are sub-millisecond; the index lags\nwrites by ~1s through the file watcher. Reach for it BEFORE *and* while\nwriting or editing code — not just for questions: one call returns the\nverbatim source PLUS who calls it and what it affects, so you edit with the\nblast radius in view. More accurate context, in far fewer tokens and\nround-trips than reading files yourself.\n\n## Use codegraph instead of reading files — for questions AND edits\n\nWhether you're answering \"how does X work\" or implementing a change (fixing\na bug, adding a feature), reach for codegraph before you Read. For\nunderstanding, answer DIRECTLY — usually with ONE `codegraph_explore` call.\n`codegraph_explore` takes either a natural-language question or a bag of\nsymbol/file names and returns the verbatim source of the relevant symbols\ngrouped by file, so it is Read-equivalent and most often the ONLY\ncodegraph call you need. Codegraph IS the pre-built search index — so\ndelegating the lookup to a separate file-reading sub-task/agent, or\nrunning your own grep + read loop, repeats work codegraph already did and\ncosts more for the same answer. Reach for raw Read/Grep only to confirm a\nspecific detail codegraph didn't cover. A direct codegraph answer is\ntypically one to a few calls; a grep/read exploration is dozens.\n\n## Tool selection by intent\n\n- **Almost any question — \"how does X work\", architecture, a bug, \"what/where is X\", or surveying an area** → `codegraph_explore` (PRIMARY — call FIRST; ONE capped call returns the verbatim source of the relevant symbols grouped by file; most often the ONLY call you need)\n- **\"How does X reach/become Y? / the flow / the path from X to Y\"** → `codegraph_explore`, naming the symbols that span the flow (e.g. `mutateElement renderScene`) — it surfaces the call path among them, including dynamic-dispatch hops (callbacks, React re-render, JSX children) grep can't follow\n- **\"What is the symbol named X?\" (just its location)** → `codegraph_search`\n- **\"What calls this?\" / \"What does this call?\" / \"What would changing this break?\"** → `codegraph_callers` / `codegraph_callees` / `codegraph_impact`\n- **Reading a source FILE (any time you'd use the `Read` tool)** → `codegraph_node` with a `file` path and no `symbol`. It returns the file's **current source with line numbers — the same `\\t` shape `Read` gives you, safe to `Edit` from** — narrowable with `offset`/`limit` exactly like `Read`, PLUS a one-line note of which files depend on it. Same bytes as `Read`, faster (served from the index), with the blast radius attached. Use it **instead of `Read`** for indexed source files; fall back to `Read` only for what codegraph doesn't index (configs, docs). Pass `symbolsOnly: true` for just the file's structure.\n- **About to read or edit a symbol you can name (or whose exact node ID you have)** → `codegraph_node` with that `symbol` (SECONDARY — the after-explore depth tool): the verbatim source (`includeCode: true`) PLUS its caller/callee trail, so before changing it you see what calls it and what your edit would break. For an OVERLOADED name it returns EVERY matching definition's body in one call, so you never Read a file to find the right overload\n- **\"What's in directory X?\"** → `codegraph_files`\n- **\"Is the index ready / what's its size?\"** → `codegraph_status`\n\n## Common chains\n\n- **Flow / \"how does X reach Y\"**: ONE `codegraph_explore` with the symbol names spanning the flow — it surfaces the call path among them (riding dynamic-dispatch hops) AND returns their source. No need to reconstruct the path with `codegraph_search` + `codegraph_callers`.\n- **Onboarding / understanding any area**: ONE `codegraph_explore` is usually the whole answer. Only follow up — `codegraph_node` for a specific symbol — if something is still unclear.\n- **Refactor planning**: `codegraph_search` → `codegraph_callers` → `codegraph_impact`. The blast-radius answer comes from impact, not from walking callers manually.\n- **Debugging a regression**: `codegraph_callers` of the suspected symbol; widen with `codegraph_impact` if an unexpected call appears.\n\n## Anti-patterns\n\n- **Trust codegraph's results — don't re-verify them with grep.** They come from a full AST parse; re-checking with grep is slower, less accurate, and wastes context.\n- **Don't grep first** when looking up a symbol by name — `codegraph_search` is faster and returns kind + location + signature.\n- **Don't chain `codegraph_search` + `codegraph_node`** to understand an area — ONE `codegraph_explore` returns the relevant symbols' source together in a single round-trip.\n- **Don't loop `codegraph_node` over many symbols** — one `codegraph_explore` call returns them all grouped by file, while each separate call re-reads the whole context and costs far more. Use `codegraph_node` for a single symbol.\n- **Don't reach for the `Read` tool on an indexed source file** — `codegraph_node` with a `file` reads it for you (same `\\t` source, `offset`/`limit` like Read, faster, with its blast radius), and with a `symbol` it returns the source plus the caller/callee trail. Reach for raw `Read` only for what codegraph doesn't index (configs, docs) or when the staleness banner flags a file as pending re-index.\n- **After editing, check the staleness banner.** When a tool response starts with \"⚠️ Some files referenced below were edited since the last index sync…\", the listed files are pending re-index — Read those specific files for accurate content. Every file NOT in that banner is fresh, so still trust codegraph. `codegraph_status` also lists pending files under \"Pending sync\".\n\n## Limitations\n\n- If a tool reports the project isn't initialized, `.codegraph/` doesn't exist yet — offer to run `codegraph init .` to build the index.\n- Index lags file writes by ~1 second.\n- Cross-file resolution is best-effort name matching; ambiguous calls may return multiple candidates.\n- No live correctness validation — that's still the TypeScript compiler / test suite / linter's job. Codegraph supplements those with structural context they don't have.\n\n## Supported languages\n\nCodegraph extracts these 38 languages (typescript, javascript, tsx, jsx, arkts, python, go, rust, java, c, cpp, csharp, razor, php, ruby, swift, kotlin, dart, svelte, vue, astro, liquid, pascal, scala, lua, luau, objc, r, solidity, nix, terraform, erlang, cfml, yaml, twig, xml, properties, gdscript). A file outside this set is not indexed, so no codegraph tool can see it — reach for Read there.\n" + "instructions": "# Codegraph — code intelligence over an indexed knowledge graph\n\nCodegraph is a SQLite knowledge graph of every symbol, edge, and file in\nthe workspace — pre-computed structure you would otherwise re-derive by\nreading files (cached intelligence: thousands of parse/trace decisions you\ndon't pay to re-reason each run). Reads are sub-millisecond; the index lags\nwrites by ~1s through the file watcher. Reach for it BEFORE *and* while\nwriting or editing code — not just for questions: one call returns the\nverbatim source PLUS who calls it and what it affects, so you edit with the\nblast radius in view. More accurate context, in far fewer tokens and\nround-trips than reading files yourself.\n\n## Use codegraph instead of reading files — for questions AND edits\n\nWhether you're answering \"how does X work\" or implementing a change (fixing\na bug, adding a feature), reach for codegraph before you Read. For\nunderstanding, answer DIRECTLY — usually with ONE `codegraph_explore` call.\n`codegraph_explore` takes either a natural-language question or a bag of\nsymbol/file names and returns the verbatim source of the relevant symbols\ngrouped by file, so it is Read-equivalent and most often the ONLY\ncodegraph call you need. Codegraph IS the pre-built search index — so\ndelegating the lookup to a separate file-reading sub-task/agent, or\nrunning your own grep + read loop, repeats work codegraph already did and\ncosts more for the same answer. Reach for raw Read/Grep only to confirm a\nspecific detail codegraph didn't cover. A direct codegraph answer is\ntypically one to a few calls; a grep/read exploration is dozens.\n\n## Tool selection by intent\n\n- **Almost any question — \"how does X work\", architecture, a bug, \"what/where is X\", or surveying an area** → `codegraph_explore` (PRIMARY — call FIRST; ONE capped call returns the verbatim source of the relevant symbols grouped by file; most often the ONLY call you need)\n- **\"How does X reach/become Y? / the flow / the path from X to Y\"** → `codegraph_explore`, naming the symbols that span the flow (e.g. `mutateElement renderScene`) — it surfaces the call path among them, including dynamic-dispatch hops (callbacks, React re-render, JSX children) grep can't follow\n- **\"What is the symbol named X?\" (just its location)** → `codegraph_search`\n- **\"What calls this?\" / \"What does this call?\" / \"What would changing this break?\"** → `codegraph_callers` / `codegraph_callees` / `codegraph_impact`\n- **Reading a source FILE (any time you'd use the `Read` tool)** → `codegraph_node` with a `file` path and no `symbol`. It returns the file's **current source with line numbers — the same `\\t` shape `Read` gives you, safe to `Edit` from** — narrowable with `offset`/`limit` exactly like `Read`, PLUS a one-line note of which files depend on it. Same bytes as `Read`, faster (served from the index), with the blast radius attached. Use it **instead of `Read`** for indexed source files; fall back to `Read` only for what codegraph doesn't index (configs, docs). Pass `symbolsOnly: true` for just the file's structure.\n- **About to read or edit a symbol you can name (or whose exact node ID you have)** → `codegraph_node` with that `symbol` (SECONDARY — the after-explore depth tool): the verbatim source (`includeCode: true`) PLUS its caller/callee trail, so before changing it you see what calls it and what your edit would break. For an OVERLOADED name it returns EVERY matching definition's body in one call, so you never Read a file to find the right overload\n- **\"What's in directory X?\"** → `codegraph_files`\n- **\"Is the index ready / what's its size?\"** → `codegraph_status`\n\n## Common chains\n\n- **Flow / \"how does X reach Y\"**: ONE `codegraph_explore` with the symbol names spanning the flow — it surfaces the call path among them (riding dynamic-dispatch hops) AND returns their source. No need to reconstruct the path with `codegraph_search` + `codegraph_callers`.\n- **Onboarding / understanding any area**: ONE `codegraph_explore` is usually the whole answer. Only follow up — `codegraph_node` for a specific symbol — if something is still unclear.\n- **Refactor planning**: `codegraph_search` → `codegraph_callers` → `codegraph_impact`. The blast-radius answer comes from impact, not from walking callers manually.\n- **Debugging a regression**: `codegraph_callers` of the suspected symbol; widen with `codegraph_impact` if an unexpected call appears.\n\n## Anti-patterns\n\n- **Trust codegraph's results — don't re-verify them with grep.** They come from a full AST parse; re-checking with grep is slower, less accurate, and wastes context.\n- **Don't grep first** when looking up a symbol by name — `codegraph_search` is faster and returns kind + location + signature.\n- **Don't chain `codegraph_search` + `codegraph_node`** to understand an area — ONE `codegraph_explore` returns the relevant symbols' source together in a single round-trip.\n- **Don't loop `codegraph_node` over many symbols** — one `codegraph_explore` call returns them all grouped by file, while each separate call re-reads the whole context and costs far more. Use `codegraph_node` for a single symbol.\n- **Don't reach for the `Read` tool on an indexed source file** — `codegraph_node` with a `file` reads it for you (same `\\t` source, `offset`/`limit` like Read, faster, with its blast radius), and with a `symbol` it returns the source plus the caller/callee trail. Reach for raw `Read` only for what codegraph doesn't index (configs, docs) or when the staleness banner flags a file as pending re-index.\n- **After editing, check the staleness banner.** When a tool response starts with \"⚠️ Some files referenced below were edited since the last index sync…\", the listed files are pending re-index — Read those specific files for accurate content. Every file NOT in that banner is fresh, so still trust codegraph. `codegraph_status` also lists pending files under \"Pending sync\".\n\n## Limitations\n\n- If a tool reports the project isn't initialized, `.codegraph/` doesn't exist yet — offer to run `codegraph init .` to build the index.\n- Index lags file writes by ~1 second.\n- Cross-file resolution is best-effort name matching; ambiguous calls may return multiple candidates.\n- Explore matches names and indexed code words lexically, not by meaning; an empty result reports word matches and may suggest indexed candidate names to retry with `codegraph_explore`.\n- No live correctness validation — that's still the TypeScript compiler / test suite / linter's job. Codegraph supplements those with structural context they don't have.\n\n## Supported languages\n\nCodegraph extracts these 38 languages (typescript, javascript, tsx, jsx, arkts, python, go, rust, java, c, cpp, csharp, razor, php, ruby, swift, kotlin, dart, svelte, vue, astro, liquid, pascal, scala, lua, luau, objc, r, solidity, nix, terraform, erlang, cfml, yaml, twig, xml, properties, gdscript). A file outside this set is not indexed, so no codegraph tool can see it — reach for Read there.\n" } } }