diff --git a/.ccc/ccc-surface.json b/.ccc/ccc-surface.json new file mode 100644 index 0000000..0d3154b --- /dev/null +++ b/.ccc/ccc-surface.json @@ -0,0 +1,8 @@ +{ + "schema": "ccc-surface/1", + "name": "codecache", + "generated": "20260921-21-51-30", + "languages": ["rust", "typescript"], + "provides": [], + "consumes": [] +} diff --git a/AGENTS.md b/AGENTS.md index aa63541..58f3f3f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,9 +1,9 @@ # AGENTS.md -This repo ships a ContextCodeCache - an in-memory code map served over MCP. Use it as the entry point for +This repo ships a CodeCaChe - an in-memory code map served over MCP. Use it as the entry point for everything you do here. -`ccc serve` listens on `http://127.0.0.1:6767/mcp` by default, but the VS Code extension starts its own +`ccc run` listens on `http://127.0.0.1:6767/mcp` by default, but the VS Code extension starts its own analyser per window on an ephemeral port so it never clashes with yours. On every start it publishes that port to `.mcp.json` (Claude Code) and `.vscode/mcp.json` (Copilot), so both auto-discover the server without being told a port. Both files are generated and gitignored - read the URL from them rather than assuming 6767. @@ -13,7 +13,7 @@ to pick the new one up. - bash/grep shouldn't be used for understanding the project - IF `ccc` tool calls are unable to find a term your are searching for; stop the session and respond with `CCC: unable to find in ccc using calls: [calls]` - Every interaction: use `ccc` tool calls to gather information about the source of this project. -- All thinking, navigation, and questions about the codebase go through the MCP server tools: (index, find, references, dependencies, file, notes, changes, test_triggers, test_targets, lints, hot, services refresh) -- When I ask to *see* the analysis, call `insights` - it opens the insights UI in my browser (needs `ccc serve --html`) +- All thinking, navigation, and questions about the codebase go through the MCP server tools: (index, find, references, dependencies, vulnerabilities, security, file, notes, changes, test_triggers, test_targets, lints, hot, services refresh) +- When I ask to *see* the analysis, call `insights` - it opens the insights UI in my browser - Make code changes in the source, never to the in-memory map. - After changing tracked source call the `ccc` tool with `refresh` to ensure you have the latest changes in-memory. diff --git a/Cargo.lock b/Cargo.lock index 27885ea..8c57bcd 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -203,7 +203,7 @@ checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" [[package]] name = "codecache" -version = "1.4.2" +version = "1.4.4" dependencies = [ "anyhow", "chrono", @@ -222,6 +222,7 @@ dependencies = [ "tree-sitter-go", "tree-sitter-javascript", "tree-sitter-odin", + "tree-sitter-proto", "tree-sitter-python", "tree-sitter-rust", "tree-sitter-typescript", @@ -721,6 +722,16 @@ dependencies = [ "tree-sitter-language", ] +[[package]] +name = "tree-sitter-proto" +version = "0.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9c93b6f1ed20de442e900eb636f2176af6063953dfaa9f76bf168dd3b490a3a1" +dependencies = [ + "cc", + "tree-sitter-language", +] + [[package]] name = "tree-sitter-python" version = "0.25.0" diff --git a/Cargo.toml b/Cargo.toml index dd75a20..619d95f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,9 +1,9 @@ [package] name = "codecache" -version = "1.4.2" +version = "1.4.4" edition = "2021" rust-version = "1.77" -description = "ContextCodeCache (.ccc) generates an agent friendly local source map" +description = "CodeCaChe (.ccc) generates an agent friendly local source map" license = "MIT" # packages the VS Code extension into dist/ alongside the binary @@ -41,6 +41,7 @@ tree-sitter-c = "0.24" tree-sitter-c-sharp = "0.23" tree-sitter-zig = "1.1" tree-sitter-odin = "1.3" +tree-sitter-proto = "0.6" crossbeam-epoch = "0.9.20" [profile.release] diff --git a/README.md b/README.md index b65e3f9..f804a1e 100644 --- a/README.md +++ b/README.md @@ -1,11 +1,11 @@ -

ContextCodeCache Logo

+

CodeCaChe Logo

[![Release CodeCaChe](https://github.com/colwill/ccc/actions/workflows/ccc-release.yaml/badge.svg)](https://github.com/colwill/ccc/actions/workflows/ccc-release.yaml) # CodeCaChe (`ccc`) -CodeCaChe provides insight into your code and improves developer experience by: +CodeCaChe tells you, and your AI agent, what a change touches before you commit it - the functions, tests, services and cross-service contracts it reaches - by: - highlighting which tests will be ran with your changes @@ -24,7 +24,7 @@ graph, and marker notes (TODO/FIXME/...). It is designed to give engineers a always-fresh index of a project, the latest changes, how those changes impact tests or other branches (compare working branch against any other branch). In addition it also provides language models a local MCP server for an always up-to-date map of your codebase, dependencies, call-graph and cross-service calls. -Supports: `C99`, `C++ (20 except modules)`, `C#`, `Rust`, `Go`, `Python`, `Zig`, `Odin`, `TypeScript` +Supports: `C99`, `C++ (20 except modules)`, `C#`, `Rust`, `Go`, `Python`, `Zig`, `Odin`, `TypeScript`, `Protobuf` & `JavaScript` - see [`LANGUAGES.md`](docs/LANGUAGES.md) for what each one resolves. ## Table of content @@ -62,11 +62,11 @@ Supports: `C99`, `C++ (20 except modules)`, `C#`, `Rust`, `Go`, `Python`, `Zig`, cargo build --release && ./target/release/ccc -- install ``` -2. **Initialise `ccc changes --init` to generate the basic `.ccc/map.json`** +2. **Initialise `ccc init` to generate the basic `.ccc/map.json`** - a. (recommended) edit dependency map `.ccc/map.json` to include service locations and dependencies + a. (recommended) edit dependency map `.ccc/map.json` to include service locations and cross-service calls -3. **Start local MCP `ccc serve --html`** +3. **Start local MCP `ccc run`** a. (recommended) visit `http://127.0.0.1:6767/insights` for Insights UI @@ -84,30 +84,30 @@ Supports: `C99`, `C++ (20 except modules)`, `C#`, `Rust`, `Go`, `Python`, `Zig`, ## Usage ```sh -ccc changes [PATH] --telemetry # changes vs the base branch: services to test, dependencies, otel -ccc check [PATH] --format json # exit non-zero if .ccc is stale - for CI -ccc tokenize [PATH] # pre-encode an existing .ccc into tokens.bin + tokens.json -ccc deps [PATH] # just the dependency delta of that report, for CI (JSON) -ccc prompts [PATH] # which claude/copilot request produced each change (JSON) -ccc serve [PATH] --html # MCP server and optional insights UI: agents query the in-memory map -ccc export [PATH] # publish what this project serves/calls, for other repos -ccc insights [PATH] --html # the insights analysis as JSON (call graph, triggers, lints) -ccc sast [PATH] # security findings; defaults to non-zero on a high finding -ccc audit [PATH] # resolve lockfiles and check against the OSV advisory db -ccc install [--dir] # install the ccc binary onto your PATH (Linux) -ccc scan [PATH] --tokens # regen PATH/.ccc (PATH defaults to ".") (opt: output token stream) +ccc run # Runs local in-memory map, MCP server and insights UI +ccc init # Generate basic `.ccc/map.json` and `.ccc/surface.json` (prev ccc-surface.json) +ccc changes [PATH] --telemetry # Changes vs base ref (services to test for CT) +ccc tokenize # Encode in-memory map of project into tokens.bin + tokens.json +ccc deps [PATH] # Dependency delta (for CI as JSON) +ccc prompts [PATH] # Which requests produced specific changes (JSON) +ccc insights [PATH] --html # The insights analysis as JSON (call graph, triggers, lints) +ccc sast [PATH] # Security findings, defaults to non-zero on a high finding +ccc audit [PATH] # Resolve lockfiles and check against the OSV advisory db +ccc install [--dir] # Install the ccc binary onto your PATH (Linux) +ccc scan [PATH] --dir= --tokens # Parse tree and report map (legacy) ``` +`ccc run` builds the project map in memory and makes it queryable by MCP via tool calls and viewable by the local web UI @ `:6767/insights`. + ## Insights -The command `(ccc serve --html)` starts the MCP server with the insights UI on `http://localhost:6767/insights`. It is disabled by default and fetches -`/insights.json` from the running server, so it tracks the in-memory ccc map at runtime. +The command `(ccc run)` starts the MCP server with the insights UI on `http://localhost:6767/insights`. and fetches `/insights.json` from the running server, so it tracks the in-memory ccc map at runtime. ```sh -ccc serve --html # then open http://127.0.0.1:6767/insights -curl -s localhost:6767/insights.json # the same data, for scripting -ccc insights # the same analysis as JSON, no server -ccc insights --html page.html # insights as a single page, for static hosting +ccc run # then open http://127.0.0.1:6767/insights +curl -s localhost:6767/insights.json # the same data, for other consumers +ccc insights # same JSON data as above via direct command +ccc insights --html page.html # output format is html, as a single page app ``` ## Extension @@ -129,7 +129,7 @@ The `.ccc/map.json` file is used to hint to ccc where to find dependencies, for "billing": ["apps/billing/**", "libs/money/**"], "gateway": ["apps/gateway/**"] }, - "deps": { + "relatives": { "gateway": ["auth"] // gateway calls auth over HTTP, so declare it! }, "externals": { @@ -138,6 +138,8 @@ The `.ccc/map.json` file is used to hint to ccc where to find dependencies, for } ``` +`relatives` are the declared relationships: service to service here, or service to a peer under `externals`. + ## Externals Calls do not stop at your project. `externals` names peer repositories - a sibling checkout, another @@ -159,7 +161,7 @@ func Charge(account string, amount int) error { ... } ``` Matching keys become real edges of the service graph, with a file and line at each end, whatever -language each side is written in. Publish a surface for others to consume with `ccc export`. +language each side is written in. Publish a surface for others to consume with `ccc init`. See [EXTERNALS.md](docs/EXTERNALS.md). @@ -182,14 +184,14 @@ Trailing prose after the marker is allowed, so a skip can say why. ## AGENTS.md -#### Note: If you're not using `ccc serve`, you can generate a `.ccc` directory using `ccc scan` and then add a block to your AGENTS.md file to scan the `.ccc` directory instead. +#### Note: If you're not using `ccc serve`, you can generate a `.ccc` directory using `ccc scan --dir` and then add a block to your AGENTS.md file to scan the `.ccc` directory instead. (recommended) For those using `ccc serve` and the MCP tools; add the following block to an AGENTS.md file at the root of your project - agents that read an [`AGENTS.md`](https://agents.md) at the repo root pick this up automatically e.g. Copilot, Claude, Cursor etc. ```md # AGENTS.md -This repo has a ContextCodeCache - a generated in-memory code map served over MCP at `http://127.0.0.1:6767/mcp`. Use it +This repo has a CodeCaChe - a generated in-memory code map served over MCP at `http://127.0.0.1:6767/mcp`. Use it as the entry point for everything you do here. - no bash, grep or sed usage for exploring the project diff --git a/docs/DEPS.md b/docs/DEPS.md new file mode 100644 index 0000000..1689c0a --- /dev/null +++ b/docs/DEPS.md @@ -0,0 +1,274 @@ +# Dependency changes + +`ccc audit` answers *what do we depend on right now, and is any of it vulnerable*. It never looks at +git. This document specifies the other half — *what did this branch do to our dependencies* — carried +in the `ccc changes` report by default, with JSON output and `--markdown` for agents. + +The question worth answering is not "did `Cargo.lock` change". Today a lockfile edit already shows up +in `changes` as a touched path with no functions and no detail, which is true and useless. The +question is: **which packages entered the tree, which left, which moved version, and did this branch +introduce an advisory that was not there before.** + +## Scope + +In, for v1: + +- added / removed / version-changed packages between a base ref and the head side, per ecosystem +- direct↔transitive and dev↔runtime transitions, which change whether a package ships +- advisories **introduced** and **resolved** by the change, not the standing set +- manifest-vs-lockfile drift: a manifest edited without its lockfile being regenerated +- `ccc changes`, `ccc deps`, `ccc run` + `/deps.json`, and the `deps` MCP tool + +Out, for v1: + +- per-entrypoint reachability (`go list -deps ./cmd/x` style). ccc resolves whole-module closures + from lockfiles; narrowing them to what one binary reaches needs a build, not a parse. +- license and provenance deltas +- the editor surface. The payload below is shaped for it — see *Editor* at the end — but no marks + are drawn in v1. + +`ccc audit` is unchanged: it stays the current-state command, with no `--base`. + +## Command surface + +The delta hangs off `changes`, not `audit`, because `changes` is the command that already owns a base +ref, `--worktree`, and the report every other surface consumes. + +``` +ccc changes [PATH] [--base ] [--worktree] [--fail-introduced] [--markdown] +ccc deps [PATH] [--base ] [--worktree] [--fail-introduced] [--markdown] [--format text|json] +ccc run [PATH] +``` + +**Nothing here is opt-in any more.** `ccc changes` computes the delta and puts it in the report; +`ccc run` answers `/deps.json` and advertises the `deps` MCP tool. That is affordable because of +the short-circuit below: a branch that touched no manifest or lockfile returns before a git blob is +read or a packet is sent, which is most branches. `--deps` is still accepted on both commands so the +pipelines and editor integrations that spell it out keep working, and on `changes` it still reads as +"the dependency section" beside `--markdown` — but it turns nothing on. + +`ccc deps` is the same delta with none of the change set around it: it prints the `deps` object +`changes` nests, so a CI step that only gates on dependencies pipes it straight in without a +`jq .deps` in the middle. `--base` and `--worktree` resolve identically on both — one shared +`changes::deps_report`, which `/deps.json` already answers from — so the two commands cannot disagree +about what the branch did. + +The one caller that still says no is `insights`, which builds a change set on every file-watch tick +and does not draw the delta. That is why `ChangesOptions.deps` survives as an internal knob even +though no CLI path sets it to false. + +| flag | effect | +|---|---| +| `--deps` | accepted and ignored on `changes` and `run`; the delta is computed and served either way | +| `--markdown` | render one section as markdown for an agent: the dependency delta, or the metric delta with `--telemetry` | +| `--fail-introduced` | exit non-zero when the change introduces an advisory; needs the network, so it fails loudly if OSV was unreachable rather than passing quietly | +| `--offline` | *not* added here — the delta resolves without OSV whenever `assess()` cannot reach the network, and says so, exactly as `audit` does | + +`--worktree` picks the head side, and nothing else changes: without it the head side is `HEAD` (the +committed view CI wants), with it the head side is the working tree. This matters more here than +elsewhere — reading lockfiles off disk in the committed view would report a dirty `Cargo.lock` as if +it were committed, so **both sides go through the same text-based resolver** rather than the disk +walk `audit::resolve` does today. + +## The refactor it needs + +`audit::resolve(root)` walks the filesystem, reads each lockfile, and dispatches on the basename to +parsers that are already pure text (`parse_toml_lock(&text, &rel, eco, &direct)` and friends). Only +the *input* layer is disk-bound: `resolve`'s walk, and the four `direct_*(dir)` helpers that read a +sibling manifest to decide which packages are direct. + +Introduce a source, and everything else is reused as-is: + +```rust +// where a resolution reads its lockfiles and manifests from +pub trait Source { + // every lockfile and manifest path this source holds, relative to the root + fn inputs(&self) -> Vec; + fn read(&self, rel: &str) -> Option; +} + +pub struct DiskSource<'a> { pub root: &'a Path } +// a committed tree, read with `git show :` +pub struct GitSource<'a> { pub root: &'a Path, pub sha: &'a str } + +pub fn resolve_with(src: &dyn Source) -> AuditReport; +pub fn resolve(root: &Path) -> AuditReport { // unchanged signature, unchanged behaviour + resolve_with(&DiskSource { root }) +} +``` + +`direct_cargo` / `direct_npm` / `direct_go` / `direct_python` take `(&dyn Source, dir)` instead of +`dir` and read through it. `GitSource::inputs()` is one `git ls-tree -r --name-only ` filtered +to `LOCK_NAMES ∪ MANIFEST_NAMES`; `read` is one `git show`. `locate()` gets the same treatment so +head-side manifest lines resolve against the head side, not the disk. + +This is the whole of the change to `audit.rs`. Nothing in the parsers, the OSV client, the CVSS +banding or the `Package`/`Finding` model moves. + +## Algorithm + +1. **Short-circuit.** `changes` already runs `git diff --name-status -z -M `. Filter that + result to basenames in `LOCK_NAMES ∪ MANIFEST_NAMES`. Empty → emit an empty `deps` section with + `changed: false` and stop. No blob is read, no package is parsed, no packet is sent. A branch that + touched no manifest costs nothing, which is most branches. +2. **Resolve both sides.** Base = `resolve_with(GitSource { sha: base_sha })`, reusing the merge-base + `resolve_base()` already computed. Head = `GitSource { sha: head_sha }`, or `DiskSource` under + `--worktree`. +3. **Follow renames.** `parse_name_status` reads `-M` rename records but flattens them — it emits + `renamed(new)` and `deleted(old)` as separate rows and drops the pairing, which is all its current + callers need. The deps pass needs the pairing, to map a moved lockfile's base path to its head + path so `deps/Cargo.lock` → `crates/x/Cargo.lock` is not a wholesale remove + add. Either return + the old path alongside the row, or leave that parser alone and run one narrow + `git diff --name-status -z -M -- ` for this pass. The second is smaller and + keeps a hot, well-tested parser untouched. +4. **Key and diff.** Key on `(ecosystem, lockfile, name)` → the set of versions under that key. Not + `(ecosystem, name)`: npm legitimately holds several versions of one package in one lockfile, and a + monorepo holds one package at different versions in different lockfiles. Diffing version *sets* + keeps both honest. +5. **Classify** each key (see below). +6. **Assess the delta only.** Query OSV for the versions that exist on exactly one side — added and + removed versions — not the full closure. A branch that bumps one package costs one bounded query, + and the standing set is `audit`'s job. `introduced` = advisories on head-only versions; + `resolved` = advisories on base-only versions that no head version carries. +7. **Locate** each change against the head manifests, reusing the `Location` attribution, so a + transitive bump points at the direct dependency whose line a person can actually edit. + +## Classification + +| kind | when | why it matters | +|---|---|---| +| `added` | key absent at base | new code in the tree | +| `removed` | key absent at head | attack surface gone; may resolve advisories | +| `upgraded` / `downgraded` | one version in, one out | the ordinary bump; direction per ecosystem semver, `unknown` when unparseable rather than guessed | +| `versions-changed` | many-in / many-out | npm's multi-version reality, reported with both lists rather than flattened into a fake bump | +| `promoted` / `demoted` | `direct` flipped, version same | a transitive package a manifest now names, or stopped naming | +| `now-ships` / `no-longer-ships` | `dev` flipped, version same | a dev-only package moved into the runtime closure. A dev advisory that never shipped now does | + +Version ordering is per-ecosystem and deliberately conservative: semver where the ecosystem +guarantees it, and `unknown` where it does not (Go pseudo-versions, PyPI epochs, npm prerelease +tags). An `unknown` direction is still reported as a change — it just does not claim which way. + +## Report shape + +`ChangesReport` gains one optional field. `ccc changes` always fills it; the option remains because +`insights` asks for the change set without it, and an absent key is honest where a null one would +not be. `schema` stays `ccc-changes/1` — additive optional fields do not bump it. + +```rust +#[serde(skip_serializing_if = "Option::is_none")] +pub deps: Option, +``` + +```rust +pub struct DepsReport { + pub schema: &'static str, // "ccc-deps/1", for /deps.json served standalone + pub base_sha: String, + pub head_sha: String, + // false when no manifest or lockfile was touched: everything below is empty + pub changed: bool, + // the base side had no lockfile at all - the whole closure reads as `added` + pub baseline: bool, + pub changes: Vec, + // advisories this change brings in, and ones it clears + pub introduced: Vec, + pub resolved: Vec, + // manifests edited without their lockfile being regenerated, and lockfiles + // that could not be read at one end + pub drift: Vec, + pub assessed: bool, + pub error: Option, + pub counts: DepsCounts, +} + +pub struct DepChange { + pub kind: DepChangeKind, + pub ecosystem: Ecosystem, + pub name: String, + pub lockfile: String, + pub from: Vec, // versions at base + pub to: Vec, // versions at head + pub direct: bool, // as of the head side + pub dev: bool, + // head manifest lines to draw this on; empty for a removed package nothing declares + pub locations: Vec, +} + +pub struct DepsCounts { + pub added: usize, + pub removed: usize, + pub upgraded: usize, + pub downgraded: usize, + pub other: usize, + pub introduced: usize, + pub resolved: usize, +} +``` + +Text format, appended to the `changes` text output, in the voice `print_audit_text` already uses: + +``` +dependencies: 3 changed (2 added, 1 upgraded) against origin/main + + tokio 1.40.0 cargo Cargo.lock direct + + tokio-util 0.7.11 cargo Cargo.lock transitive, via tokio + ~ serde 1.0.203 -> 1.0.210 cargo Cargo.lock direct + +introduced (1): + RUSTSEC-2024-0011 high tokio-util 0.7.11 - fixed in 0.7.12 + Cargo.toml:24 - via tokio +``` + +A branch that changed nothing prints one line: `dependencies: unchanged against origin/main`. + +## Serving it + +`ccc run` answers `GET /deps.json?base=` and advertises the `deps` MCP tool, with no flag to +turn either on. The delta short-circuits on a branch that touched no manifest, so an agent that asks +about a project which never moved a lockfile gets one line back and the server reads nothing. + +Cache it the way `Analysis` is cached — keyed on `(map generation ts, base)` — not the way +`audit_report` is, which is keyed on generation alone. The base ref is a parameter here, and two +agents asking about two refs must not evict each other into a re-query. + +The MCP tool description follows the `vulnerabilities` one in tone: say what it answers, say what it +does not. It answers *what this branch did to the dependency tree*; it does not answer *what we +depend on* (that is `vulnerabilities`) and it cannot see a change that was never committed unless the +server was started against a working tree. + +## Failure modes + +Every one of these is reported, never fatal, matching `audit`'s standing rule — a resolution that +half-worked must never read as a clean result. + +| situation | behaviour | +|---|---| +| base ref missing / shallow clone | the existing `resolve_base` error, unchanged: `changes` cannot run at all without a base | +| git not on PATH | `deps` present, `changed: false`, `error` set | +| no lockfile at base (new project, first lock) | `baseline: true`; every package reads as `added` and the counts say so, rather than a triumphant "1,204 dependencies added" with no context | +| lockfile deleted at head | its packages read as `removed`, and `drift` notes the manifest that now pins nothing | +| manifest edited, lockfile untouched | `drift` entry: *"Cargo.toml changed but Cargo.lock did not — the declared range moved and the pinned version did not"* | +| OSV unreachable | `assessed: false`, `error` set, `changes` still complete. `--fail-introduced` exits non-zero, because "we could not check" is not "it is fine" | +| unpinned requirements.txt | already an `Unresolved` in `audit`; carried through to `drift` unchanged | + +## Tests + +Mirroring the existing split in `changes.rs` and `audit.rs`: + +- unit, on the classifier alone: synthetic `Vec` pairs covering every row of the + classification table, plus the npm multi-version case and the unparseable-version case +- unit, on `GitSource`: a lockfile read at a sha resolves to the same packages `DiskSource` gets from + the same content +- end-to-end, in the style of `changes_end_to_end_git`: init a repo, commit a `Cargo.toml` + + `Cargo.lock`, branch, bump one dependency, and assert one `upgraded` change and no others — then + the same with `--worktree` and the bump uncommitted +- the short-circuit: a branch that touches only `.rs` files produces `changed: false` and reads no + blob (assert via a source that panics on `read`) +- `baseline`, drift, and the OSV-unreachable path, each asserting the report is still well-formed + +## Editor + +Deferred, but the payload above is already the right shape for it: `DepChange.locations` carries +manifest lines exactly as `Finding.locations` does, so `VulnerabilityMarks` in the extension can draw +"upgraded on this branch" or "advisory introduced here" on a manifest line with no new plumbing. It +needs a second payload alongside `vulnerabilities`, and nothing else — the spawned server now +answers `/deps.json` without being asked. Nothing in this spec should be designed around that; it is +listed so v1 does not close the door on it. diff --git a/docs/EXTERNALS.md b/docs/EXTERNALS.md index e7a32dc..6873f7b 100644 --- a/docs/EXTERNALS.md +++ b/docs/EXTERNALS.md @@ -9,7 +9,9 @@ that call: there is no symbol to resolve, and the code on the other side is not - **`externals` in `.ccc/map.json`** names the peer repositories. - **`ccc:serves` / `ccc:calls` comments** name the key both ends agree on. -Matching keys become real edges of the service graph, with a file and line at each end. +Matching keys become real edges of the service graph, with a file and line at each end. For gRPC +the key needs no comment at all: the `.proto` schema supplies it +([gRPC without comments](#grpc-without-comments)). ## The hints @@ -62,7 +64,7 @@ function happens to appear later. "gateway": ["gateway/**"], "shared": ["shared/**"] }, - "deps": { "gateway": ["billing"] }, + "relatives": { "gateway": ["billing"] }, "externals": { "billing": { "repo": "acme/billing", @@ -71,20 +73,20 @@ function happens to appear later. }, "ledger": { "repo": "acme/ledger", - "surface": "https://artifacts.internal/ledger/ccc-surface.json", + "surface": "https://artifacts.internal/ledger/surface.json", "auth": "env:CCC_TOKEN" } } } ``` -An external is a service like any other: `deps` may name it, and edges end at it. A name cannot be -both a service and an external — it is either code in this repo or code in another one. +An external is a service like any other: `relatives` may name it, and edges end at it. +A name cannot be both a service and an external — it is either code in this repo or code in another one. | field | meaning | |---|---| | `path` | A directory to parse: a sibling checkout, or another corner of a monorepo. Relative paths resolve against the repo root. | -| `surface` | A file, a directory containing `ccc-surface.json`, or an `http(s)` URL, holding a surface published with `ccc export`. | +| `surface` | A file, a directory containing `surface.json` (or the older `ccc-surface.json`, still read), or an `http(s)` URL, holding a surface published with `ccc init`. | | `auth` | `env:VARIABLE` — the variable holding a bearer token for a private URL. Only this form is accepted; a literal token in a file that belongs in git is a mistake, not a feature. | | `repo` | `owner/repo`, for display. | | `lang` | The peer's language, for display when no surface is reachable. | @@ -100,8 +102,8 @@ analysis runs exactly as before. ## Publishing a surface ```sh -ccc export --name billing --repo acme/billing # -> .ccc/ccc-surface.json -ccc export --name billing -o - # stdout, for a CI artifact +ccc init --name billing --repo acme/billing # -> .ccc/surface.json +ccc init --name billing -o - # stdout, for a CI artifact ``` A surface is only what a repository publishes and consumes — no bodies, no call graph, no private @@ -130,9 +132,9 @@ language, and no access to its source — which is what makes a private peer wor place. Publish it from CI on merge: ```yaml -- run: ccc export --name billing --repo ${{ github.repository }} +- run: ccc init --name billing --repo ${{ github.repository }} - uses: actions/upload-artifact@v4 - with: { name: ccc-surface, path: .ccc/ccc-surface.json } + with: { name: ccc-surface, path: .ccc/surface.json } ``` `consumes` is what lets a repository learn that something out there calls **in** to it — the one @@ -166,6 +168,57 @@ The VS Code extension shows a crossing as a CodeLens above the call — `↑ cal acme/billing` — and opens the handler when that peer is checked out locally, or says where it lives when it is not. +## gRPC without comments + +An rpc is the rare call that crosses languages on purpose, and the schema every side was generated from already names it. + +Each rpc is keyed by its wire name, `acme.billing.v1.Billing/CreateInvoice`, +and code is tied to it by evidence — never by a name alone, because the name changes with the +generator: `CreateInvoice` in Go, `createInvoice` in TypeScript, `create_invoice` in Rust. + +| side | tied when the name matches (case and `_` ignored) and | evidence | +|---|---|---| +| handler | a parameter is the rpc's request message | `request-type` | +| handler | its owning type is named after the service, and the file imports the generated code | `service-owner` | +| caller | the receiver is the service's generated stub (`BillingClient`, `BillingStub`, …) | `stub-type` | +| caller | the file imports the generated code (`billing_pb2_grpc`, `…/billing/v1`, `billing_connect`) | `generated-import` | + +Where the evidence fits two rpcs equally, nothing is linked. Generated files themselves +(`*.pb.go`, `*_pb2_grpc.py`, `*_connect.ts`, `*Grpc.cs`, …) are never read as handlers or callers. + +What the links feed: + +- **`/references`** — looking up an rpc by any spelling (`create_invoice`, `Billing.CreateInvoice`, + the wire name) adds an `rpcs` block: the schema, its handlers and its callers, in every language. +- **Coverage** — a Python test through the stub covers the Go handler (evidence `rpc`). +- **The call graph** — client → rpc in the schema → handler. +- **`ccc changes`** — `rpc` edges from each caller's service to each handler's, and from both to the + service holding the schema, so a changed `.proto` reaches everything generated from it. +- **Surfaces** — `ccc init` publishes handlers under `provides` and calls under `consumes`, marked + `"via": "rpc"`, so a peer repository links to them exactly as it would to a `ccc:` comment. + +### Schemas kept elsewhere + +Schemas inside the project are found by the scan. When they live in a shared protos repository, a +buf module or a vendored directory, name them under `contracts` in `.ccc/map.json`: + +```json +{ + "contracts": ["../protos/acme/**/*.proto", "third_party/proto"] +} +``` + +Each entry is a file, a directory (every `.proto` under it) or a glob, relative to the repository +root and free to leave it. Ignore files are not consulted — vendored schemas are routinely ignored, +and naming them here is the opt-in. A peer checkout named by `path` is linked through its own +`contracts` and this repository's, since the schema is often in neither. + +A surface published before rpc links existed lacks `"rpc_endpoints": true`. A `path` peer whose +published surface is that old is parsed instead; a `surface`-only peer needs to re-run `ccc init`. + +An rpc call that nothing in view serves is not reported as an unanswered key: that is usually a +third-party API rather than a typo. + ## What this does and does not establish `ccc:serves` and `ccc:calls` are **statements by an author**, not inferences. That is exactly why diff --git a/docs/LANGUAGES.md b/docs/LANGUAGES.md index 7a0f143..cd135e0 100644 --- a/docs/LANGUAGES.md +++ b/docs/LANGUAGES.md @@ -14,6 +14,7 @@ Every language ccc can analyse. | Go | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | JavaScript | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | n/a | ✅ | ✅ | ✅ | | Odin | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | +| Protobuf | ✅ | ✅ | ✅ | ✅ | n/a | ✅ | ✅ | ✅ | n/a | ✅ | ✅ | | Python | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | n/a | ✅ | ✅ | ✅ | | Rust | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | TypeScript | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | @@ -35,6 +36,7 @@ n/a - the language has no such concept | Go | `.go` | | JavaScript | `.js`, `.jsx`, `.mjs`, `.cjs` | | Odin | `.odin` | +| Protobuf | `.proto` | | Python | `.py`, `.pyi` | | Rust | `.rs` | | TypeScript | `.ts`, `.mts`, `.cts` | @@ -48,6 +50,13 @@ Please create a PR to change this if you want to solve that specific case. ## Approximations +Protobuf is a schema, so it is mapped onto the same shapes: a +`message` is a struct, a `service` an interface, each `rpc` a method of its +service taking its request message and returning its response (`stream` is +dropped), and enum values are constants. It has no calls or bodies, so the +calls and metrics columns do not apply. Code in other languages is linked to +each rpc through its generated stubs - see [gRPC without comments](EXTERNALS.md#grpc-without-comments). + One deliberate approximation is worth naming: Zig's `errdefer` counts as a guard even though it only runs on the error path. Reading a correctly written cleanup as a leak is the worse of the two errors, and the heuristic is diff --git a/docs/MCP.md b/docs/MCP.md index 79eead7..11ec621 100644 --- a/docs/MCP.md +++ b/docs/MCP.md @@ -1,12 +1,12 @@ # MCP -`ccc serve` parses the project into an **in-memory copy of the map** and +`ccc run` parses the project into an **in-memory copy of the map** and serves it over local HTTP, so AI agents query the code map directly instead of reading `.ccc` files from disk. A file watcher rescans automatically when source changes (default: every 2s; `--no-watch` to disable): ```sh -ccc serve # http://127.0.0.1:6767 (MCP endpoint at /mcp -- insights at /insights) +ccc run # http://127.0.0.1:6767 (MCP endpoint at /mcp -- insights at /insights) ``` ```sh diff --git a/docs/PIPELINES.md b/docs/PIPELINES.md index 5c6c80c..e90fab7 100644 --- a/docs/PIPELINES.md +++ b/docs/PIPELINES.md @@ -7,7 +7,7 @@ maps the diff down to function granularity, groups files into named *services* Service A calls Service B and B changes, both land in the test set: ```sh -ccc changes --init # scaffold .ccc/map.json from your top-level dirs +ccc init # scaffold .ccc/map.json from your top-level dirs, and the surface ccc changes # one line of JSON: services_to_test, edges, untested, ... ccc changes --format text # human-readable summary ccc changes --fail-untested # gate: exit 1 when changed functions lack test references diff --git a/docs/STREAM.md b/docs/STREAM.md index 5b1812e..57048d0 100644 --- a/docs/STREAM.md +++ b/docs/STREAM.md @@ -8,7 +8,7 @@ > For exact Claude token counts, use Anthropic's `count_tokens` endpoint. > `tokens.json` carries this caveat inline (`approximate: true` + a `note`). -`ccc tokenize` (or `ccc scan --tokens`) encodes the whole `.ccc` corpus with a +`ccc tokenize` (or `ccc scan --dir --tokens`) encodes the whole map with a pretrained tiktoken vocabulary (`o200k_base` by default, `--encoding cl100k_base` also supported) and writes: @@ -29,5 +29,5 @@ let ids: &[u32] = cache.file("src-main.rs.md").unwrap(); // raw tokens, ready let text = cache.decode(ids)?; // optional: back to markdown ``` -Token artifacts are derived, so a plain `ccc scan` clears them; re-run with +Token artifacts are derived, so a `ccc scan --dir` clears them; re-run with `--tokens` (or `ccc tokenize`) to refresh. diff --git a/extensions/vscode/package.json b/extensions/vscode/package.json index 32744fc..0c214d1 100644 --- a/extensions/vscode/package.json +++ b/extensions/vscode/package.json @@ -2,7 +2,7 @@ "name": "ccc-codecache", "displayName": "CodeCaChe (ccc)", "description": "Test-coverage, untested-changes and cross-service hints from the ccc static analyser, inline in your editor.", - "version": "1.4.2", + "version": "1.4.4", "publisher": "colwill", "license": "MIT", "icon": "media/icon.png", @@ -520,11 +520,11 @@ "ccc": [ { "id": "ccc.testTriggers", - "name": "[CCC] Test Impact" + "name": "Tests Triggered" }, { "id": "ccc.complexity", - "name": "[CCC] Complexity" + "name": "Complexity" } ] } diff --git a/extensions/vscode/src/hover.ts b/extensions/vscode/src/hover.ts index 463651c..f050c42 100644 --- a/extensions/vscode/src/hover.ts +++ b/extensions/vscode/src/hover.ts @@ -450,7 +450,7 @@ function complexityHover(fn: FileFunc): vscode.MarkdownString { const score = fn.complexity_score ?? 0; const md = new vscode.MarkdownString(); md.supportThemeIcons = true; - md.appendMarkdown(`**[ccc] Complexity ${score}/10** - _${SCORE_DESCRIPTION[score] ?? ''}_`); + md.appendMarkdown(`**Complexity ${score}/10** - _${SCORE_DESCRIPTION[score] ?? ''}_`); const parts: string[] = []; if (typeof fn.complexity === 'number') parts.push(`${fn.complexity} independent path(s)`); if (typeof fn.branches === 'number' && fn.branches > 0) { diff --git a/extensions/vscode/src/types.ts b/extensions/vscode/src/types.ts index 90b6c88..1bb2cf2 100644 --- a/extensions/vscode/src/types.ts +++ b/extensions/vscode/src/types.ts @@ -186,7 +186,7 @@ export interface ServicesEdge { export interface ServicesSection { // grouping provenance; the only signal of how services were derived source: string; - declared_deps: Record; + declared_relatives: Record; services: { name: string; globs: string[]; files: number; funcs: number; paths: string[] }[]; edges: ServicesEdge[]; unassigned_files: string[]; diff --git a/install.sh b/install.sh index aafe51e..59e98b8 100755 --- a/install.sh +++ b/install.sh @@ -1,9 +1,9 @@ #!/usr/bin/env bash # SPDX-License-Identifier: MIT -# This script is part of ContextCodeCache (ccc) and is distributed under the +# This script is part of CodeCaChe (ccc) and is distributed under the # MIT License (see https://github.com/colwill/ccc/LICENSE). # -# ContextCodeCache (ccc) installer +# CodeCaChe (ccc) installer # Auto-detects OS and architecture, downloads the matching release asset from # https://github.com/colwill/ccc, then lets the binary install itself onto # your PATH with `ccc install` (defaults to ~/.local/bin; no sudo). @@ -88,4 +88,4 @@ else fi say "installed $("$BIN" --version)" -say "try: ccc serve or ccc scan ." +say "try: ccc run or ccc scan ." diff --git a/src/changes.rs b/src/changes.rs index ed1534a..29d3a2e 100644 --- a/src/changes.rs +++ b/src/changes.rs @@ -4,6 +4,7 @@ //! `--service` flags), diffs the branch against a base ref. // ccc:skip +use crate::contracts::ContractIndex; use crate::coverage; use crate::extract::BDD_REGISTRARS; use crate::model::{Boundary, FileCache}; @@ -49,12 +50,16 @@ pub struct ChangesOptions { pub struct ChangesConfig { #[serde(default)] pub services: BTreeMap>, - #[serde(default)] - pub deps: BTreeMap>, + // Declared relationships + #[serde(default, alias = "deps")] + pub relatives: BTreeMap>, // peer repositories: another checkout, or a published surface. Their names // are service names too, so `deps` may point at them. #[serde(default)] pub externals: BTreeMap, + // `.proto` schemas kept outside the project + #[serde(default)] + pub contracts: Vec, } impl ChangesConfig { @@ -64,6 +69,7 @@ impl ChangesConfig { }; let raw = fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?; + warn_legacy_deps_key(&path, &raw); serde_json::from_str(&raw).with_context(|| format!("parsing {}", path.display())) } @@ -77,6 +83,25 @@ impl ChangesConfig { } +// `relatives` was called `deps` back when it could only name a service +fn warn_legacy_deps_key(path: &Path, raw: &str) { + static WARNED: std::sync::Once = std::sync::Once::new(); + let uses_deps = serde_json::from_str::(raw) + .ok() + .is_some_and(|v| v.get("deps").is_some()); + if !uses_deps { + return; + } + WARNED.call_once(|| { + eprintln!( + "warning: {} uses `deps`, which is now `relatives` - it may name a peer \ + repository from `externals`, not only a service in this repo. `deps` is \ + still read and will be removed in 2.0.0.", + path.display() + ); + }); +} + #[derive(Debug, Serialize, Clone)] pub struct ChangedFile { pub path: String, @@ -226,7 +251,7 @@ pub fn changes(root: &Path, root_label: &str, opts: &ChangesOptions) -> Result = config.services.keys().cloned().collect(); known.extend(config.externals.keys().cloned()); bail!( - "map.json deps mention unknown service '{t}' \ + "map.json relatives mention unknown service '{t}' \ (known: {})", known.join(", ") ); @@ -272,7 +297,10 @@ pub fn changes_with_caches( } let matchers = build_matchers(&config.services)?; let service_names: Vec = config.services.keys().cloned().collect(); - let externals = crate::externals::resolve_all(root, &config.externals); + // the rpcs every schema declares, and the code in any language tied to them + let schemas = crate::contracts::load_schemas(root, &config.contracts); + let contracts = ContractIndex::build(caches, &schemas); + let externals = crate::externals::resolve_all(root, &config.externals, &contracts.schemas); let (base_label, base_sha) = resolve_base(root, opts.base.as_deref())?; let head_sha = git(root, &["rev-parse", "HEAD"])?.trim().to_string(); @@ -358,14 +386,22 @@ pub fn changes_with_caches( // `insights`, so the two reports cannot disagree about what is covered. let project_ids: BTreeSet = manifest_identities(root).into_iter().map(|(id, _)| id).collect(); - let cov = coverage::build(caches, &project_ids); + let cov = coverage::build(caches, &project_ids, &contracts); // cross-service edges + per-symbol caller map - let (mut edges, symbol_callers, unresolved_calls) = detect_edges(&idx, &config.deps); + let (mut edges, mut symbol_callers, mut unresolved_calls) = detect_edges(&idx, &config.relatives); + merge_rpc_edges( + &mut edges, + &mut symbol_callers, + &mut unresolved_calls, + caches, + &matchers, + &contracts, + ); // boundary crossings: the calls that leave the process, which the call - // graph cannot see and the author had to name - let crossings = detect_crossings(caches, &matchers, &externals); + // graph cannot see - named by an author, or derived from a schema + let crossings = detect_crossings(caches, &matchers, &externals, &contracts); merge_crossings(&mut edges, &crossings); // changed files -> services @@ -562,6 +598,8 @@ fn crossing_json(c: &crate::externals::Crossing) -> Value { "line": c.line, "function": c.function, "external": c.external, + // `rpc` when a schema tied it, `annotation` when a `ccc:` comment did + "via": if c.rpc { crate::contracts::VIA } else { "annotation" }, // absent when nothing answers this key "remote": c.remote.as_ref().map(|r| serde_json::json!({ "function": r.function, @@ -604,7 +642,7 @@ pub fn init_config(root: &Path) -> Result { if services.is_empty() { bail!("no supported source files found under {}", root.display()); } - let out = serde_json::json!({ "services": services, "deps": {} }); + let out = serde_json::json!({ "services": services, "relatives": {} }); fs::create_dir_all(path.parent().unwrap())?; fs::write(&path, format!("{}\n", serde_json::to_string_pretty(&out)?))?; Ok(path) @@ -1178,38 +1216,46 @@ pub(crate) fn detect_crossings( caches: &[FileCache], matchers: &[(String, GlobSet)], externals: &[crate::externals::ExternalService], + contracts: &ContractIndex, ) -> Vec { use crate::externals::{norm_key, Crossing, Endpoint}; - // every handler this repo publishes, by key - let mut local_handlers: BTreeMap> = BTreeMap::new(); + // every boundary endpoint here + let mut local: Vec<(Boundary, Endpoint)> = Vec::new(); for cache in caches { let file = path_str(&cache.rel_path); - let services = assign(matchers, &file); for ann in &cache.annotations { - if ann.boundary != Boundary::Serves { - continue; - } let endpoint = Endpoint { key: ann.key.clone(), transport: ann.transport.clone(), function: ann.function.clone(), file: file.clone(), line: ann.line, - service: services.first().cloned(), + service: None, + via: None, }; - for service in services.iter() { - local_handlers - .entry(norm_key(&ann.key)) - .or_default() - .push((service.clone(), endpoint.clone())); - } - if services.is_empty() { - local_handlers - .entry(norm_key(&ann.key)) - .or_default() - .push((String::new(), endpoint.clone())); - } + local.push((ann.boundary, endpoint)); + } + } + let (rpc_provides, rpc_consumes) = contracts.endpoints(caches); + local.extend(rpc_provides.into_iter().map(|e| (Boundary::Serves, e))); + local.extend(rpc_consumes.into_iter().map(|e| (Boundary::Calls, e))); + + // every handler this repo publishes, by key + let mut local_handlers: BTreeMap> = BTreeMap::new(); + for (boundary, endpoint) in &local { + if *boundary != Boundary::Serves { + continue; + } + let services = assign(matchers, &endpoint.file); + let mut endpoint = endpoint.clone(); + endpoint.service = services.first().cloned(); + let owners = if services.is_empty() { vec![String::new()] } else { services }; + for service in owners { + local_handlers + .entry(norm_key(&endpoint.key)) + .or_default() + .push((service, endpoint.clone())); } } @@ -1217,37 +1263,39 @@ pub(crate) fn detect_crossings( // outbound: a call here, matched against peers first, then against this // repo's own handlers - for cache in caches { - let file = path_str(&cache.rel_path); - let services = assign(matchers, &file); - for ann in &cache.annotations { - if ann.boundary != Boundary::Calls { - continue; - } - let key = norm_key(&ann.key); - let from = services.first().cloned().unwrap_or_default(); + for (boundary, call) in &local { + if *boundary != Boundary::Calls { + continue; + } + let rpc = call.via.is_some(); + let key = norm_key(&call.key); + let from = assign(matchers, &call.file).first().cloned().unwrap_or_default(); - let mut matched = false; - for external in externals { - let Some(surface) = &external.surface else { - continue; - }; - for endpoint in surface.provides.iter().filter(|e| norm_key(&e.key) == key) { - matched = true; - out.push(Crossing { - key: ann.key.clone(), - transport: pick_transport(&ann.transport, &endpoint.transport), - from: from.clone(), - to: external.name.clone(), - file: file.clone(), - line: ann.line, - function: ann.function.clone(), - remote: Some(endpoint.clone()), - external: true, - }); - } + let mut matched = false; + for external in externals { + let Some(surface) = &external.surface else { + continue; + }; + for endpoint in surface.provides.iter().filter(|e| norm_key(&e.key) == key) { + matched = true; + out.push(Crossing { + key: call.key.clone(), + transport: pick_transport(&call.transport, &endpoint.transport), + from: from.clone(), + to: external.name.clone(), + file: call.file.clone(), + line: call.line, + function: call.function.clone(), + remote: Some(endpoint.clone()), + external: true, + rpc: rpc || endpoint.via.is_some(), + }); } + } + // this repo's own rpc handlers are reached through the call graph + // already, as `rpc` service edges + if !rpc { for (service, endpoint) in local_handlers.get(&key).into_iter().flatten() { // a handler in the same service is an internal detail, not a // boundary crossing @@ -1256,33 +1304,37 @@ pub(crate) fn detect_crossings( } matched = true; out.push(Crossing { - key: ann.key.clone(), - transport: pick_transport(&ann.transport, &endpoint.transport), + key: call.key.clone(), + transport: pick_transport(&call.transport, &endpoint.transport), from: from.clone(), to: service.clone(), - file: file.clone(), - line: ann.line, - function: ann.function.clone(), + file: call.file.clone(), + line: call.line, + function: call.function.clone(), remote: Some(endpoint.clone()), external: false, + rpc: false, }); } + } - // A call naming a key nobody answers is worth reporting: it is - // either a typo at one end, or a peer that was never configured. - if !matched { - out.push(Crossing { - key: ann.key.clone(), - transport: ann.transport.clone(), - from, - to: String::new(), - file: file.clone(), - line: ann.line, - function: ann.function.clone(), - remote: None, - external: false, - }); - } + // A call naming a key nobody answers is worth reporting: it is either + // a typo at one end, or a peer that was never configured. An rpc with + // no handler anywhere in view is more often a third-party API, which + // is not a mistake to report. + if !matched && !rpc { + out.push(Crossing { + key: call.key.clone(), + transport: call.transport.clone(), + from, + to: String::new(), + file: call.file.clone(), + line: call.line, + function: call.function.clone(), + remote: None, + external: false, + rpc: false, + }); } } @@ -1305,6 +1357,7 @@ pub(crate) fn detect_crossings( function: endpoint.function.clone(), remote: Some(consumed.clone()), external: true, + rpc: endpoint.via.is_some() || consumed.via.is_some(), }); } } @@ -1316,6 +1369,88 @@ pub(crate) fn detect_crossings( out } +// Service edges an rpc makes: the caller's service depends on each handler's, +// and both depend on the service holding the schema - so a changed `.proto` +// reaches everything generated from it, in every language. +fn merge_rpc_edges( + edges: &mut Vec, + symbol_callers: &mut BTreeMap>, + unresolved: &mut Vec, + caches: &[FileCache], + matchers: &[(String, GlobSet)], + contracts: &ContractIndex, +) { + let services_of = |fi: usize| assign(matchers, &path_str(&caches[fi].rel_path)); + let symbol = |key: &str, file: &str, line: usize, kind: &str| EdgeSymbol { + symbol: key.to_string(), + file: file.to_string(), + line, + via: crate::contracts::VIA.to_string(), + kind: kind.to_string(), + }; + let mut linked: BTreeSet<(String, usize)> = BTreeSet::new(); + for link in &contracts.callers { + let c = &contracts.contracts[link.contract]; + let file = path_str(&caches[link.file].rel_path); + let line = caches[link.file].calls[link.call].line; + linked.insert((file.clone(), line)); + for from in services_of(link.file) { + for h in contracts.handlers_of(link.contract) { + for to in services_of(h.def.0).into_iter().filter(|to| *to != from) { + add_edge_symbol(edges, &from, &to, symbol(&c.key, &file, line, "call")); + symbol_callers + .entry(caches[h.def.0].funcs[h.def.1].name.clone()) + .or_default() + .insert(from.clone()); + } + } + if let Some((schema, _)) = c.def { + for to in services_of(schema).into_iter().filter(|to| *to != from) { + add_edge_symbol(edges, &from, &to, symbol(&c.key, &file, line, "type")); + } + } + } + } + // a handler depends on its schema just as a caller does + for h in &contracts.handlers { + let c = &contracts.contracts[h.contract]; + let Some((schema, _)) = c.def else { continue }; + let file = path_str(&caches[h.def.0].rel_path); + let line = caches[h.def.0].funcs[h.def.1].line; + for from in services_of(h.def.0) { + for to in services_of(schema).into_iter().filter(|to| *to != from) { + add_edge_symbol(edges, &from, &to, symbol(&c.key, &file, line, "type")); + } + } + } + // a call the schema accounts for is not an unresolved one + unresolved.retain(|u| !linked.contains(&(u.file.clone(), u.line))); + edges.sort_by(|a, b| (&a.from, &a.to).cmp(&(&b.from, &b.to))); +} + +// one piece of evidence on the edge `from -> to`, creating the edge if needed +fn add_edge_symbol(edges: &mut Vec, from: &str, to: &str, symbol: EdgeSymbol) { + match edges.iter_mut().find(|e| e.from == from && e.to == to) { + Some(edge) => { + let seen = edge + .symbols + .iter() + .any(|s| s.symbol == symbol.symbol && s.line == symbol.line && s.file == symbol.file); + if !seen { + edge.symbols.push(symbol); + } + edge.detected = true; + } + None => edges.push(ServiceEdge { + from: from.to_string(), + to: to.to_string(), + declared: false, + detected: true, + symbols: vec![symbol], + }), + } +} + // One side may name a transport and the other leave it out; prefer whichever // actually said something. fn pick_transport(a: &str, b: &str) -> String { @@ -1333,35 +1468,19 @@ fn merge_crossings(edges: &mut Vec, crossings: &[crate::externals:: if crossing.from.is_empty() || crossing.to.is_empty() { continue; } + let via = if crossing.rpc { + crate::contracts::VIA + } else { + Via::Annotation.label() + }; let symbol = EdgeSymbol { symbol: crossing.key.clone(), file: crossing.file.clone(), line: crossing.line, - via: Via::Annotation.label().to_string(), + via: via.to_string(), kind: crossing.transport.clone(), }; - match edges - .iter_mut() - .find(|e| e.from == crossing.from && e.to == crossing.to) - { - Some(edge) => { - if !edge - .symbols - .iter() - .any(|s| s.symbol == symbol.symbol && s.line == symbol.line && s.file == symbol.file) - { - edge.symbols.push(symbol); - } - edge.detected = true; - } - None => edges.push(ServiceEdge { - from: crossing.from.clone(), - to: crossing.to.clone(), - declared: false, - detected: true, - symbols: vec![symbol], - }), - } + add_edge_symbol(edges, &crossing.from, &crossing.to, symbol); } edges.sort_by(|a, b| (&a.from, &a.to).cmp(&(&b.from, &b.to))); } @@ -1370,7 +1489,7 @@ fn merge_crossings(edges: &mut Vec, crossings: &[crate::externals:: fn via_rank(via: &str) -> usize { match via { "annotation" => 0, - "receiver-type" => 1, + "receiver-type" | "rpc" => 1, "qualifier" => 2, "project" => 3, "import" => 4, @@ -1951,6 +2070,131 @@ diff --git a/gone.rs b/gone.rs build_matchers(&services).expect("globs") } + const BILLING_PROTO: &str = "syntax = \"proto3\";\n\ + package acme.billing.v1;\n\ + message CreateInvoiceRequest { string customer = 1; }\n\ + message Invoice { string id = 1; }\n\ + service Billing { rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice); }\n"; + + // a typescript client and a go server, with the schema in a third service + fn rpc_repo() -> (Vec, Vec<(String, GlobSet)>) { + use crate::languages::Language; + let caches = vec![ + fixture( + Language::TypeScript, + "gateway/charge.ts", + "import { BillingClient } from \"../gen/acme/billing/v1/billing_grpc_pb\";\n\ + export function charge() {\n\ + \x20 const client = new BillingClient(\"addr\");\n\ + \x20 client.createInvoice({});\n\ + }\n", + ), + fixture( + Language::Go, + "billing/server.go", + "package billing\n\ + import billingv1 \"github.com/acme/gen/acme/billing/v1\"\n\ + type server struct{}\n\ + func (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\ + \treturn nil, nil\n\ + }\n", + ), + fixture(Language::Proto, "shared/billing.proto", BILLING_PROTO), + ]; + let mut services = BTreeMap::new(); + for s in ["gateway", "billing", "shared"] { + services.insert(s.to_string(), vec![format!("{s}/**")]); + } + (caches, build_matchers(&services).expect("globs")) + } + + // no comment at either end: the schema is what joins them + #[test] + fn an_rpc_joins_a_client_to_its_handler_across_languages() { + let (caches, matchers) = rpc_repo(); + let contracts = ContractIndex::build(&caches, &[]); + let mut edges = Vec::new(); + let mut callers = BTreeMap::new(); + let mut unresolved = vec![UnresolvedCall { + symbol: "createInvoice".into(), + file: "gateway/charge.ts".into(), + line: 4, + from: "gateway".into(), + reason: "no-evidence".into(), + candidates: Vec::new(), + }]; + merge_rpc_edges(&mut edges, &mut callers, &mut unresolved, &caches, &matchers, &contracts); + + let found: Vec<(&str, &str, &str)> = edges + .iter() + .flat_map(|e| e.symbols.iter().map(move |s| (e.from.as_str(), e.to.as_str(), s.kind.as_str()))) + .collect(); + assert_eq!( + found, + vec![ + // the handler depends on the schema it implements + ("billing", "shared", "type"), + ("gateway", "billing", "call"), + ("gateway", "shared", "type"), + ] + ); + assert!(edges.iter().flat_map(|e| &e.symbols).all(|s| s.via == "rpc" + && s.symbol == "acme.billing.v1.Billing/CreateInvoice")); + // the go handler's report names the typescript service that calls it + assert_eq!(callers["CreateInvoice"], BTreeSet::from(["gateway".to_string()])); + // and the call the name-based resolver gave up on is accounted for + assert!(unresolved.is_empty(), "{unresolved:?}"); + + // a changed schema reaches both ends + let impact = impact_closure(&BTreeSet::from(["shared".to_string()]), &edges); + let hit: BTreeSet<&str> = impact.iter().map(|i| i.service.as_str()).collect(); + assert_eq!(hit, BTreeSet::from(["shared", "billing", "gateway"])); + } + + // Two repositories: the client here, the server in a peer that published + // a surface. The rpc key is the same string at both ends because both were + // derived from the same schema. + #[test] + fn an_rpc_reaches_a_handler_in_a_peer_repository() { + let (mut caches, matchers) = rpc_repo(); + caches.retain(|c| !c.rel_path.starts_with("billing")); + let contracts = ContractIndex::build(&caches, &[]); + let peer = crate::externals::ExternalService { + name: "billing".to_string(), + config: Default::default(), + source: "surface".to_string(), + surface: Some(crate::externals::Surface { + schema: crate::externals::SURFACE_SCHEMA.to_string(), + name: "billing".to_string(), + generated: "t".to_string(), + repo: None, + languages: vec!["go".to_string()], + provides: vec![crate::externals::Endpoint { + key: "acme.billing.v1.Billing/CreateInvoice".to_string(), + transport: "grpc".to_string(), + function: "CreateInvoice".to_string(), + file: "svc/server.go".to_string(), + line: 4, + service: None, + via: Some("rpc".to_string()), + }], + consumes: Vec::new(), + rpc_endpoints: true, + }), + error: None, + }; + let crossings = detect_crossings(&caches, &matchers, &[peer], &contracts); + assert_eq!(crossings.len(), 1, "{crossings:?}"); + let c = &crossings[0]; + assert_eq!((c.from.as_str(), c.to.as_str()), ("gateway", "billing")); + assert!(c.rpc && c.external); + assert_eq!(c.remote.as_ref().unwrap().file, "svc/server.go"); + + // with no peer serving it, an rpc is not reported as a dangling key + let crossings = detect_crossings(&caches, &matchers, &[], &contracts); + assert!(crossings.is_empty(), "{crossings:?}"); + } + // The monorepo case: two directories in one repo, joined by a key rather // than by a call the parser could ever follow. #[test] @@ -1967,7 +2211,7 @@ diff --git a/gone.rs b/gone.rs "// ccc:serves queue audit.events\npub fn record(e: &str) -> usize { e.len() }\n", ), ]; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[], &ContractIndex::default()); assert_eq!(crossings.len(), 1, "{crossings:?}"); let c = &crossings[0]; assert_eq!((c.from.as_str(), c.to.as_str()), ("gateway", "shared")); @@ -2005,12 +2249,14 @@ diff --git a/gone.rs b/gone.rs file: "svc/charge.go".to_string(), line: 42, service: None, + via: None, }], consumes: Vec::new(), + rpc_endpoints: false, }), error: None, }; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[peer]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[peer], &ContractIndex::default()); assert_eq!(crossings.len(), 1, "{crossings:?}"); let c = &crossings[0]; assert_eq!(c.to, "billing"); @@ -2034,7 +2280,7 @@ diff --git a/gone.rs b/gone.rs "package b\n\n// ccc:serves grpc billing.v1.charge\nfunc Charge() {}\n", ), ]; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[], &ContractIndex::default()); assert_eq!(crossings.len(), 1, "{crossings:?}"); assert!(crossings[0].remote.is_some()); } @@ -2048,7 +2294,7 @@ diff --git a/gone.rs b/gone.rs "gateway/main.rs", "pub fn a() {\n // ccc:calls grpc nobody.Answers\n}\n", )]; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[], &ContractIndex::default()); assert_eq!(crossings.len(), 1); assert!(crossings[0].remote.is_none()); assert_eq!(crossings[0].to, ""); @@ -2062,7 +2308,7 @@ diff --git a/gone.rs b/gone.rs "gateway/main.rs", "// ccc:serves queue x.y\npub fn h() {}\n\npub fn a() {\n // ccc:calls queue x.y\n}\n", )]; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[], &ContractIndex::default()); assert!(crossings.iter().all(|c| c.remote.is_none()), "{crossings:?}"); } @@ -2093,11 +2339,13 @@ diff --git a/gone.rs b/gone.rs file: "svc/charge.go".to_string(), line: 7, service: None, + via: None, }], + rpc_endpoints: false, }), error: None, }; - let crossings = detect_crossings(&caches, &two_service_matchers(), &[peer]); + let crossings = detect_crossings(&caches, &two_service_matchers(), &[peer], &ContractIndex::default()); assert_eq!(crossings.len(), 1, "{crossings:?}"); let c = &crossings[0]; assert_eq!((c.from.as_str(), c.to.as_str()), ("billing", "shared")); @@ -2112,7 +2360,7 @@ diff --git a/gone.rs b/gone.rs "svc/charge.go", "package svc\n\n// ccc:serves grpc billing.v1.Charge\nfunc Charge() {}\n\nfunc C() {\n\t// ccc:calls grpc ledger.v1.Write\n}\n", )]; - let surface = crate::externals::Surface::from_caches("billing", "t", &caches); + let surface = crate::externals::Surface::from_caches("billing", "t", &caches, &ContractIndex::default()); assert_eq!(surface.provides.len(), 1); assert_eq!(surface.consumes.len(), 1); assert_eq!(surface.languages, vec!["go".to_string()]); @@ -2124,11 +2372,34 @@ diff --git a/gone.rs b/gone.rs // `deps` may name a peer, but a name cannot be both a local service and a // repository somewhere else. + // The rename has to be invisible to a map.json written before it, and + // loud enough that the file gets fixed. + #[test] + fn the_old_deps_spelling_is_still_read_and_reported() { + let dir = std::env::temp_dir().join(format!("ccc-relatives-{}", std::process::id())); + let _ = fs::remove_dir_all(&dir); + fs::create_dir_all(dir.join(".ccc")).unwrap(); + let write = |body: &str| fs::write(dir.join(".ccc").join(CONFIG_NAME), body).unwrap(); + + write(r#"{"services":{"gateway":["gateway/**"]},"deps":{"gateway":["auth"]}}"#); + let cfg = ChangesConfig::load(&dir).expect("legacy map.json still loads"); + assert_eq!(cfg.relatives["gateway"], vec!["auth".to_string()]); + + // both spellings in one file is a contradiction, not a merge + write(r#"{"services":{},"deps":{"a":[]},"relatives":{"a":["b"]}}"#); + let err = format!("{:#}", ChangesConfig::load(&dir).expect_err("both keys")); + assert!(err.contains("duplicate field"), "{err}"); + + write(r#"{"services":{},"relatives":{"gateway":["auth"]}}"#); + assert!(ChangesConfig::load(&dir).is_ok()); + let _ = fs::remove_dir_all(&dir); + } + #[test] fn a_dep_may_name_an_external_but_a_name_cannot_be_both() { let cfg: ChangesConfig = serde_json::from_str( r#"{"services":{"gateway":["gateway/**"]}, - "deps":{"gateway":["billing"]}, + "relatives":{"gateway":["billing"]}, "externals":{"billing":{"path":"../billing"}}}"#, ) .expect("parse"); @@ -2136,8 +2407,13 @@ diff --git a/gone.rs b/gone.rs assert_eq!(cfg.externals["billing"].path.as_deref(), Some("../billing")); // unknown keys stay ignored, so an older ccc reads a newer map.json let old: ChangesConfig = - serde_json::from_str(r#"{"services":{},"deps":{},"future_field":42}"#).expect("parse"); + serde_json::from_str(r#"{"services":{},"relatives":{},"future_field":42}"#).expect("parse"); assert!(old.services.is_empty()); + // `relatives` was `deps` when it could only name a service in this + // repo. A map.json written then still declares what it declared. + let legacy: ChangesConfig = + serde_json::from_str(r#"{"services":{},"deps":{"gateway":["auth"]}}"#).expect("parse"); + assert_eq!(legacy.relatives["gateway"], vec!["auth".to_string()]); } #[test] @@ -2264,7 +2540,7 @@ diff --git a/gone.rs b/gone.rs let dir = std::env::temp_dir().join(format!("ccc-legacy-cfg-{}", std::process::id())); let _ = fs::remove_dir_all(&dir); fs::create_dir_all(dir.join(".ccc")).unwrap(); - let cfg = r#"{ "services": { "billing": ["billing/**"] }, "deps": { "billing": [] } }"#; + let cfg = r#"{ "services": { "billing": ["billing/**"] }, "relatives": { "billing": [] } }"#; // every name this file has had still resolves, one at a time for old in LEGACY_CONFIG_NAMES { @@ -2563,7 +2839,7 @@ diff --git a/gone.rs b/gone.rs "billing": ["billing/**"], "gateway": ["gateway/**"] }, - "deps": { "gateway": ["auth"] } + "relatives": { "gateway": ["auth"] } }"#; let base: &[(&str, &str)] = &[ (".ccc/map.json", MAP_JSON), diff --git a/src/contracts.rs b/src/contracts.rs new file mode 100644 index 0000000..00e047f --- /dev/null +++ b/src/contracts.rs @@ -0,0 +1,672 @@ +//! gRPC contracts: the rpcs a `.proto` schema declares, and the code in every +//! other language that implements or calls them. + +use crate::changes::{module_segments, path_str}; +use crate::externals::Endpoint; +use crate::languages::Language; +use crate::model::FileCache; +use globset::{GlobBuilder, GlobMatcher}; +use std::collections::{BTreeMap, HashMap}; +use std::path::{Component, Path, PathBuf}; + +pub const TRANSPORT: &str = "grpc"; +// how an endpoint derived here is told apart from a `ccc:serves` comment +pub const VIA: &str = "rpc"; + +// what a generated stub type is called after its service's name: go +// `BillingClient`, python `BillingStub`, grpc-web `BillingPromiseClient`, java +// `BillingBlockingStub` +const STUB_SUFFIXES: &[&str] = &[ + "client", + "stub", + "asyncstub", + "blockingstub", + "futurestub", + "promiseclient", + "asyncclient", + "grpcclient", + "clientimpl", +]; +// what a generated module adds to its schema's file stem: `billing_pb2_grpc`, +// `billing_grpc_pb`, `billing_connect`, tonic's `billing_client`, java +// `BillingGrpc` +const GENERATED_MARKERS: &[&str] = &[ + "pb", "grpc", "connect", "twirp", "proto", "client", "server", "servicer", "stub", +]; + +// one rpc of one service +#[derive(Debug, Clone)] +pub struct Contract { + // the wire name: `acme.billing.v1.Billing/CreateInvoice` + pub key: String, + pub package: String, + pub service: String, + pub method: String, + pub request: Option, + pub response: Option, + // the schema it is declared in, relative to the project root + pub file: String, + pub line: usize, + // (file, func) in the project's caches; `None` for a schema from `contracts` + pub def: Option<(usize, usize)>, + stem: String, +} + +// How code was tied to an rpc. Strongest first. +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Debug)] +pub enum LinkEvidence { + // a handler that takes the rpc's request message + RequestType, + // a call through the service's generated stub + StubType, + // a handler on a type named after the service, beside an import of the + // generated code + ServiceOwner, + // a call in a file that imports the generated code + GeneratedImport, +} + +impl LinkEvidence { + pub fn label(self) -> &'static str { + match self { + LinkEvidence::RequestType => "request-type", + LinkEvidence::StubType => "stub-type", + LinkEvidence::ServiceOwner => "service-owner", + LinkEvidence::GeneratedImport => "generated-import", + } + } +} + +// a function that implements an rpc +#[derive(Debug, Clone)] +pub struct Handler { + pub contract: usize, + // (file, func) in the project's caches + pub def: (usize, usize), + pub evidence: LinkEvidence, +} + +// a call site that invokes an rpc +#[derive(Debug, Clone)] +pub struct Caller { + pub contract: usize, + pub file: usize, + // index into that file's `calls` + pub call: usize, + pub evidence: LinkEvidence, +} + +#[derive(Debug, Default)] +pub struct ContractIndex { + pub contracts: Vec, + // the schemas from outside the project it was built with, kept so a peer + // repository is linked through the same ones + pub schemas: Vec, + pub handlers: Vec, + pub callers: Vec, + by_call: HashMap<(usize, usize), usize>, + by_def: HashMap<(usize, usize), usize>, +} + +impl ContractIndex { + // the index for a project: its own schemas plus the ones `.ccc/map.json` + // names under `contracts` + pub fn for_root(root: &Path, caches: &[FileCache]) -> ContractIndex { + let patterns = crate::changes::ChangesConfig::load(root) + .map(|c| c.contracts) + .unwrap_or_default(); + ContractIndex::build(caches, &load_schemas(root, &patterns)) + } + + pub fn build(caches: &[FileCache], schemas: &[FileCache]) -> ContractIndex { + let mut contracts = Vec::new(); + for (fi, c) in caches.iter().enumerate() { + declare(c, Some(fi), &mut contracts); + } + for c in schemas { + if !caches.iter().any(|p| p.rel_path == c.rel_path) { + declare(c, None, &mut contracts); + } + } + // one schema vendored into two places is still one contract; the + // project's own copy was declared first and wins + let mut seen = std::collections::BTreeSet::new(); + contracts.retain(|c: &Contract| seen.insert(c.key.clone())); + + let mut by_fold: BTreeMap> = BTreeMap::new(); + for (i, c) in contracts.iter().enumerate() { + by_fold.entry(fold(&c.method)).or_default().push(i); + } + // (owning type, method) the project defines - a call through one of + // those is a local method call, whatever it is named + let mut project_methods = std::collections::BTreeSet::new(); + for c in caches { + for f in &c.funcs { + if let Some(o) = &f.owner { + project_methods.insert((o.as_str(), f.name.as_str())); + } + } + } + + let mut idx = ContractIndex { contracts, schemas: schemas.to_vec(), ..Default::default() }; + if by_fold.is_empty() { + return idx; + } + for (fi, cache) in caches.iter().enumerate() { + if cache.language == Language::Proto || is_generated(&cache.rel_path) { + continue; + } + let generated: Vec = + idx.contracts.iter().map(|c| imports_generated(cache, c)).collect(); + + for (ki, f) in cache.funcs.iter().enumerate() { + let Some(cands) = by_fold.get(&fold(&f.name)) else { continue }; + let best = pick(cands.iter().filter_map(|&ci| { + let c = &idx.contracts[ci]; + // a client wrapper shaped like the rpc is not a handler of it + if f.owner.as_deref().is_some_and(|o| is_stub(o, &c.service)) { + return None; + } + if c.request.as_ref().is_some_and(|r| f.param_types.contains(r)) { + return Some((ci, LinkEvidence::RequestType)); + } + let owned = f.owner.as_deref().is_some_and(|o| { + o.to_ascii_lowercase().contains(&c.service.to_ascii_lowercase()) + }); + (owned && generated[ci]).then_some((ci, LinkEvidence::ServiceOwner)) + })); + if let Some((contract, evidence)) = best { + idx.by_def.insert((fi, ki), idx.handlers.len()); + idx.handlers.push(Handler { contract, def: (fi, ki), evidence }); + } + } + + for (ci_call, call) in cache.calls.iter().enumerate() { + let Some(cands) = by_fold.get(&fold(&call.name)) else { continue }; + let best = pick(cands.iter().filter_map(|&ci| { + let c = &idx.contracts[ci]; + if let Some(t) = call.recv_type.as_deref() { + if is_stub(t, &c.service) { + return Some((ci, LinkEvidence::StubType)); + } + // the project's own method on a type it defines + if project_methods.contains(&(t, call.name.as_str())) { + return None; + } + } + generated[ci].then_some((ci, LinkEvidence::GeneratedImport)) + })); + if let Some((contract, evidence)) = best { + idx.by_call.insert((fi, ci_call), idx.callers.len()); + idx.callers.push(Caller { contract, file: fi, call: ci_call, evidence }); + } + } + } + idx + } + + // the rpc a call site invokes, if it invokes one + pub fn caller(&self, file: usize, call: usize) -> Option<&Caller> { + self.by_call.get(&(file, call)).map(|&i| &self.callers[i]) + } + + // the rpc a definition implements, if it implements one + pub fn handler(&self, def: (usize, usize)) -> Option<&Handler> { + self.by_def.get(&def).map(|&i| &self.handlers[i]) + } + + pub fn handlers_of(&self, contract: usize) -> impl Iterator { + self.handlers.iter().filter(move |h| h.contract == contract) + } + + pub fn callers_of(&self, contract: usize) -> impl Iterator { + self.callers.iter().filter(move |c| c.contract == contract) + } + + // Every definition a call to `contract` reaches: the rpc in the schema + // when the project holds it, and each handler. + pub fn targets(&self, contract: usize) -> Vec<(usize, usize)> { + self.contracts[contract] + .def + .into_iter() + .chain(self.handlers_of(contract).map(|h| h.def)) + .collect() + } + + // The rpcs a lookup names: `CreateInvoice`, `create_invoice`, + // `Billing.CreateInvoice`, `acme.billing.v1.Billing/CreateInvoice`. A + // qualifier has to name the service or a segment of its package. + pub fn matching(&self, name: &str, qualifier: Option<&str>) -> Vec { + let folded = fold(name); + self.contracts + .iter() + .enumerate() + .filter(|(_, c)| fold(&c.method) == folded) + .filter(|(_, c)| { + qualifier.map_or(true, |q| { + module_segments(q).all(|seg| { + seg.eq_ignore_ascii_case(&c.service) + || c.package.split('.').any(|p| p.eq_ignore_ascii_case(seg)) + }) + }) + }) + .map(|(i, _)| i) + .collect() + } + + // The boundary endpoints this project serves and calls, for its surface: + // what lets a peer in another repository link to it without a comment + // written at either end. + pub fn endpoints(&self, caches: &[FileCache]) -> (Vec, Vec) { + let endpoint = |contract: usize, file: usize, function: &str, line: usize| Endpoint { + key: self.contracts[contract].key.clone(), + transport: TRANSPORT.to_string(), + function: function.to_string(), + file: path_str(&caches[file].rel_path), + line, + service: None, + via: Some(VIA.to_string()), + }; + let provides = self + .handlers + .iter() + .map(|h| { + let f = &caches[h.def.0].funcs[h.def.1]; + endpoint(h.contract, h.def.0, &f.name, f.line) + }) + .collect(); + let consumes = self + .callers + .iter() + .map(|c| { + let site = &caches[c.file].calls[c.call]; + endpoint(c.contract, c.file, &site.caller, site.line) + }) + .collect(); + (provides, consumes) + } +} + +// the strongest link, or none when two rpcs tie for it +fn pick(links: impl Iterator) -> Option<(usize, LinkEvidence)> { + let mut links: Vec<(usize, LinkEvidence)> = links.collect(); + links.sort_by_key(|&(_, e)| e); + match links[..] { + [] => None, + [only] => Some(only), + [(a, ea), (b, eb), ..] => (a == b || ea < eb).then_some((a, ea)), + } +} + +// every rpc a schema file declares +fn declare(cache: &FileCache, fi: Option, out: &mut Vec) { + if cache.language != Language::Proto { + return; + } + let package = cache.modules.first().cloned().unwrap_or_default(); + let stem = cache + .rel_path + .file_stem() + .and_then(|s| s.to_str()) + .unwrap_or_default() + .to_string(); + for (ki, f) in cache.funcs.iter().enumerate() { + let Some(service) = f.owner.clone() else { continue }; + let key = if package.is_empty() { + format!("{service}/{}", f.name) + } else { + format!("{package}.{service}/{}", f.name) + }; + out.push(Contract { + key, + package: package.clone(), + service, + method: f.name.clone(), + request: f.param_types.first().cloned(), + response: f.ret.as_deref().map(crate::extract::normalize_type), + file: path_str(&cache.rel_path), + line: f.line, + def: fi.map(|fi| (fi, ki)), + stem: stem.clone(), + }); + } +} + +// an rpc's name as every language's generator spells it: `CreateInvoice`, +// `createInvoice` and `create_invoice` are one method +fn fold(name: &str) -> String { + name.chars() + .filter(|c| *c != '_') + .map(|c| c.to_ascii_lowercase()) + .collect() +} + +// is `ty` the generated stub for `service` +fn is_stub(ty: &str, service: &str) -> bool { + let ty = ty.to_ascii_lowercase(); + ty.strip_prefix(&service.to_ascii_lowercase()) + .is_some_and(|rest| STUB_SUFFIXES.contains(&rest)) +} + +// Does this file import code generated from the contract's schema? Either a +// module named after the schema file (`billing_pb2_grpc`, `billing_grpc_pb`), +// or one whose path ends in the schema's package (`.../billing/v1`, +// `acme.billing.v1`, `Acme.Billing.V1`). +fn imports_generated(cache: &FileCache, c: &Contract) -> bool { + let stem = c.stem.to_ascii_lowercase(); + let pkg: Vec = c.package.split('.').map(|s| s.to_ascii_lowercase()).collect(); + // one segment names too little to tell a generated package from any other + let tail = (pkg.len() >= 2).then(|| &pkg[pkg.len() - 2..]); + cache.imports.iter().any(|imp| { + let segs: Vec = module_segments(&imp.module).map(|s| s.to_ascii_lowercase()).collect(); + let by_package = tail.is_some_and(|t| segs.windows(t.len()).any(|w| w == t)); + let by_stem = segs + .iter() + .map(String::as_str) + .chain(imp.names.iter().map(String::as_str)) + .any(|seg| { + seg.to_ascii_lowercase() + .strip_prefix(&stem) + .is_some_and(|rest| GENERATED_MARKERS.iter().any(|m| rest.contains(m))) + }); + by_package || by_stem + }) +} + +// Code protoc wrote. It defines every rpc and calls none, so reading it as +// handlers would link every contract to its own stubs. +pub fn is_generated(path: &Path) -> bool { + let Some(name) = path.file_name().and_then(|n| n.to_str()) else { + return false; + }; + let name = name.to_ascii_lowercase(); + let stem = name.split('.').next().unwrap_or(&name); + name.contains(".pb.") + || name.contains(".connect.") + || name.contains(".twirp.") + || name.contains(".pb2") + || [ + "_pb2", "_pb2_grpc", "_pb", "_grpc_pb", "_grpc_web_pb", "_connect", "_connectweb", + ] + .iter() + .any(|s| stem.ends_with(s)) + // c# puts the stubs in `BillingGrpc.cs` + || (name.ends_with(".cs") && stem.ends_with("grpc")) +} + +// Parse the schemas `contracts` names. A pattern is a file, a directory (every +// `.proto` under it), or a glob; relative ones resolve against `root`, and may +// leave it (`../protos/**/*.proto`). Ignore files are not consulted: vendored +// schemas are routinely ignored, and naming them here is the opt-in. +pub fn load_schemas(root: &Path, patterns: &[String]) -> Vec { + let mut files: Vec = Vec::new(); + for pattern in patterns { + let (base, glob) = split_glob(pattern); + let base = if base.is_absolute() { base } else { root.join(base) }; + let matcher: Option = glob.and_then(|g| { + GlobBuilder::new(&g) + .literal_separator(true) + .build() + .ok() + .map(|g| g.compile_matcher()) + }); + if base.is_file() { + files.push(base); + continue; + } + for dent in ignore::WalkBuilder::new(&base).standard_filters(false).build().flatten() { + let path = dent.path(); + if Language::from_path(path) != Some(Language::Proto) || !path.is_file() { + continue; + } + let rel = path.strip_prefix(&base).unwrap_or(path); + if matcher.as_ref().map_or(true, |m| m.is_match(rel)) { + files.push(path.to_path_buf()); + } + } + } + files.sort(); + files.dedup(); + crate::scan::build_caches(root, &files) +} + +// `../protos/acme/**/*.proto` -> (`../protos/acme`, `**/*.proto`) +fn split_glob(pattern: &str) -> (PathBuf, Option) { + let mut base = PathBuf::new(); + let mut rest: Vec = Vec::new(); + for comp in Path::new(pattern).components() { + let s = comp.as_os_str().to_string_lossy().to_string(); + let globby = s.contains(['*', '?', '[', '{']); + if rest.is_empty() && !globby { + match comp { + Component::CurDir => {} + _ => base.push(comp), + } + } else { + rest.push(s); + } + } + let glob = (!rest.is_empty()).then(|| rest.join("/")); + (base, glob) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::extract::extract; + use crate::model::FileCache; + + fn cache(rel: &str, src: &str) -> FileCache { + let lang = Language::from_path(Path::new(rel)).unwrap(); + let ex = extract(lang, src).unwrap(); + FileCache { + cache_name: rel.to_string(), + display_name: rel.to_string(), + rel_path: PathBuf::from(rel), + language: lang, + lines: src.lines().count(), + consts: ex.consts, + funcs: ex.funcs, + refs: ex.refs, + notes: ex.notes, + calls: ex.calls, + uses: ex.uses, + imports: ex.imports, + types: ex.types, + modules: ex.modules, + annotations: ex.annotations, + } + } + + const SCHEMA: &str = "syntax = \"proto3\";\n\ + package acme.billing.v1;\n\ + message CreateInvoiceRequest { string customer = 1; }\n\ + message Invoice { string id = 1; }\n\ + service Billing {\n\ + \x20 rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice);\n\ + }\n"; + + fn polyglot() -> Vec { + vec![ + cache( + "gosvc/server.go", + "package gosvc\n\ + import billingv1 \"github.com/acme/protos/gen/go/acme/billing/v1\"\n\ + type server struct{}\n\ + func (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\ + \treturn nil, nil\n\ + }\n", + ), + cache( + "gosvc/client.go", + "package gosvc\n\ + import billingv1 \"github.com/acme/protos/gen/go/acme/billing/v1\"\n\ + func Charge(conn *grpc.ClientConn) {\n\ + \tclient := billingv1.NewBillingClient(conn)\n\ + \tclient.CreateInvoice(ctx, &billingv1.CreateInvoiceRequest{})\n\ + }\n", + ), + cache( + "gosvc/billing_grpc.pb.go", + "package billingv1\n\ + type billingClient struct{}\n\ + func (c *billingClient) CreateInvoice(ctx context.Context, in *CreateInvoiceRequest) (*Invoice, error) {\n\ + \treturn nil, nil\n\ + }\n", + ), + cache( + "py/service.py", + "from acme.billing.v1 import billing_pb2, billing_pb2_grpc\n\ + class BillingService(billing_pb2_grpc.BillingServicer):\n\ + \x20 def CreateInvoice(self, request, context):\n\ + \x20 return billing_pb2.Invoice()\n", + ), + cache( + "py/client.py", + "from acme.billing.v1 import billing_pb2_grpc\n\ + def charge(channel):\n\ + \x20 stub = billing_pb2_grpc.BillingStub(channel)\n\ + \x20 return stub.CreateInvoice(None)\n", + ), + cache( + "ts/charge.ts", + "import { createClient } from \"@connectrpc/connect\";\n\ + import { Billing } from \"./gen/acme/billing/v1/billing_connect\";\n\ + export async function charge(t: any) {\n\ + \x20 const client = createClient(Billing, t);\n\ + \x20 await client.createInvoice({});\n\ + }\n", + ), + cache( + "rs/src/lib.rs", + "use billing::v1::billing_client::BillingClient;\n\ + pub async fn charge() {\n\ + \x20 let mut client = BillingClient::connect(\"http://x\").await.unwrap();\n\ + \x20 client.create_invoice(req).await;\n\ + }\n", + ), + // a same-named method with no tie to the schema + cache( + "ts/ledger.ts", + "export function record(l: Ledger) { l.createInvoice(); }\n", + ), + cache("proto/acme/billing/v1/billing.proto", SCHEMA), + ] + } + + fn at(caches: &[FileCache], file: usize) -> &str { + caches[file].rel_path.to_str().unwrap() + } + + #[test] + fn every_language_is_tied_to_the_rpc_by_evidence() { + let caches = polyglot(); + let idx = ContractIndex::build(&caches, &[]); + assert_eq!(idx.contracts.len(), 1); + assert_eq!(idx.contracts[0].key, "acme.billing.v1.Billing/CreateInvoice"); + assert_eq!(idx.contracts[0].request.as_deref(), Some("CreateInvoiceRequest")); + + let handlers: Vec<(&str, &str)> = idx + .handlers + .iter() + .map(|h| (at(&caches, h.def.0), h.evidence.label())) + .collect(); + // the generated client is neither a handler nor a caller + assert_eq!( + handlers, + vec![("gosvc/server.go", "request-type"), ("py/service.py", "service-owner")] + ); + + let callers: Vec<(&str, &str)> = idx + .callers + .iter() + .map(|c| (at(&caches, c.file), c.evidence.label())) + .collect(); + // `ts/ledger.ts` shares the name and nothing else + assert_eq!( + callers, + vec![ + ("gosvc/client.go", "stub-type"), + ("py/client.py", "generated-import"), + ("ts/charge.ts", "generated-import"), + ("rs/src/lib.rs", "stub-type"), + ] + ); + // a call reaches the schema and every handler + assert_eq!(idx.targets(0).len(), 3); + } + + #[test] + fn a_schema_outside_the_project_still_links_its_callers() { + let mut caches = polyglot(); + let schema = caches.pop().unwrap(); + let idx = ContractIndex::build(&caches, &[schema]); + assert_eq!(idx.contracts[0].def, None); + assert_eq!(idx.callers.len(), 4); + // with no definition in the project, a call reaches the handlers only + assert_eq!(idx.targets(0).len(), 2); + } + + #[test] + fn two_services_with_one_method_name_link_by_stub_and_otherwise_not_at_all() { + let two = "syntax = \"proto3\";\npackage acme.v1;\n\ + service Billing { rpc Get(A) returns (B); }\n\ + service Ledger { rpc Get(C) returns (D); }\n"; + let caches = vec![ + cache("acme.proto", two), + cache( + "a.go", + "package a\nimport v1 \"x/acme/v1\"\n\ + func F(c v1.LedgerClient) { c.Get(ctx, nil) }\n\ + func G(c other) { c.Get(ctx, nil) }\n", + ), + ]; + let idx = ContractIndex::build(&caches, &[]); + let linked: Vec<&str> = idx + .callers + .iter() + .map(|c| idx.contracts[c.contract].service.as_str()) + .collect(); + // `G` imports the generated package, but so would a call to either + // service - that is a tie, and a tie links nothing + assert_eq!(linked, vec!["Ledger"]); + } + + #[test] + fn a_lookup_matches_any_spelling_and_a_qualifier_must_agree() { + let idx = ContractIndex::build(&polyglot(), &[]); + assert_eq!(idx.matching("create_invoice", None), vec![0]); + assert_eq!(idx.matching("createInvoice", Some("Billing")), vec![0]); + assert_eq!(idx.matching("CreateInvoice", Some("acme.billing.v1.Billing")), vec![0]); + assert!(idx.matching("CreateInvoice", Some("Ledger")).is_empty()); + } + + #[test] + fn generated_files_are_recognised_by_name() { + for f in [ + "x/billing.pb.go", + "x/billing_grpc.pb.go", + "x/billing_pb2.py", + "x/billing_pb2_grpc.py", + "x/billing_grpc_pb.js", + "x/billing_pb.d.ts", + "x/billing_connect.ts", + "x/billingv1connect/billing.connect.go", + "x/BillingGrpc.cs", + ] { + assert!(is_generated(Path::new(f)), "{f}"); + } + for f in ["x/server_grpc.go", "x/billing.go", "x/grpc.rs", "x/client.ts"] { + assert!(!is_generated(Path::new(f)), "{f}"); + } + } + + #[test] + fn a_glob_splits_into_the_directory_to_walk_and_the_pattern_under_it() { + assert_eq!( + split_glob("../protos/acme/**/*.proto"), + (PathBuf::from("../protos/acme"), Some("**/*.proto".to_string())) + ); + assert_eq!(split_glob("./third_party/proto"), (PathBuf::from("third_party/proto"), None)); + } +} diff --git a/src/coverage.rs b/src/coverage.rs index 93947e7..3cc700a 100644 --- a/src/coverage.rs +++ b/src/coverage.rs @@ -1,29 +1,7 @@ //! Which tests exercise which functions. -//! -//! One relation, built once and read by everything that reports coverage: -//! `changes` (`tested` / `tested_by`), `insights::test_targets` (`covered` / -//! `covered_by`) and the trigger walk. They used to compute it separately, each -//! keyed on the callee's bare name, which made a test a "cover" of every -//! same-named function in the repository - a Rust test calling `std::fs::write` -//! was reported as covering a TypeScript method called `write`. -//! -//! A test reference is tied to a *definition*, addressed by (file, index into -//! that file's `funcs`), and only when something beyond the name agrees: -//! -//! 1. calls whose receiver or qualifier resolves to nothing in the project -//! are external - `std::fs::write` covers nothing, it leaves the project; -//! 2. a candidate must be in the caller's runtime family, so a name shared by -//! two ecosystems is never one definition; -//! 3. what is left needs evidence - the receiver's declared type, the same -//! file, an import, or a qualifier naming the defining file - and where -//! the evidence fits more than one definition it produces nothing, the -//! same discipline `insights::build_graph` applies to call edges. -//! -//! Name-only matching survives in one corner: an untyped language calling a -//! name with exactly one definition in the project. It is labelled as such, -//! since it is the weakest thing here that is still worth reporting. use crate::changes::{is_test_fn_name, is_test_path, module_segments, names_project, path_str}; +use crate::contracts::ContractIndex; use crate::extract::TOP_LEVEL; use crate::model::FileCache; use std::collections::{BTreeMap, BTreeSet}; @@ -65,6 +43,9 @@ pub enum Evidence { Import, // a qualifier segment names the defining file's module, type, stem or dir Qualifier, + // the call goes through an rpc's generated stub, which reaches the rpc and + // whatever implements it, in any language - see `contracts` + Rpc, // untyped language, and the name has exactly one definition in the project NameOnly, } @@ -77,6 +58,7 @@ impl Evidence { Evidence::SamePackage => "same-package", Evidence::Import => "import", Evidence::Qualifier => "qualifier", + Evidence::Rpc => "rpc", Evidence::NameOnly => "name-only", } } @@ -270,7 +252,11 @@ fn stem_of(c: &FileCache) -> &str { // Build the coverage relation. `project_ids` are the identities declared by // manifests (crate name, go module path, npm name), so an integration test // calling `mycrate::parse` is understood as staying inside the project. -pub fn build(caches: &[FileCache], project_ids: &BTreeSet) -> CoverageIndex { +pub fn build( + caches: &[FileCache], + project_ids: &BTreeSet, + contracts: &ContractIndex, +) -> CoverageIndex { let aliases = file_aliases(caches); let imported = imported_names(caches); let alias_any: BTreeSet<&str> = aliases.iter().flatten().map(String::as_str).collect(); @@ -319,7 +305,7 @@ pub fn build(caches: &[FileCache], project_ids: &BTreeSet) -> CoverageIn for (a, c) in caches.iter().enumerate() { let path = path_str(&c.rel_path); let file_is_test = is_test_path(&path); - for call in &c.calls { + for (ci, call) in c.calls.iter().enumerate() { if !(file_is_test || call.test_ctx || is_test_fn_name(&call.caller)) { continue; } @@ -327,7 +313,13 @@ pub fn build(caches: &[FileCache], project_ids: &BTreeSet) -> CoverageIn if selectable { tests.insert((a, call.caller.as_str())); } - let matched = resolve( + // An rpc is the one call that crosses runtime families on purpose: + // a python test through the stub exercises the go handler behind + // it. Settled before `resolve`, whose family rule would refuse it. + let rpc = contracts + .caller(a, ci) + .map(|link| (contracts.targets(link.contract), Evidence::Rpc)); + let matched = rpc.or_else(|| resolve( Site { file: a, name: call.name.as_str(), @@ -348,7 +340,7 @@ pub fn build(caches: &[FileCache], project_ids: &BTreeSet) -> CoverageIn families: &families, dirs: &dirs, }, - ); + )); let Some((defs, evidence)) = matched else { external_calls += 1; continue; @@ -583,7 +575,43 @@ mod tests { } fn index(caches: &[FileCache]) -> CoverageIndex { - build(caches, &BTreeSet::new()) + build(caches, &BTreeSet::new(), &ContractIndex::default()) + } + + // A python test through the generated stub exercises the go handler behind + // it. The family rule refuses that pairing for every other call, and must + // keep doing so: without the schema this is two unrelated functions. + #[test] + fn a_test_through_an_rpc_stub_covers_the_handler_in_another_language() { + let (_dir, caches) = caches( + "rpc", + &[ + ("proto/billing.proto", "syntax = \"proto3\";\npackage acme.billing.v1;\nmessage CreateInvoiceRequest { string customer = 1; }\nmessage Invoice { string id = 1; }\nservice Billing { rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice); }\n"), + ("svc/server.go", "package svc\nimport billingv1 \"github.com/acme/gen/acme/billing/v1\"\ntype server struct{}\nfunc (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\treturn nil, nil\n}\n"), + ( + "tests/test_billing.py", + "from acme.billing.v1 import billing_pb2_grpc\n\ + def test_create_invoice(channel):\n\ + \x20 stub = billing_pb2_grpc.BillingStub(channel)\n\ + \x20 stub.CreateInvoice(None)\n", + ), + ], + ); + let handler = def(&caches, "svc/server.go", "CreateInvoice"); + let schema = def(&caches, "proto/billing.proto", "CreateInvoice"); + + assert!(!index(&caches).is_covered(handler), "no schema index, no link"); + + let contracts = ContractIndex::build(&caches, &[]); + let cov = build(&caches, &BTreeSet::new(), &contracts); + for d in [handler, schema] { + let by: Vec<(&str, &str)> = cov + .covering(d) + .iter() + .map(|r| (r.site.name.as_str(), r.evidence.label())) + .collect(); + assert_eq!(by, vec![("test_create_invoice", "rpc")]); + } } fn def(caches: &[FileCache], file: &str, name: &str) -> (usize, usize) { @@ -762,7 +790,7 @@ mod tests { } } } - let cov = build(&caches, &BTreeSet::new()); + let cov = build(&caches, &BTreeSet::new(), &ContractIndex::default()); let (mut before, mut after, mut total) = (0usize, 0usize, 0usize); let mut lost: Vec<(String, String, usize)> = Vec::new(); diff --git a/src/externals.rs b/src/externals.rs index d981367..399a4f5 100644 --- a/src/externals.rs +++ b/src/externals.rs @@ -1,22 +1,17 @@ //! Cross-repo links. //! -//! calls do not stop at the process: a gateway calls a billing -//! service that lives in another repository, in another language, behind an -//! HTTP or gRPC hop that no parser can follow. `.ccc/map.json` names those -//! peers under `externals`, and `ccc:serves` / `ccc:calls` comments name the -//! key both ends agree on. Matching keys become real edges of the service -//! graph, with a file and line at each end. -//! +//! calls do not stop at the process //! A peer is reached one of two ways, and both reduce to the same [`Surface`]: //! //! - `path` - a directory: a sibling checkout, or another corner of a //! monorepo. ccc parses it and derives the surface itself. //! - `surface` - a file or URL holding a surface this peer published with -//! `ccc export`. No source, no toolchain for its language, no clone. +//! `ccc init`. No source, no toolchain for its language, no clone. //! //! Peer files deliberately never join `caches` //! everything downstream keys on paths relative to *this* root +use crate::contracts::ContractIndex; use crate::model::{Boundary, FileCache}; use crate::scan; use anyhow::{bail, Context, Result}; @@ -27,7 +22,19 @@ use std::process::Command; pub const SURFACE_SCHEMA: &str = "ccc-surface/1"; // the conventional file name so `surface` can name a directory -pub const SURFACE_NAME: &str = "ccc-surface.json"; +pub const SURFACE_NAME: &str = "surface.json"; +// what `ccc export` wrote before `ccc init` took the job. Still read, never +// written: a peer repo that has not regenerated is not a peer that stopped +// existing +pub const LEGACY_SURFACE_NAME: &str = "ccc-surface.json"; + +// the surface a directory publishes, current name first +fn surface_in(dir: &Path) -> Option { + std::iter::once(SURFACE_NAME) + .chain(std::iter::once(LEGACY_SURFACE_NAME)) + .map(|n| dir.join(n)) + .find(|p| p.is_file()) +} // a published surface *should* be small, anything larger considered a pebkac const MAX_SURFACE_BYTES: u64 = 8 * 1024 * 1024; const FETCH_TIMEOUT_SECS: u64 = 20; @@ -44,6 +51,10 @@ pub struct Endpoint { // the service inside that repo when it names more than one #[serde(default, skip_serializing_if = "Option::is_none")] pub service: Option, + // `rpc` when derived from a `.proto` schema rather than written in a + // `ccc:` comment + #[serde(default, skip_serializing_if = "Option::is_none")] + pub via: Option, } // what a repository publishes and consumes and nothing else @@ -62,14 +73,28 @@ pub struct Surface { pub provides: Vec, #[serde(default)] pub consumes: Vec, + // rpc endpoints were derived from `.proto` schemas. A surface without the + // flag predates that and says nothing about the rpcs its repo serves + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub rpc_endpoints: bool, } impl Surface { // derive a surface from an already-parsed tree - pub fn from_caches(name: &str, generated: &str, caches: &[FileCache]) -> Surface { - let mut provides = Vec::new(); - let mut consumes = Vec::new(); + pub fn from_caches( + name: &str, + generated: &str, + caches: &[FileCache], + contracts: &ContractIndex, + ) -> Surface { + // the rpcs this repo serves and calls need no comment at either end + let (mut provides, mut consumes) = contracts.endpoints(caches); let mut languages = BTreeSet::new(); + for e in provides.iter().chain(&consumes) { + if let Some(c) = caches.iter().find(|c| crate::changes::path_str(&c.rel_path) == e.file) { + languages.insert(c.language.as_str().to_string()); + } + } for cache in caches { if cache.annotations.is_empty() { @@ -85,6 +110,7 @@ impl Surface { file: file.clone(), line: ann.line, service: None, + via: None, }; match ann.boundary { Boundary::Serves => provides.push(endpoint), @@ -103,6 +129,7 @@ impl Surface { languages: languages.into_iter().collect(), provides, consumes, + rpc_endpoints: true, } } @@ -169,17 +196,25 @@ impl ExternalService { } // Resolve every peer named in the config. Errors are captured per peer. +// `schemas` are this repo's `contracts`: a peer checkout is linked through +// them as well as its own, since the schema often lives in neither repo. pub fn resolve_all( root: &Path, externals: &BTreeMap, + schemas: &[FileCache], ) -> Vec { externals .iter() - .map(|(name, config)| resolve_one(root, name, config)) + .map(|(name, config)| resolve_one(root, name, config, schemas)) .collect() } -fn resolve_one(root: &Path, name: &str, config: &ExternalRepo) -> ExternalService { +fn resolve_one( + root: &Path, + name: &str, + config: &ExternalRepo, + schemas: &[FileCache], +) -> ExternalService { let mut service = ExternalService { name: name.to_string(), config: config.clone(), @@ -194,7 +229,7 @@ fn resolve_one(root: &Path, name: &str, config: &ExternalRepo) -> ExternalServic let dir = resolve_path(root, path); service.source = format!("path {}", path); if dir.is_dir() { - match surface_from_dir(name, &dir) { + match surface_from_dir(name, &dir, schemas) { Ok(surface) => { service.surface = Some(surface); return service; @@ -238,14 +273,18 @@ fn resolve_path(root: &Path, path: &str) -> PathBuf { } // Parse a peer checkout and reduce it to its surface. -fn surface_from_dir(name: &str, dir: &Path) -> Result { +fn surface_from_dir(name: &str, dir: &Path, schemas: &[FileCache]) -> Result { // a checkout that already publishes one is cheaper, and is what its owners // consider their contract - let published = dir.join(".ccc").join(SURFACE_NAME); - if published.is_file() { + if let Some(published) = surface_in(&dir.join(".ccc")) { let raw = std::fs::read_to_string(&published) .with_context(|| format!("reading {}", published.display()))?; - if let Ok(mut surface) = Surface::parse(&raw, &published.display().to_string()) { + // one published before rpc endpoints existed would hide every rpc + // the checkout serves, and the checkout is right here to read + if let Ok(mut surface) = Surface::parse(&raw, &published.display().to_string()) + .map_err(|_| ()) + .and_then(|s| if s.rpc_endpoints { Ok(s) } else { Err(()) }) + { surface.name = name.to_string(); return Ok(surface); } @@ -255,7 +294,14 @@ fn surface_from_dir(name: &str, dir: &Path) -> Result { let files = scan::collect_files(dir) .with_context(|| format!("scanning external '{name}' at {}", dir.display()))?; let caches = scan::build_caches(dir, &files); - Ok(Surface::from_caches(name, &crate::render::now_ts(), &caches)) + // the peer's own `contracts`, then ours + let peer_patterns = crate::changes::ChangesConfig::load(dir) + .map(|c| c.contracts) + .unwrap_or_default(); + let mut all = crate::contracts::load_schemas(dir, &peer_patterns); + all.extend(schemas.iter().cloned()); + let contracts = ContractIndex::build(&caches, &all); + Ok(Surface::from_caches(name, &crate::render::now_ts(), &caches, &contracts)) } // Read a surface from a file, a directory holding one, or a URL. @@ -266,7 +312,8 @@ fn load_surface(root: &Path, location: &str, auth: Option<&str>) -> Result MAX_SURFACE_BYTES { @@ -344,6 +391,8 @@ pub struct Crossing { pub remote: Option, // the peer is another repository rather than a service in this one pub external: bool, + // tied by a `.proto` schema rather than a `ccc:` comment + pub rpc: bool, } // index a peers endpoints by key @@ -359,3 +408,86 @@ pub fn index_by_key(endpoints: &[Endpoint]) -> BTreeMap<&str, Vec<&Endpoint>> { pub fn norm_key(key: &str) -> String { key.trim().to_ascii_lowercase() } + +#[cfg(test)] +mod tests { + use super::*; + + // A peer checkout serves an rpc from a schema only this repo holds, and + // sits beside a surface it published before rpc endpoints existed. The + // stale surface would hide the handler, so the checkout is read instead. + #[test] + fn a_peer_serves_an_rpc_through_this_repos_schema() { + let base = std::env::temp_dir().join(format!("ccc-peer-rpc-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&base); + let (root, peer) = (base.join("app"), base.join("billing")); + std::fs::create_dir_all(root.join("proto")).unwrap(); + std::fs::create_dir_all(peer.join(".ccc")).unwrap(); + std::fs::write( + root.join("proto/billing.proto"), + "syntax = \"proto3\";\npackage acme.billing.v1;\n\ + message CreateInvoiceRequest { string customer = 1; }\n\ + message Invoice { string id = 1; }\n\ + service Billing { rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice); }\n", + ) + .unwrap(); + std::fs::write( + peer.join("server.go"), + "package billing\nimport billingv1 \"github.com/acme/gen/acme/billing/v1\"\n\ + type server struct{}\n\ + func (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\ + \treturn nil, nil\n}\n", + ) + .unwrap(); + std::fs::write( + peer.join(".ccc").join(SURFACE_NAME), + format!("{{\"schema\": \"{SURFACE_SCHEMA}\", \"name\": \"billing\", \"generated\": \"old\"}}"), + ) + .unwrap(); + + let schemas = crate::contracts::load_schemas(&root, &["proto".to_string()]); + let config = ExternalRepo { path: Some("../billing".to_string()), ..Default::default() }; + let peers = resolve_all(&root, &BTreeMap::from([("billing".to_string(), config)]), &schemas); + let _ = std::fs::remove_dir_all(&base); + + let surface = peers[0].surface.as_ref().expect("resolved"); + assert!(surface.rpc_endpoints); + let served: Vec<(&str, &str, Option<&str>)> = surface + .provides + .iter() + .map(|e| (e.key.as_str(), e.file.as_str(), e.via.as_deref())) + .collect(); + assert_eq!( + served, + vec![("acme.billing.v1.Billing/CreateInvoice", "server.go", Some("rpc"))] + ); + assert_eq!(surface.languages, vec!["go".to_string()]); + } + + // Renaming what we write must not unpublish what peers already published. + // A repo that last ran `ccc export` still holds `ccc-surface.json`, and it + // is still a peer. + #[test] + fn a_peer_that_published_under_the_old_name_is_still_found() { + let dir = std::env::temp_dir().join(format!("ccc-surf-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(&dir).expect("mkdir"); + + assert!(surface_in(&dir).is_none(), "an empty directory publishes nothing"); + + std::fs::write(dir.join(LEGACY_SURFACE_NAME), "{}").expect("write"); + assert_eq!( + surface_in(&dir).map(|p| p.file_name().unwrap().to_owned()), + Some(LEGACY_SURFACE_NAME.into()) + ); + + // and once it regenerates, the current name is the one that answers + std::fs::write(dir.join(SURFACE_NAME), "{}").expect("write"); + assert_eq!( + surface_in(&dir).map(|p| p.file_name().unwrap().to_owned()), + Some(SURFACE_NAME.into()) + ); + + let _ = std::fs::remove_dir_all(&dir); + } +} diff --git a/src/extract.rs b/src/extract.rs index 88d38d9..23e77e9 100644 --- a/src/extract.rs +++ b/src/extract.rs @@ -455,6 +455,11 @@ fn type_scope(node: Node, ctx: &Ctx) -> Option<(String, Option)> { let name = type_def_name(node, ctx)?; Some((oneline(text(name, ctx.src)), Some("self".to_string()))) } + // rpcs are the service's methods; there is no receiver to call through + (Language::Proto, "service") => { + let name = type_def_name(node, ctx)?; + Some((oneline(text(name, ctx.src)), None)) + } _ => None, } } @@ -560,7 +565,7 @@ fn const_eligible(ctx: &Ctx) -> bool { | Language::C | Language::Zig => !ctx.in_type(), Language::CSharp => true, - Language::Rust | Language::Go | Language::Odin => true, + Language::Rust | Language::Go | Language::Odin | Language::Proto => true, } } @@ -659,18 +664,11 @@ pub fn normalize_type(raw: &str) -> String { // side may be absent: an unannotated JS parameter has no type, a C++ // `void f(int)` has no name. fn param_pairs(node: Node, ctx: &Ctx) -> Vec<(Option, Option)> { - let kinds = ctx.lang.param_list_kinds(); - let mut list = None; - let mut stack = vec![node]; - while let Some(n) = stack.pop() { - if kinds.contains(&n.kind()) { - list = Some(n); - break; - } - let mut c = n.walk(); - stack.extend(n.children(&mut c)); + // an rpc takes exactly one unnamed request message + if ctx.lang == Language::Proto { + return proto_rpc_types(node, ctx).0.map(|t| (None, Some(t))).into_iter().collect(); } - let Some(list) = list else { return Vec::new() }; + let Some(list) = param_list(node, ctx) else { return Vec::new() }; let mut out = Vec::new(); let mut cursor = list.walk(); for p in list.named_children(&mut cursor) { @@ -770,6 +768,8 @@ fn bind_declaration(node: Node, ctx: &mut Ctx) { let value = node .child_by_field_name("value") .or_else(|| node.child_by_field_name("declarator")) + // go `x := ...` + .or_else(|| node.child_by_field_name("right")) .map(name_of); let ty = declared.or_else(|| value.as_deref().and_then(constructed_type)); let Some(ty) = ty else { return }; @@ -823,7 +823,8 @@ fn constructed_type(value: &str) -> Option { // `Client::new(...)` / `Client::default()` - the type is the qualifier if let Some((head, tail)) = v.split_once("::") { let ctor = tail.split(['(', ':']).next().unwrap_or(""); - if matches!(ctor, "new" | "default" | "from" | "with_capacity" | "create") { + // `connect` is how a tonic client is built (`BillingClient::connect(addr)`) + if matches!(ctor, "new" | "default" | "from" | "with_capacity" | "create" | "connect") { let name = head.rsplit("::").next()?.trim(); return (!name.is_empty()).then(|| name.to_string()); } @@ -845,8 +846,24 @@ fn constructed_type(value: &str) -> Option { .next() .is_some_and(|c| c.is_ascii_uppercase()) && head.chars().all(|c| c.is_alphanumeric() || c == '_' || c == ':'); - if is_type_name && (v.contains('{') || v.contains('(')) { - return Some(head.rsplit("::").next()?.to_string()); + // `Client::lookup(..)` is a call through the type, not a literal of it + let last = head.rsplit("::").next()?; + let last_is_type = last.chars().next().is_some_and(|c| c.is_ascii_uppercase()); + if is_type_name && last_is_type && (v.contains('{') || v.contains('(')) { + return Some(last.to_string()); + } + // go's constructor convention: `NewClient(..)` / `billingv1.NewBillingClient(..)` + // returns the type it is named after + if v.contains('(') { + let callee = v.split('(').next()?.trim(); + let func = callee.rsplit('.').next()?; + if let Some(ty) = func.strip_prefix("New") { + let is_type = ty.chars().next().is_some_and(|c| c.is_ascii_uppercase()) + && ty.chars().all(|c| c.is_alphanumeric() || c == '_'); + if is_type { + return Some(ty.to_string()); + } + } } None } @@ -868,20 +885,36 @@ fn func_metrics(node: Node, own_name: &str, ctx: &Ctx) -> FuncMetrics { } fn count_params(node: Node, ctx: &Ctx) -> usize { + if ctx.lang == Language::Proto { + return usize::from(proto_rpc_types(node, ctx).0.is_some()); + } + // `self`/`this` receivers are parameters of the call, not the API + param_list(node, ctx).map_or(0, |n| { + n.named_children(&mut n.walk()) + .filter(|c| !c.kind().contains("self")) + .count() + }) +} + +// The node holding a definition's parameters. The `parameters` field names it +// exactly where the grammar has one; otherwise the first list in source order. +// A go method has three lists - receiver, parameters, results - and a search +// that met the results first read the return types as the parameters. +fn param_list<'a>(node: Node<'a>, ctx: &Ctx) -> Option> { let kinds = ctx.lang.param_list_kinds(); + if let Some(n) = node.child_by_field_name("parameters").filter(|n| kinds.contains(&n.kind())) { + return Some(n); + } let mut stack = vec![node]; while let Some(n) = stack.pop() { if kinds.contains(&n.kind()) { - // `self`/`this` receivers are parameters of the call, not the API - return n - .named_children(&mut n.walk()) - .filter(|c| !c.kind().contains("self")) - .count(); + return Some(n); } - let mut cursor = n.walk(); - stack.extend(n.children(&mut cursor)); + let mut c = n.walk(); + let children: Vec = n.children(&mut c).collect(); + stack.extend(children.into_iter().rev()); } - 0 + None } fn walk_metrics( @@ -1055,6 +1088,11 @@ fn func_name<'a>(node: Node<'a>, ctx: &Ctx) -> Option<(String, Node<'a>)> { if let Some(n) = node.child_by_field_name("name") { return Some((oneline(text(n, ctx.src)), n)); } + // `rpc Charge(...)` names itself through an `rpc_name` wrapper + if ctx.lang == Language::Proto { + let n = first_child_of_kind(node, "rpc_name")?; + return Some((oneline(text(n, ctx.src)), n)); + } // odin declares by binding a name to a procedure literal if ctx.lang == Language::Odin { if let Some(n) = first_child_of_kind(node, "identifier") { @@ -1182,11 +1220,16 @@ fn type_def_name<'a>(node: Node<'a>, ctx: &Ctx) -> Option> { } // `Codec :: struct { ... }` - an unlabelled identifier child Language::Odin => first_child_of_kind(node, "identifier"), + // `message Invoice` -> `message_name`, and likewise for enum/service + Language::Proto => first_child_of_kind(node, &format!("{}_name", node.kind())), _ => None, } } fn func_return(node: Node, ctx: &Ctx) -> Option { + if ctx.lang == Language::Proto { + return proto_rpc_types(node, ctx).1; + } // odin wraps the signature in a `procedure` node if ctx.lang == Language::Odin { let proc = first_child_of_kind(node, "procedure")?; @@ -1208,6 +1251,20 @@ fn func_return(node: Node, ctx: &Ctx) -> Option { } } +// (request, response) of an rpc. Neither side is a labelled field, so they +// are told apart by order: `rpc Watch(stream Req) returns (stream Resp)` +fn proto_rpc_types(node: Node, ctx: &Ctx) -> (Option, Option) { + if node.kind() != "rpc" { + return (None, None); + } + let mut cursor = node.walk(); + let mut types = node + .children(&mut cursor) + .filter(|c| c.kind() == "message_or_enum_type") + .map(|c| oneline(text(c, ctx.src))); + (types.next(), types.next()) +} + fn extract_consts(node: Node, ctx: &mut Ctx) { match ctx.lang { Language::Rust => { @@ -1392,6 +1449,8 @@ fn extract_consts(node: Node, ctx: &mut Ctx) { } } } + // no `const_kinds`, so never reached + Language::Proto => {} } } @@ -1537,6 +1596,23 @@ fn receiver_type(callee: Node, qualifier: Option<&str>, ctx: &Ctx) -> Option bool { .trim_start() .strip_prefix("pub") .is_some_and(|rest| rest.starts_with(['(', ' ', '\t'])), + Language::Proto => stmt.split_whitespace().nth(1) == Some("public"), _ => false, } } @@ -1848,6 +1925,15 @@ fn parse_import(lang: Language, stmt: &str) -> Vec<(String, Vec)> { .collect(); vec![(module, names)] } + // `import "google/protobuf/timestamp.proto";` / `import public "a.proto";` + // binds no names - the package inside the file does + Language::Proto => { + let module = stmt.split('"').nth(1).unwrap_or("").to_string(); + if module.is_empty() { + return Vec::new(); + } + vec![(module, Vec::new())] + } Language::Odin => { // `import "core:fmt"` / `import os "core:os"` let module = stmt.split('"').nth(1).unwrap_or("").to_string(); @@ -1987,6 +2073,18 @@ fn loose_name(node: Node, ctx: &Ctx) -> Option<(Option, String)> { }; named(Some(q), name) } + // proto `google.protobuf.Timestamp` is a flat run of identifiers + "message_or_enum_type" => { + let mut cursor = node.walk(); + let parts: Vec = node.named_children(&mut cursor).collect(); + let (&last, head) = parts.split_last()?; + let qualifier = head + .iter() + .map(|n| oneline(text(*n, src))) + .collect::>() + .join("."); + Some(((!qualifier.is_empty()).then_some(qualifier), oneline(text(last, src)))) + } // js/ts `a.b` "member_expression" => named( node.child_by_field_name("object"), @@ -2925,6 +3023,64 @@ mod tests { assert_eq!(get.ret.as_deref(), Some("T")); } + #[test] + fn a_proto_schema_indexes_messages_enums_and_rpcs() { + let src = "syntax = \"proto3\";\n\ + package acme.billing.v1;\n\ + import \"google/protobuf/timestamp.proto\";\n\ + import public \"acme/common.proto\";\n\ + message Invoice {\n\ + \x20 google.protobuf.Timestamp created = 1;\n\ + \x20 message LineItem { string sku = 1; }\n\ + }\n\ + enum Status { STATUS_UNSPECIFIED = 0; PAID = 1; }\n\ + service Billing {\n\ + \x20 // creates one\n\ + \x20 rpc Create(CreateRequest) returns (Invoice);\n\ + \x20 rpc Watch(stream WatchRequest) returns (stream Invoice);\n\ + }\n"; + let ex = extract(Language::Proto, src).unwrap(); + let types: Vec<(&str, &str)> = + ex.types.iter().map(|t| (t.name.as_str(), t.kind.as_str())).collect(); + assert_eq!( + types, + vec![ + ("Invoice", "struct"), + ("LineItem", "struct"), + ("Status", "enum"), + ("Billing", "interface"), + ] + ); + assert_eq!(ex.modules, vec!["acme.billing.v1".to_string()]); + let consts: Vec<(&str, Option<&str>)> = + ex.consts.iter().map(|c| (c.name.as_str(), c.ty.as_deref())).collect(); + assert_eq!(consts, vec![("STATUS_UNSPECIFIED", Some("Status")), ("PAID", Some("Status"))]); + + // an rpc is a method of its service, taking its request and returning + // its response - `stream` is a modifier, not part of either type + let create = ex.funcs.iter().find(|f| f.name == "Create").unwrap(); + assert_eq!(create.owner.as_deref(), Some("Billing")); + assert_eq!(create.param_types, vec!["CreateRequest".to_string()]); + assert_eq!(create.ret.as_deref(), Some("Invoice")); + assert_eq!(create.comment.as_deref(), Some("creates one")); + let watch = ex.funcs.iter().find(|f| f.name == "Watch").unwrap(); + assert_eq!(watch.param_types, vec!["WatchRequest".to_string()]); + assert_eq!(watch.ret.as_deref(), Some("Invoice")); + + let imports: Vec<(&str, bool)> = + ex.imports.iter().map(|i| (i.module.as_str(), i.reexport)).collect(); + assert_eq!( + imports, + vec![("google/protobuf/timestamp.proto", false), ("acme/common.proto", true)] + ); + // a qualified field type is a use of that type + assert!( + ex.uses.iter().any(|u| u.name == "Timestamp" + && u.qualifier.as_deref() == Some("google.protobuf")), + "{:?}", ex.uses.iter().map(|u| &u.name).collect::>() + ); + } + #[test] fn an_unqualified_call_means_a_sibling_method_only_where_the_language_says_so() { // C++, C# and Zig look in the enclosing type before the file diff --git a/src/html.rs b/src/html.rs index d6069d8..71845e2 100644 --- a/src/html.rs +++ b/src/html.rs @@ -2,7 +2,7 @@ //! //! The generated page embeds the report JSON verbatim, renders it with //! Tailwind (CDN), and carries an HTMX-powered "live query" panel that talks -//! to a running `ccc serve` instance via its `/fragment/*` endpoints - so the +//! to a running `ccc run` instance via its `/fragment/*` endpoints - so the //! same file both *views* the report and *queries* the live map. use anyhow::{Context, Result}; @@ -96,7 +96,7 @@ const TEMPLATE: &str = r####" connecting… -

Queries a running ccc serve for the current tree - start it in the project root. The report above stays as generated.

+

Queries a running ccc run for the current tree - start it in the project root. The report above stays as generated.

@@ -133,7 +133,7 @@ const TEMPLATE: &str = r####"
-
generated by github.com/colwill/ccc changes --html · report embedded below · live panel needs ccc serve
+
generated by github.com/colwill/ccc changes --html · report embedded below · live panel needs ccc run
@@ -270,7 +270,7 @@ document.body.addEventListener('htmx:sendError', e => { const out = sel ? document.querySelector(sel) : e.detail.elt; if (out) out.innerHTML = out.id === 'health-chip' ? '● offline' - : '

server unreachable - is ccc serve running?

'; + : '

server unreachable - is ccc run running?

'; }); @@ -337,9 +337,9 @@ const INSIGHTS_TEMPLATE: &str = r####"
- served by ccc serve --html · data at /insights.json + served by ccc run · data at /insights.json
diff --git a/src/insights.rs b/src/insights.rs index 56b8533..b3f10bc 100644 --- a/src/insights.rs +++ b/src/insights.rs @@ -7,6 +7,7 @@ //! line and measurement it came from so a reader can check it. The UI is //! labelled accordingly; do not present these as proofs. +use crate::contracts::ContractIndex; use crate::coverage::{self, CoverageIndex, TestSite}; use crate::extract::TOP_LEVEL; use crate::languages::Language; @@ -101,7 +102,7 @@ impl<'a> Graph<'a> { // the callee is imported from it - but at function granularity rather than // file granularity. A call with no evidence, or with evidence for more than // one target, produces no edge: an absent edge is better than a wrong one. -fn build_graph<'a>(caches: &'a [FileCache]) -> Graph<'a> { +fn build_graph<'a>(caches: &'a [FileCache], contracts: &ContractIndex) -> Graph<'a> { let mut nodes = Vec::new(); // (file, name) -> every node with that name, in definition order. A name // is not unique within a file: overloads share one, and so does an @@ -265,6 +266,40 @@ fn build_graph<'a>(caches: &'a [FileCache]) -> Graph<'a> { } } } + + // rpc edges, the only ones allowed to cross languages: a call through a + // generated stub reaches the rpc in the schema, and the schema reaches + // each handler. With the schema outside the project the call reaches the + // handlers directly. + let pos_of: BTreeMap = + g.nodes.iter().copied().enumerate().map(|(p, id)| (id, p)).collect(); + let node = |(f, k): (usize, usize)| pos_of.get(&NodeId(f, k)).copied(); + for link in &contracts.callers { + let call = &caches[link.file].calls[link.call]; + let Some(from) = owner_of(link.file, call.caller.as_str(), call.line) else { + continue; + }; + let targets: Vec<(usize, usize)> = match contracts.contracts[link.contract].def { + Some(def) => vec![def], + None => contracts.handlers_of(link.contract).map(|h| h.def).collect(), + }; + for to in targets.into_iter().filter_map(node) { + if from != to && g.out[from].insert(to) { + g.into[to].insert(from); + g.call_sites[to] += 1; + } + } + } + for h in &contracts.handlers { + let (Some(from), Some(to)) = (contracts.contracts[h.contract].def.and_then(node), node(h.def)) + else { + continue; + }; + if g.out[from].insert(to) { + g.into[to].insert(from); + g.call_sites[to] += 1; + } + } g } @@ -737,7 +772,9 @@ fn lints(g: &Graph) -> (Vec, bool) { struct ServiceCtx { source: String, map: BTreeMap>, - deps: BTreeMap>, + // declared relationships from `map.json`: service to service, or service + // to a peer under `externals` + relatives: BTreeMap>, // per cache index, the services that own that file of_file: Vec>, // the grouping degenerated to one unit per file, so "service" means @@ -759,7 +796,7 @@ impl ServiceCtx { } } -fn service_ctx(g: &Graph, root: &Path) -> ServiceCtx { +fn service_ctx(g: &Graph, root: &Path, contracts: &ContractIndex) -> ServiceCtx { let cfg = ChangesConfig::load(root).unwrap_or_default(); let paths: Vec = g.caches.iter().map(|c| changes::path_str(&c.rel_path)).collect(); let (mut map, mut source) = if cfg.services.is_empty() { @@ -813,9 +850,9 @@ fn service_ctx(g: &Graph, root: &Path) -> ServiceCtx { // Peer repositories, and the boundary crossings that reach them. Resolving // a peer may parse a whole other checkout, so this rides the same memoised // analysis as everything else rather than running per request. - let externals = crate::externals::resolve_all(root, &cfg.externals); + let externals = crate::externals::resolve_all(root, &cfg.externals, &contracts.schemas); let crossings = match changes::build_matchers(&map) { - Ok(matchers) => changes::detect_crossings(&g.caches, &matchers, &externals), + Ok(matchers) => changes::detect_crossings(&g.caches, &matchers, &externals, contracts), Err(_) => Vec::new(), }; @@ -823,7 +860,7 @@ fn service_ctx(g: &Graph, root: &Path) -> ServiceCtx { per_file: source.starts_with("one unit per file"), source: source.to_string(), map, - deps: cfg.deps, + relatives: cfg.relatives, of_file, externals, crossings, @@ -902,7 +939,7 @@ fn services(g: &Graph, ctx: &ServiceCtx) -> Value { // the same list - flagged `declared` so a reader can tell them from the // ones that were detected, exactly as `changes` reports them. let mut declared: BTreeSet<(String, String)> = BTreeSet::new(); - for (from, tos) in &ctx.deps { + for (from, tos) in &ctx.relatives { for to in tos { declared.insert((from.clone(), to.clone())); edges.entry((from.clone(), to.clone())).or_default(); @@ -943,7 +980,7 @@ fn services(g: &Graph, ctx: &ServiceCtx) -> Value { json!({ "source": ctx.source, - "declared_deps": ctx.deps, + "declared_relatives": ctx.relatives, "externals": ctx.externals.iter().map(|e| e.json()).collect::>(), "external_names": external_names.iter().collect::>(), "crossings": ctx.crossings.iter().map(|c| json!({ @@ -1702,15 +1739,16 @@ pub fn insights( base: Option<&str>, ) -> Value { let started = Instant::now(); - let g = build_graph(caches); + let contracts = ContractIndex::for_root(root, caches); + let g = build_graph(caches, &contracts); // The coverage relation, built from the same evidence rules as the graph // and shared by every section that reports what a test reaches. let project_ids: BTreeSet = changes::manifest_identities(root) .into_iter() .map(|(id, _)| id) .collect(); - let cov = coverage::build(caches, &project_ids); - let ctx = service_ctx(&g, root); + let cov = coverage::build(caches, &project_ids, &contracts); + let ctx = service_ctx(&g, root, &contracts); let n = g.nodes.len(); // The change set is computed once, here, and every consumer refers to this @@ -1776,8 +1814,8 @@ pub fn insights( // trees are expected to leave their own boundary. Without them every real // service gets one - but not when the grouping degenerated to one unit per // file, where it would just redraw the same tree once per module. - let flame_keys: Vec = if !ctx.deps.is_empty() { - ctx.deps.keys().cloned().collect() + let flame_keys: Vec = if !ctx.relatives.is_empty() { + ctx.relatives.keys().cloned().collect() } else if ctx.per_file { Vec::new() } else { @@ -1804,7 +1842,7 @@ pub fn insights( let (t, cut) = flame(&g, &ctx, &svc_roots, &mut b); groups.push(json!({ "service": key, - "declares": ctx.deps.get(key), + "declares": ctx.relatives.get(key), "roots": t, "truncated": cut, })); @@ -1925,6 +1963,44 @@ mod tests { use super::*; use crate::scan; + fn graph_of(caches: &[FileCache]) -> Graph<'_> { + build_graph(caches, &ContractIndex::build(caches, &[])) + } + + // the call graph follows an rpc from a typescript client, through the + // schema, into the go handler + #[test] + fn an_rpc_is_a_path_through_the_schema_between_languages() { + let (_dir, caches) = map( + "rpc-graph", + &[ + ("proto/billing.proto", "syntax = \"proto3\";\npackage acme.billing.v1;\nmessage CreateInvoiceRequest { string customer = 1; }\nmessage Invoice { string id = 1; }\nservice Billing { rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice); }\n"), + ("svc/server.go", "package svc\nimport billingv1 \"github.com/acme/gen/acme/billing/v1\"\ntype server struct{}\nfunc (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\treturn nil, nil\n}\n"), + ( + "web/charge.ts", + "import { BillingClient } from \"./gen/billing_grpc_pb\";\n\ + export function charge() {\n\ + \x20 const client = new BillingClient(\"addr\");\n\ + \x20 client.createInvoice({});\n\ + }\n", + ), + ], + ); + let g = graph_of(&caches); + let at = |file: &str, name: &str| { + (0..g.nodes.len()) + .find(|&i| g.file(i) == file && g.name(i) == name) + .unwrap_or_else(|| panic!("no node {name} in {file}")) + }; + let (client, schema, handler) = ( + at("web/charge.ts", "charge"), + at("proto/billing.proto", "CreateInvoice"), + at("svc/server.go", "CreateInvoice"), + ); + assert_eq!(g.out[client], BTreeSet::from([schema])); + assert_eq!(g.out[schema], BTreeSet::from([handler])); + } + // `tag` must be unique per test: these run in parallel in one process fn map(tag: &str, files: &[(&str, &str)]) -> (tempdir::Dir, Vec) { let dir = tempdir::Dir::new(tag); @@ -2027,7 +2103,7 @@ mod tests { fn main() { let _ = charge(1); }\n", ), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let pos = |name: &str| (0..g.nodes.len()).find(|&i| g.name(i) == name).unwrap(); // same-file edge from `refs` assert!(g.out[pos("charge")].contains(&pos("helper"))); @@ -2046,7 +2122,7 @@ mod tests { ("src/b.rs", "pub fn run() -> u8 { 2 }\n"), ("src/c.rs", "fn go() -> u8 { run() }\n"), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let go = (0..g.nodes.len()).find(|&i| g.name(i) == "go").unwrap(); assert!( g.out[go].is_empty(), @@ -2071,7 +2147,7 @@ mod tests { "from mypkg.cli import main\n\nraise SystemExit(main())\n", ), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let pos = |name: &str| (0..g.nodes.len()).find(|&i| g.name(i) == name).unwrap(); let (top, main) = (pos(TOP_LEVEL), pos("main")); assert!(g.is_module(top)); @@ -2096,7 +2172,7 @@ mod tests { ("mypkg/__init__.py", "from .cli import run\n\n__all__ = [\"run\"]\n"), ("mypkg/app.py", "from mypkg import run\n\n\ndef go():\n return run()\n"), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let pos = |name: &str| (0..g.nodes.len()).find(|&i| g.name(i) == name).unwrap(); assert!(g.out[pos("go")].contains(&pos("run"))); drop(dir); @@ -2113,7 +2189,7 @@ mod tests { "from mypkg.cli import run\n\n\ndef go():\n return run()\n", ), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let cli = caches.iter().position(|c| c.rel_path.ends_with("cli.py")).unwrap(); let go = (0..g.nodes.len()).find(|&i| g.name(i) == "go").unwrap(); let called: Vec = g.out[go].iter().map(|&i| g.file(i)).collect(); @@ -2174,7 +2250,7 @@ mod tests { for pair in PAIRS { let (dir, caches) = map(pair.tag, pair.files); - let g = build_graph(&caches); + let g = graph_of(&caches); let pos = |name: &str| { (0..g.nodes.len()) .find(|&i| g.name(i) == name) @@ -2203,7 +2279,7 @@ mod tests { \x20 private int Settle(int a) { return a; }\n\ }\n", )]); - let g = build_graph(&caches); + let g = graph_of(&caches); let at = |line: usize| { (0..g.nodes.len()) .find(|&i| g.func(i).line == line) @@ -2236,7 +2312,7 @@ mod tests { ]; for (tag, files) in cases { let (dir, caches) = map(tag, files); - let g = build_graph(&caches); + let g = graph_of(&caches); let pos = |name: &str| { (0..g.nodes.len()) .find(|&i| g.name(i) == name) @@ -2259,10 +2335,10 @@ mod tests { fn top() -> u8 { mid() }\n\ fn spin(n: u8) -> u8 { spin(n) }\n", )]); - let g = build_graph(&caches); + let g = graph_of(&caches); let mut budget = FLAME_NODES; let roots: Vec = (0..g.nodes.len()).filter(|&i| g.is_root(i)).collect(); - let ctx = service_ctx(&g, dir.path()); + let ctx = service_ctx(&g, dir.path(), &ContractIndex::default()); let (tree, _) = flame(&g, &ctx, &roots, &mut budget); let top = tree.iter().find(|n| n["name"] == "top").unwrap(); // top -> mid -> leaf, so the root covers three frames @@ -2296,7 +2372,7 @@ mod tests { "void t() { char* q = (char*)malloc(4); }\n", ), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let (found, _) = lints(&g); let rules: Vec<&str> = found.iter().map(|f| f["rule"].as_str().unwrap()).collect(); assert!(rules.contains(&"leak-risk"), "{rules:?}"); @@ -2365,7 +2441,7 @@ mod tests { \x20 return 1\n", ), ]); - let g = build_graph(&caches); + let g = graph_of(&caches); let (found, _) = lints(&g); let leaks: Vec<&str> = found .iter() @@ -2392,7 +2468,7 @@ mod tests { ( ".ccc/map.json", r#"{"services":{"auth":["auth/**"],"billing":["billing/**"],"gateway":["gateway/**"]}, - "deps":{"gateway":["auth"]}}"#, + "relatives":{"gateway":["auth"]}}"#, ), ("auth/lib.rs", "pub fn verify(t: &str) -> bool { !t.is_empty() }\n"), ("billing/charge.rs", "pub fn charge(c: u64) -> u64 { c }\n"), @@ -2405,8 +2481,8 @@ mod tests { ("tools/codegen.rs", "pub fn helper() -> u64 { 7 }\n"), ], ); - let g = build_graph(&caches); - let s = services(&g, &service_ctx(&g, dir.path())); + let g = graph_of(&caches); + let s = services(&g, &service_ctx(&g, dir.path(), &ContractIndex::default())); assert_eq!(s["source"], ".ccc/map.json"); let names: Vec<&str> = s["services"] @@ -2446,14 +2522,14 @@ mod tests { // one flame graph per service that declares deps, with the frames a call // reached by leaving its caller's service marked #[test] - fn flame_groups_follow_declared_deps_and_mark_crossings() { + fn flame_groups_follow_declared_relatives_and_mark_crossings() { let (dir, caches) = map( "flamedeps", &[ ( ".ccc/map.json", r#"{"services":{"gateway":["gateway/**"],"billing":["billing/**"],"store":["store/**"]}, - "deps":{"gateway":["billing"]}}"#, + "relatives":{"gateway":["billing"]}}"#, ), ("store/db.rs", "pub fn fetch(id: u64) -> u64 { id } "), @@ -2529,8 +2605,8 @@ pub fn helper() -> u64 { 1 } ), ], ); - let g = build_graph(&caches); - let s = services(&g, &service_ctx(&g, dir.path())); + let g = graph_of(&caches); + let s = services(&g, &service_ctx(&g, dir.path(), &ContractIndex::default())); let edge = s["edges"] .as_array() .unwrap() @@ -2694,14 +2770,14 @@ pub fn helper() -> u64 { 1 } // Declaring a dependency in map.json must never stand in for analysing it #[test] - fn declared_deps_are_still_resolved_not_skipped() { + fn declared_relatives_are_still_resolved_not_skipped() { let (dir, caches) = map( "declared", &[ ( ".ccc/map.json", r#"{"services":{"gateway":["gateway/**"],"auth":["auth/**"],"queue":["queue/**"]}, - "deps":{"gateway":["auth","queue"]}}"#, + "relatives":{"gateway":["auth","queue"]}}"#, ), ("auth/lib.rs", "pub fn verify(t: &str) -> bool { !t.is_empty() }\n"), // traversed, declared, and genuinely uncallable @@ -2712,8 +2788,8 @@ pub fn helper() -> u64 { 1 } ), ], ); - let g = build_graph(&caches); - let s = services(&g, &service_ctx(&g, dir.path())); + let g = graph_of(&caches); + let s = services(&g, &service_ctx(&g, dir.path(), &ContractIndex::default())); let edge = |to: &str| { s["edges"] .as_array() diff --git a/src/languages.rs b/src/languages.rs index 5732d2f..75fd65e 100644 --- a/src/languages.rs +++ b/src/languages.rs @@ -16,6 +16,7 @@ pub enum Language { CSharp, Zig, Odin, + Proto, } impl Language { @@ -31,6 +32,7 @@ impl Language { Language::CSharp, Language::Zig, Language::Odin, + Language::Proto, ]; pub fn from_path(path: &Path) -> Option { @@ -48,6 +50,7 @@ impl Language { "cs" | "csx" => Language::CSharp, "zig" => Language::Zig, "odin" => Language::Odin, + "proto" => Language::Proto, _ => return None, }) } @@ -65,6 +68,7 @@ impl Language { Language::CSharp => "csharp", Language::Zig => "zig", Language::Odin => "odin", + Language::Proto => "proto", } } @@ -82,6 +86,7 @@ impl Language { Language::CSharp => tree_sitter_c_sharp::LANGUAGE.into(), Language::Zig => tree_sitter_zig::LANGUAGE.into(), Language::Odin => tree_sitter_odin::LANGUAGE.into(), + Language::Proto => tree_sitter_proto::LANGUAGE.into(), } } @@ -108,6 +113,8 @@ impl Language { ], Language::Zig => &["function_declaration"], Language::Odin => &["procedure_declaration"], + // an rpc is the one callable a schema declares + Language::Proto => &["rpc"], } } @@ -127,6 +134,8 @@ impl Language { // the same node; the type case is split back out in `extract` Language::Zig => &["variable_declaration"], Language::Odin => &["const_declaration"], + // enum values arrive through `variant_kinds`; proto has no other constants + Language::Proto => &[], } } @@ -139,6 +148,7 @@ impl Language { Language::Go => &["call_expression"], Language::Cpp | Language::C | Language::Zig | Language::Odin => &["call_expression"], Language::CSharp => &["invocation_expression", "object_creation_expression"], + Language::Proto => &[], } } @@ -156,6 +166,8 @@ impl Language { Language::CSharp => &["member_access_expression"], Language::Zig => &["field_expression"], Language::Odin => &["member_expression"], + // `google.protobuf.Timestamp` in a field or rpc signature + Language::Proto => &["message_or_enum_type"], } } @@ -178,6 +190,7 @@ impl Language { | Language::Odin => &[], // told apart from a struct field by its parent, in `extract` Language::Zig => &["container_field"], + Language::Proto => &["enum_field"], } } @@ -194,6 +207,7 @@ impl Language { Language::CSharp => &["using_directive"], Language::Zig => &["variable_declaration"], Language::Odin => &["import_declaration"], + Language::Proto => &["import"], } } @@ -204,7 +218,7 @@ impl Language { Language::Python | Language::JavaScript | Language::TypeScript | Language::Tsx | Language::Go => &["comment"], Language::Cpp | Language::C | Language::CSharp | Language::Zig - | Language::Odin => &["comment"], + | Language::Odin | Language::Proto => &["comment"], } } @@ -219,6 +233,7 @@ impl Language { | Language::CSharp | Language::Zig | Language::Odin + | Language::Proto | Language::TypeScript | Language::Tsx ) @@ -235,6 +250,8 @@ impl Language { Language::CSharp => "csharp", Language::Zig => "zig", Language::Odin => "odin", + // generated stubs are what other languages call, never the schema itself + Language::Proto => "proto", } } @@ -242,7 +259,7 @@ impl Language { pub fn package_scoped(self) -> bool { matches!( self, - Language::Go | Language::CSharp | Language::C | Language::Cpp + Language::Go | Language::CSharp | Language::C | Language::Cpp | Language::Proto ) } @@ -294,6 +311,11 @@ impl Language { ("union_declaration", "union"), ("bit_field_declaration", "struct"), ], + Language::Proto => &[ + ("message", "struct"), + ("enum", "enum"), + ("service", "interface"), + ], Language::TypeScript | Language::Tsx => &[ ("class_declaration", "class"), ("abstract_class_declaration", "class"), @@ -316,6 +338,7 @@ impl Language { Language::Rust => &["mod_item"], Language::CSharp => &["namespace_declaration", "file_scoped_namespace_declaration"], Language::Odin => &["package_declaration"], + Language::Proto => &["package"], Language::TypeScript | Language::Tsx => &["internal_module", "module"], _ => &[], } @@ -352,6 +375,7 @@ impl Language { ], Language::Zig => &["for_statement", "while_statement"], Language::Odin => &["for_statement"], + Language::Proto => &[], } } @@ -389,6 +413,7 @@ impl Language { ], Language::Zig => &["if_statement", "if_expression", "switch_case"], Language::Odin => &["if_statement", "switch_case"], + Language::Proto => &[], } } @@ -401,6 +426,8 @@ impl Language { &["parameter_list"] } Language::Zig | Language::Odin => &["parameters"], + // an rpc takes its one request type bare; `extract` reads it off the rpc + Language::Proto => &[], } } @@ -481,6 +508,7 @@ impl Language { ("setInterval", "clearInterval"), ("createObjectURL", "revokeObjectURL"), ], + Language::Proto => &[], } } @@ -498,6 +526,7 @@ impl Language { Language::JavaScript | Language::TypeScript | Language::Tsx => { "monomorphic call site (JIT inlines these)" } + Language::Proto => "n/a - a schema has no bodies to inline", } } @@ -549,6 +578,8 @@ impl Language { Language::Zig => Some("type"), Language::Odin => None, Language::JavaScript => None, + // the response type sits after `returns` with no field of its own + Language::Proto => None, } } } @@ -600,7 +631,7 @@ mod tests { #[test] fn every_language_is_reachable_by_extension() { let found: BTreeSet<&str> = ["a.rs", "a.py", "a.js", "a.ts", "a.tsx", "a.go", "a.cpp", - "a.c", "a.cs", "a.zig", "a.odin"] + "a.c", "a.cs", "a.zig", "a.odin", "a.proto"] .iter() .filter_map(|f| Language::from_path(Path::new(f))) .map(Language::as_str) diff --git a/src/lib.rs b/src/lib.rs index 3f653db..80bf234 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -16,9 +16,13 @@ pub mod sast; pub mod scan; pub mod serve; pub mod changes; +pub mod contracts; pub mod telemetry; pub mod tokenize; +// `check` is deprecated but still re-exported: deprecating it is what warns +// callers, and pulling the re-export would break them instead +#[allow(deprecated)] pub use scan::{check, scan, Change, ChangeKind, CheckReport, ScanReport}; pub use serve::{serve, ServeOptions}; pub use changes::{init_config, changes, ChangesOptions, ChangesReport}; diff --git a/src/main.rs b/src/main.rs index 0ed146c..99ef8c8 100644 --- a/src/main.rs +++ b/src/main.rs @@ -7,7 +7,7 @@ use std::process::ExitCode; #[derive(Parser)] #[command( name = "ccc", - about = "Scan a project and generate a ContextCodeCache (.ccc) directory", + about = "Map a project for agents and CI: serve it, diff it, or write it out", version )] struct Cli { @@ -15,6 +15,8 @@ struct Cli { command: Command, } +const CACHE_DIR: &str = ".ccc"; + #[derive(Clone, Copy, Debug, ValueEnum)] enum OutputFormat { // Human-readable summary (default). @@ -25,19 +27,37 @@ enum OutputFormat { #[derive(Subcommand)] enum Command { - // scan a project and (re)generate the `.ccc` directory + // parse a project and report what the map holds. Writes nothing without + // `--dir`: the map is what every other command reads, and it is built in + // memory whether or not a copy is left on disk Scan { #[arg(default_value = ".")] path: PathBuf, - #[arg(long)] + // Also write the markdown cache. Bare `--dir` writes `/.ccc`; + // `--dir=DIR` writes there instead, relative to `PATH`. The value needs + // `=` so a bare `--dir` cannot swallow the path argument after it. + #[arg(long, value_name = "DIR", num_args = 0..=1, require_equals = true, + default_missing_value = CACHE_DIR)] + dir: Option, + // pre-encode the written markdown into a token stream. Needs `--dir`: + // the stream is an encoding of those files, so they have to exist + #[arg(long, requires = "dir")] tokens: bool, #[arg(long, default_value = "o200k_base")] encoding: String, }, - // verify `.ccc` is up to date; exit non-zero if it would change + // Deprecated, and hidden from `--help` because of it: still callable so the + // pipelines built on it keep passing, but no longer offered. It verifies a + // written cache, and a written cache is now the exception - `ccc scan` + // builds the map in memory and only `--dir` puts a copy on disk for + // something outside `ccc` to read. + #[command(hide = true)] Check { #[arg(default_value = ".")] path: PathBuf, + // the cache to verify, relative to `PATH` (default `.ccc`) + #[arg(long, value_name = "DIR", default_value = CACHE_DIR)] + dir: PathBuf, // output format: `text` (default) or `json` (changed files as an array) #[arg(long, value_enum, default_value_t = OutputFormat::Text)] format: OutputFormat, @@ -45,11 +65,16 @@ enum Command { Tokenize { #[arg(default_value = ".")] path: PathBuf, + // the cache to encode, relative to `PATH` (default `.ccc`) + #[arg(long, value_name = "DIR", default_value = CACHE_DIR)] + dir: PathBuf, #[arg(long, default_value = "o200k_base")] encoding: String, }, - // publish what this project serves and calls across process boundaries - Export { + // Set `.ccc/` up: a starter service map, and the surface other repos + // consume. Both come out of one parse of the tree, and both are things a + // person edits afterwards rather than output to be regenerated blindly. + Init { #[arg(default_value = ".")] path: PathBuf, // name other repos will know this one by; defaults to the directory @@ -58,8 +83,8 @@ enum Command { // "owner/repo", recorded for display on the consuming side #[arg(long, value_name = "OWNER/REPO")] repo: Option, - // where to write it; defaults to `/.ccc/ccc-surface.json`. - // `-` writes to stdout. + // write the surface here instead of `/.ccc/surface.json`. + // `-` writes it to stdout #[arg(short, long, value_name = "FILE")] out: Option, }, @@ -84,8 +109,11 @@ enum Command { // exit non-zero when changed functions have no detected test reference #[arg(long)] fail_untested: bool, - // write a starter `.ccc/map.json` inferred from top-level directories - #[arg(long)] + // Moved to `ccc init`. Kept hidden only so the old spelling fails with + // somewhere to go: unlike the other retired flags this one *did* + // something, and quietly accepting it would hand back a change report + // where a scaffolded config was asked for + #[arg(long, hide = true)] init: bool, // include uncommitted edits and untracked files in the diff. CI wants // the committed view (the default); a local run usually wants this @@ -106,7 +134,7 @@ enum Command { #[arg(long)] fail_introduced: bool, // also write a single-file HTML view of the report (Tailwind + HTMX - // live-query panel against `ccc serve`), e.g. ccc-changes-rust.html + // live-query panel against `ccc run`), e.g. ccc-changes-rust.html #[arg(long, value_name = "FILE")] html: Option, // render --html from an existing changes JSON report instead of running @@ -150,9 +178,12 @@ enum Command { #[arg(long, value_enum, default_value_t = OutputFormat::Json)] format: OutputFormat, }, - // serve the code map over HTTP for AI agents: REST endpoints - // (/find /references /dependencies ...) + an MCP endpoint at /mcp - Serve { + // run the code map over HTTP for AI agents and people: REST endpoints + // (/find /references /dependencies ...), an MCP endpoint at /mcp, and the + // insights UI at /insights. `serve` was the original name; kept so existing + // scripts and editor integrations keep working + #[command(alias = "serve")] + Run { #[arg(default_value = ".")] path: PathBuf, // bind address (loopback by default; think twice before widening) @@ -167,9 +198,13 @@ enum Command { // disable file watching (rescan only via POST /refresh) #[arg(long)] no_watch: bool, - // also serve the human-facing insights UI at /insights + // The insights UI is served either way now. Accepted so the scripts and + // editor integrations that pass it keep working, but it turns nothing on #[arg(long)] html: bool, + // leave the human-facing UI off and serve the agent endpoints alone + #[arg(long, conflicts_with = "html")] + no_html: bool, #[arg(long)] deps: bool, }, @@ -211,8 +246,13 @@ enum Command { }, // install this `ccc` binary onto your PATH (Linux; defaults to ~/.local/bin) Install { - // directory to install into (default: ~/.local/bin) - #[arg(long)] + // Directory to install into, positionally: `ccc install ~/bin`. Every + // other command takes its path this way, and naming a destination is + // the only argument this one has. + #[arg(value_name = "DIR")] + path: Option, + // the same directory as a flag, for anyone already spelling it out + #[arg(long, value_name = "DIR", conflicts_with = "path")] dir: Option, // overwrite an existing `ccc` in the target directory #[arg(long)] @@ -236,34 +276,50 @@ fn run() -> Result { match cli.command { Command::Scan { path, + dir, tokens, encoding, } => { let root = canonical(&path); - let report = codecache::scan(&root)?; + // a relative `--dir` is relative to what was scanned, so the same + // flag means the same place whatever directory it was run from + let out = dir.map(|d| root.join(d)); + let report = codecache::scan(&root, out.as_deref())?; let t = report.totals; - println!( - "Wrote {} ({} files: {} funcs, {} consts, {} refs, {} notes)", - report.ccc_dir.display(), - report.files, - t.funcs, - t.consts, - t.refs, - t.notes + let mapped = format!( + "{} files: {} funcs, {} consts, {} refs, {} notes", + report.files, t.funcs, t.consts, t.refs, t.notes ); + match &report.out_dir { + Some(d) => println!("Mapped {mapped}\nWrote {}", d.display()), + // nothing was written, and saying so is the difference between + // "it worked" and "where did my files go" + None => println!("Mapped {mapped} (in memory; --dir writes the markdown)"), + } if tokens { - run_tokenize(&root, &encoding)?; + let d = out.as_deref().expect("clap enforces --tokens requires --dir"); + // exactly what was written above, not a re-read of it + run_tokenize(&report.rendered, d, &encoding)?; } Ok(ExitCode::SUCCESS) } - Command::Check { path, format } => { + Command::Check { path, dir, format } => { + // stderr, so a `--format json` pipeline reading stdout is unaffected + eprintln!( + "warning: `ccc check` is deprecated and will be removed. It verifies a written \ + cache, and `ccc scan` no longer writes one unless `--dir` asks. If you still \ + commit a cache, regenerate it with `ccc scan --dir` and let your VCS report the \ + diff; everything else reads the map `ccc` builds in memory." + ); // Canonicalize for the check itself (so results match `scan`, which // does the same), but keep the original `path` for building the // repo-relative cache paths reported in JSON. - let report = codecache::check(&canonical(&path))?; + let root = canonical(&path); + #[allow(deprecated)] + let report = codecache::check(&root, &root.join(&dir))?; match format { - OutputFormat::Text => print_check_text(&report), - OutputFormat::Json => print_check_json(&path, &report)?, + OutputFormat::Text => print_check_text(&report, &dir), + OutputFormat::Json => print_check_json(&path, &dir, &report)?, } if report.up_to_date { Ok(ExitCode::SUCCESS) @@ -271,20 +327,36 @@ fn run() -> Result { Ok(ExitCode::FAILURE) } } - Command::Export { + Command::Init { path, name, repo, out, } => { let root = canonical(&path); + // The service map first: it is the file a person is meant to edit, + // and an existing one is their work, not something to overwrite. + match codecache::changes::ChangesConfig::path(&root) { + Some(existing) => println!("kept {}", existing.display()), + None => { + let cfg = codecache::init_config(&root)?; + println!( + "Wrote {} - edit the service globs, then re-run `ccc changes`", + cfg.display() + ); + } + } + let files = codecache::scan::collect_files(&root)?; let caches = codecache::scan::build_caches(&root, &files); let label = name.unwrap_or_else(|| path_str(&root)); + // what the schemas tie this repo to is published with the rest + let contracts = codecache::contracts::ContractIndex::for_root(&root, &caches); let mut surface = codecache::Surface::from_caches( &label, &codecache::render::now_ts(), &caches, + &contracts, ); surface.repo = repo; let body = serde_json::to_string_pretty(&surface)?; @@ -304,20 +376,25 @@ fn run() -> Result { } std::fs::write(&target, format!("{body}\n")) .with_context(|| format!("writing {}", target.display()))?; - eprintln!( - "{}: {} provided, {} consumed -> {}", - surface.name, + println!( + "Wrote {} - {} provided, {} consumed", + target.display(), surface.provides.len(), surface.consumes.len(), - target.display() ); } } Ok(ExitCode::SUCCESS) } - Command::Tokenize { path, encoding } => { + Command::Tokenize { path, dir, encoding } => { let root = canonical(&path); - run_tokenize(&root, &encoding)?; + // the map, built here and now - not read back off a `.ccc` that may + // not exist and may not match the source if it does + let files = codecache::scan::collect_files(&root)?; + let caches = codecache::scan::build_caches(&root, &files); + let corpus = + codecache::scan::render_all(&root, &caches, &codecache::render::now_ts()); + run_tokenize(&corpus, &root.join(dir), &encoding)?; Ok(ExitCode::SUCCESS) } Command::Changes { @@ -339,9 +416,10 @@ fn run() -> Result { let _ = deps; let root = canonical(&path); if init { - let cfg = codecache::init_config(&root)?; - println!("Wrote {} - edit the service globs, then re-run `ccc changes`", cfg.display()); - return Ok(ExitCode::SUCCESS); + return Err(anyhow!( + "`ccc changes --init` has moved to `ccc init`, which writes the same \ + `.ccc/map.json` and the surface beside it" + )); } // render the HTML view from a saved report, no analysis @@ -469,22 +547,29 @@ fn run() -> Result { } Ok(ExitCode::SUCCESS) } - Command::Serve { + Command::Run { path, addr, port, watch_interval, no_watch, html, - deps + no_html, + deps, } => { let watch = if no_watch || watch_interval == 0 { None } else { Some(std::time::Duration::from_secs(watch_interval)) }; - let _ = deps; // discard - let opts = codecache::ServeOptions { addr, port, watch, html }; + // neither flag turns anything on; only `--no-html` turns one off + let _ = (deps, html); + let opts = codecache::ServeOptions { + addr, + port, + watch, + html: !no_html, + }; codecache::serve(&canonical(&path), &opts)?; Ok(ExitCode::SUCCESS) } @@ -545,7 +630,7 @@ fn run() -> Result { }; Ok(if gating > 0 { ExitCode::FAILURE } else { ExitCode::SUCCESS }) } - Command::Install { dir, force } => run_install(dir, force), + Command::Install { path, dir, force } => run_install(path.or(dir), force), } } @@ -795,11 +880,18 @@ fn print_changes_text(r: &ChangesReport) { } } -fn print_check_text(report: &CheckReport) { +fn print_check_text(report: &CheckReport, dir: &Path) { + let name = path_str(dir); if report.up_to_date { - println!(".ccc is up to date"); + println!("{name} is up to date"); } else { - eprintln!(".ccc is out of date; run `ccc scan`:"); + // name the flag that writes, because a bare `ccc scan` no longer does + let flag = if dir == Path::new(CACHE_DIR) { + "--dir".to_string() + } else { + format!("--dir={name}") + }; + eprintln!("{name} is out of date; run `ccc scan {flag}`:"); for c in &report.changes { eprintln!(" {:9} {}", format!("{}:", c.kind.as_str()), c.file); } @@ -809,8 +901,8 @@ fn print_check_text(report: &CheckReport) { // Emit `{ root, up_to_date, files[], changes[] }` as one JSON line. `files` is // the repo-relative paths of the changed cache entries — ready to hand to // another GitHub Action via `fromJSON(...)`. -fn print_check_json(root: &Path, report: &CheckReport) -> Result<()> { - let ccc_rel = rel_join(root, Path::new(".ccc")); +fn print_check_json(root: &Path, dir: &Path, report: &CheckReport) -> Result<()> { + let ccc_rel = rel_join(root, dir); let changes: Vec<_> = report .changes .iter() @@ -864,10 +956,16 @@ fn html_title(p: &Path) -> String { .to_string() } -fn run_tokenize(root: &Path, encoding: &str) -> Result<()> { +// `corpus` is the rendered map to encode and `ccc` where the stream lands - +// beside the markdown when there is any, on its own when there is not +fn run_tokenize( + corpus: &std::collections::BTreeMap, + ccc: &Path, + encoding: &str, +) -> Result<()> { let enc = Encoding::parse(encoding) .ok_or_else(|| anyhow!("unknown encoding '{encoding}' (use o200k_base or cl100k_base)"))?; - let report = codecache::tokenize(root, enc)?; + let report = codecache::tokenize(corpus, ccc, enc)?; println!( "Wrote {} ({} tokens from {} files, {} bytes, {} encoding; round-trip verified)", report.bin_path.display(), diff --git a/src/render.rs b/src/render.rs index 408a105..75e0097 100644 --- a/src/render.rs +++ b/src/render.rs @@ -1,5 +1,5 @@ //! rendering of `FileCache` entries and the `CCC.md` index to markdown per the -//! ContextCodeCache spec in PLAN.md +//! CodeCaChe spec in PLAN.md use crate::model::{Counts, FileCache}; use std::fmt::Write as _; @@ -117,7 +117,7 @@ pub fn render_index(root: &Path, caches: &[FileCache], ts: &str) -> String { // agent guide kept at the very top in metadata let _ = writeln!(out, "---"); - let _ = writeln!(out, "ContextCodeCache - agent guide"); + let _ = writeln!(out, "CodeCaChe - agent guide"); let _ = writeln!(out); let _ = writeln!(out, "what: a GENERATED map of this project's source. Each source file has a"); let _ = writeln!(out, " `-..md` entry listing the submodules it declares,"); @@ -132,18 +132,20 @@ pub fn render_index(root: &Path, caches: &[FileCache], ts: &str) -> String { let _ = writeln!(out, " as APPROXIMATE tiktoken (o200k) ids - for a downstream model that"); let _ = writeln!(out, " shares that vocabulary, NOT for Claude (different tokenizer; its API"); let _ = writeln!(out, " takes text, not token ids). Feed Claude the markdown above as text."); - let _ = writeln!(out, "query: `ccc serve` exposes this map over local HTTP - REST endpoints"); + let _ = writeln!(out, "query: `ccc run` exposes this map over local HTTP - REST endpoints"); let _ = writeln!(out, " (/find /references /dependencies /file /notes) plus an MCP"); let _ = writeln!(out, " endpoint at /mcp - so agents can query instead of reading files."); let _ = writeln!(out, "keep-fresh: whenever you change tracked source, regenerate with"); - let _ = writeln!(out, " `ccc scan` (add `--tokens` to refresh the token stream). CI runs"); - let _ = writeln!(out, " `ccc check`, which fails when `.ccc` is out of date."); + let _ = writeln!(out, " `ccc scan --dir` (add `--tokens` to refresh the token stream). A"); + let _ = writeln!(out, " bare `ccc scan` only builds the map in memory and writes nothing."); + let _ = writeln!(out, " A stale copy is a diff your VCS will show you; `ccc check`"); + let _ = writeln!(out, " still reports one but is deprecated."); let _ = writeln!(out, "do-not-edit: never hand-edit files under `.ccc` - they are overwritten on"); let _ = writeln!(out, " the next scan. To change the cache, change the source, then rescan."); let _ = writeln!(out, "---"); let _ = writeln!(out); - let _ = writeln!(out, "# ContextCodeCache ({}) UTC", ts); + let _ = writeln!(out, "# CodeCaChe ({}) UTC", ts); let _ = writeln!(out, "### project: {}", root_label); let _ = writeln!( out, @@ -156,7 +158,7 @@ pub fn render_index(root: &Path, caches: &[FileCache], ts: &str) -> String { totals.mods, totals.reexports ); - let _ = writeln!(out, "### regenerate: `ccc scan`"); + let _ = writeln!(out, "### regenerate: `ccc scan --dir`"); let _ = writeln!(out, "### files"); for c in caches { let n = c.counts(); diff --git a/src/scan.rs b/src/scan.rs index 9b18bf8..9d94dbe 100644 --- a/src/scan.rs +++ b/src/scan.rs @@ -32,7 +32,10 @@ const MAX_FILE_BYTES: u64 = 2_000_000; pub struct ScanReport { pub files: usize, pub totals: Counts, - pub ccc_dir: PathBuf, + // The markdown as rendered, kept only when it was written + pub rendered: BTreeMap, + // where the markdown was written + pub out_dir: Option, } pub struct CheckReport { @@ -159,7 +162,9 @@ fn build_one(root: &Path, path: &Path) -> Option { }) } -fn render_all(root: &Path, caches: &[FileCache], ts: &str) -> BTreeMap { +// the whole cache as `name -> markdown`, which is what gets written, compared +// against, or encoded - never re-derived from whatever is on disk +pub fn render_all(root: &Path, caches: &[FileCache], ts: &str) -> BTreeMap { let mut map = BTreeMap::new(); for c in caches { map.insert(c.cache_name.clone(), render::render_file(c, ts)); @@ -168,21 +173,30 @@ fn render_all(root: &Path, caches: &[FileCache], ts: &str) -> BTreeMap Result { +// Parse `root` into the map, and write the markdown only when `out` names +// somewhere to put it. Every other command builds this same map in memory and +// never reads the files, so writing them is a choice rather than a step. +pub fn scan(root: &Path, out: Option<&Path>) -> Result { let files = collect_files(root)?; let caches = build_caches(root, &files); - let ts = render::now_ts(); - let rendered = render_all(root, &caches, &ts); - let ccc = root.join(".ccc"); - fs::create_dir_all(&ccc).with_context(|| format!("creating {}", ccc.display()))?; - clear_generated(&ccc)?; - crate::tokenize::clear(&ccc)?; - for (name, content) in &rendered { - let path = ccc.join(name); - fs::write(&path, content).with_context(|| format!("writing {}", path.display()))?; - } + let (out_dir, rendered) = match out { + Some(dir) => { + let ts = render::now_ts(); + let rendered = render_all(root, &caches, &ts); + fs::create_dir_all(dir).with_context(|| format!("creating {}", dir.display()))?; + clear_generated(dir)?; + crate::tokenize::clear(dir)?; + for (name, content) in &rendered { + let path = dir.join(name); + fs::write(&path, content).with_context(|| format!("writing {}", path.display()))?; + } + (Some(dir.to_path_buf()), rendered) + } + // rendering is only ever done to be written, so without a destination + // it is not done at all + None => (None, BTreeMap::new()), + }; let mut totals = Counts::default(); for c in &caches { @@ -191,18 +205,27 @@ pub fn scan(root: &Path) -> Result { Ok(ScanReport { files: caches.len(), totals, - ccc_dir: ccc, + rendered, + out_dir, }) } -// verify .ccc outputs for CI -pub fn check(root: &Path) -> Result { +// Verify a written cache for CI. Deprecated alongside the `ccc check` command: +// it answers "does the copy on disk match the source", and since `scan` only +// writes a copy when `--dir` asks for one, most projects no longer have a copy +// to be stale. The map every other entry point reads is built from source each +// time and cannot go out of date. +#[deprecated( + since = "1.4.3", + note = "a written cache is now opt-in; regenerate with `scan(root, Some(dir))` and diff it, \ + or read the in-memory map that every other entry point builds" +)] +pub fn check(root: &Path, ccc: &Path) -> Result { let files = collect_files(root)?; let caches = build_caches(root, &files); let ts = render::now_ts(); let expected = render_all(root, &caches, &ts); - let ccc = root.join(".ccc"); let mut changes = Vec::new(); for (name, content) in &expected { @@ -224,7 +247,7 @@ pub fn check(root: &Path) -> Result { if ccc.is_dir() { // clean stale let mut existing = BTreeSet::new(); - for entry in fs::read_dir(&ccc)? { + for entry in fs::read_dir(ccc)? { let name = entry?.file_name().to_string_lossy().to_string(); if name.ends_with(".md") { existing.insert(name); @@ -257,3 +280,80 @@ fn clear_generated(ccc: &Path) -> Result<()> { } Ok(()) } + +#[cfg(test)] +mod tests { + use super::*; + + fn fixture(name: &str) -> PathBuf { + let dir = std::env::temp_dir().join(format!("ccc-scan-{name}-{}", std::process::id())); + let _ = fs::remove_dir_all(&dir); + fs::create_dir_all(dir.join("src")).expect("mkdir"); + fs::write(dir.join("src/lib.rs"), "pub fn charge(c: u64) -> u64 { c }\n").expect("write"); + dir + } + + fn md_names(dir: &Path) -> BTreeSet { + fs::read_dir(dir) + .map(|rd| { + rd.flatten() + .map(|e| e.file_name().to_string_lossy().to_string()) + .filter(|n| n.ends_with(".md")) + .collect() + }) + .unwrap_or_default() + } + + // The map is the product; the markdown is a copy of it somebody asked for. + // A scan that writes nothing has to leave the tree exactly as it found it - + // not an empty `.ccc`, not a directory at all. + #[test] + fn a_scan_without_a_destination_writes_nothing() { + let dir = fixture("memory"); + let report = scan(&dir, None).expect("scan"); + assert_eq!(report.files, 1); + assert_eq!(report.totals.funcs, 1); + assert!(report.out_dir.is_none()); + assert!(!dir.join(".ccc").exists(), "a bare scan created .ccc"); + let _ = fs::remove_dir_all(&dir); + } + + // `check` is deprecated, not gone - it still has to work for the pipelines + // that call it, so it is still covered + #[allow(deprecated)] + #[test] + fn a_destination_gets_the_markdown_and_check_reads_it_back() { + let dir = fixture("written"); + for out in [dir.join(".ccc"), dir.join("docs/map")] { + let report = scan(&dir, Some(&out)).expect("scan"); + assert_eq!(report.out_dir.as_deref(), Some(out.as_path())); + let names = md_names(&out); + assert!(names.contains("CCC.md"), "{names:?}"); + assert!(names.contains("src-lib.rs.md"), "{names:?}"); + // and the cache just written is by definition up to date + assert!(check(&dir, &out).expect("check").up_to_date); + } + // a destination that was never written reads as wholly missing rather + // than as up to date + let report = check(&dir, &dir.join("nowhere")).expect("check"); + assert!(!report.up_to_date); + assert!(report.changes.iter().all(|c| c.kind == ChangeKind::Missing)); + let _ = fs::remove_dir_all(&dir); + } + + // rewriting a destination clears what a previous scan left, so a deleted + // source file does not linger as a cache entry nothing maps to + #[test] + fn rewriting_a_destination_clears_the_last_one() { + let dir = fixture("stale"); + let out = dir.join(".ccc"); + fs::write(dir.join("src/gone.rs"), "pub fn gone() {}\n").expect("write"); + scan(&dir, Some(&out)).expect("scan"); + assert!(md_names(&out).contains("src-gone.rs.md")); + + fs::remove_file(dir.join("src/gone.rs")).expect("rm"); + scan(&dir, Some(&out)).expect("scan"); + assert!(!md_names(&out).contains("src-gone.rs.md")); + let _ = fs::remove_dir_all(&dir); + } +} diff --git a/src/serve.rs b/src/serve.rs index 362dfab..e600631 100644 --- a/src/serve.rs +++ b/src/serve.rs @@ -1,10 +1,11 @@ -//! `ccc serve` local REST/MCP endpoints for AI agents. +//! `ccc run` local REST/MCP endpoints for AI agents, plus the insights UI. //! //! On startup the whole project is parsed into an in-memory map (the same //! model `.ccc` is rendered from); every query answers from memory. A watcher //! thread polls a walk fingerprint (path + mtime + size) and swaps a freshly //! parsed map in whenever source changes - `/refresh` forces it immediately. +use crate::contracts::ContractIndex; use crate::model::{FileCache, Counts}; use crate::{audit, deps, insights, render, sast, scan}; use anyhow::Result; @@ -34,7 +35,7 @@ impl Default for ServeOptions { addr: "127.0.0.1".into(), port: 6767, watch: Some(std::time::Duration::from_secs(2)), - html: false, + html: true, } } } @@ -44,6 +45,8 @@ struct MapState { root_label: String, ts: String, caches: Vec, + // the rpcs the schemas declare and the code tied to them; rebuilt with `caches` + contracts: ContractIndex, externals: Vec, facade: Option, watch_secs: Option, @@ -80,11 +83,12 @@ impl MapState { root: root.to_path_buf(), root_label, ts: render::now_ts(), + contracts: ContractIndex::for_root(root, &caches), caches, externals: manifest_deps(root), facade: cargo_package_name(root), watch_secs: None, - html: false, + html: true, // `--port 0` picks one at runtime origin: { let d = ServeOptions::default(); @@ -150,6 +154,7 @@ impl MapState { let before = self.caches.len(); let files = scan::collect_files(&self.root)?; self.caches = scan::build_caches(&self.root, &files); + self.contracts = ContractIndex::for_root(&self.root, &self.caches); self.externals = manifest_deps(&self.root); self.facade = cargo_package_name(&self.root); self.ts = render::now_ts(); @@ -160,6 +165,7 @@ impl MapState { // swap in a fresh map (built outside lock by watcher) fn swap_in(&mut self, caches: Vec) { self.caches = caches; + self.contracts = ContractIndex::for_root(&self.root, &self.caches); self.externals = manifest_deps(&self.root); self.facade = cargo_package_name(&self.root); self.ts = render::now_ts(); @@ -1046,9 +1052,13 @@ fn q_references(map: &MapState, symbol: &str) -> Result { if !exported_as.is_empty() { out["exported_as"] = Value::Array(exported_as); } + let rpcs = rpc_references(map, symbol, qualifier, name); + if !rpcs.is_empty() { + out["rpcs"] = Value::Array(rpcs); + } // A miss is an answer, not an error: say what was covered and what to try, // so the lookup can be carried on rather than abandoned. - if out["counts"]["definitions"] == 0 && total_refs == 0 { + if out["counts"]["definitions"] == 0 && total_refs == 0 && out.get("rpcs").is_none() { out["miss"] = json!(true); out["suggestions"] = Value::Array(nearest_names(map, name, 8)); add_miss_evidence(map, &mut out, qualifier); @@ -1056,6 +1066,59 @@ fn q_references(map: &MapState, symbol: &str) -> Result { Ok(out) } +// The rpcs a lookup names, with everything tied to each one in any language: +// the schema that declares it, the handlers that implement it, and the call +// sites that reach it through a generated stub. Nothing else in `references` +// crosses a language, because no name does - `CreateInvoice` in Go is +// `createInvoice` in TypeScript and `create_invoice` in Rust. +fn rpc_references(map: &MapState, symbol: &str, qualifier: Option<&str>, name: &str) -> Vec { + // the wire form `acme.billing.v1.Billing/CreateInvoice` + let (qualifier, name) = match symbol.rsplit_once('/') { + Some((q, n)) if !n.is_empty() => (Some(q).filter(|q| !q.is_empty()), n), + _ => (qualifier, name), + }; + let idx = &map.contracts; + let lang = |fi: usize| map.caches[fi].language.as_str(); + idx.matching(name, qualifier) + .into_iter() + .map(|ci| { + let c = &idx.contracts[ci]; + let handlers: Vec = idx + .handlers_of(ci) + .map(|h| { + let f = &map.caches[h.def.0].funcs[h.def.1]; + json!({ + "file": map.path_of(&map.caches[h.def.0]), "line": f.line, + "function": f.name, "owner": f.owner, "language": lang(h.def.0), + "evidence": h.evidence.label(), + }) + }) + .collect(); + let callers: Vec = idx + .callers_of(ci) + .take(REFS_CAP) + .map(|l| { + let site = &map.caches[l.file].calls[l.call]; + json!({ + "file": map.path_of(&map.caches[l.file]), "line": site.line, + "name": site.name, "caller": site.caller, "language": lang(l.file), + "test_ctx": site.test_ctx, "evidence": l.evidence.label(), + }) + }) + .collect(); + json!({ + "key": c.key, + "schema": {"file": c.file, "line": c.line, "in_project": c.def.is_some()}, + "request": c.request, + "response": c.response, + "handlers": handlers, + "callers": callers, + "counts": {"handlers": idx.handlers_of(ci).count(), "callers": idx.callers_of(ci).count()}, + }) + }) + .collect() +} + // package name from a root Cargo.toml, if any - the name code uses to // qualify calls through the crate facade fn cargo_package_name(root: &Path) -> Option { @@ -1680,7 +1743,7 @@ fn mcp_tools() -> Value { ), tool( "references", - "CALL THIS BEFORE renaming a symbol, changing a signature, or deleting anything that looks unused - it answers what calls this, who imports it, and is this dead. Use instead of `grep -rn 'foo('`, which misses imports and type-only uses while inventing hits in comments. Definitions, call sites, qualified value usages (enum variants, consts: `Encoding::O200kBase`) and import bindings of an exact name. Type definitions and imports are covered, so a struct used only through its type, or a crate pulled in for a derive, is still found. Qualified names (`serde_json::to_string`, `client.charge`, `Encoding::parse`) narrow by file, owning type and import module, and definitions that merely share the bare name are listed separately rather than passed off as the symbol. Re-exports are followed, so a crate-facade path (`mycrate::thing`, from a `pub use` in lib.rs) resolves to the definition in the module it actually lives in, and any symbol a module root republishes is reported under `published as` - renaming one is a breaking change even when every call site is local. Each hit carries its enclosing caller and test context, so production callers are distinguishable from test ones at a glance. A miss is an answer, not an error - it names the kinds searched, the nearest indexed names, and whether the qualifier is a declared dependency.", + "CALL THIS BEFORE renaming a symbol, changing a signature, or deleting anything that looks unused - it answers what calls this, who imports it, and is this dead. Use instead of `grep -rn 'foo('`, which misses imports and type-only uses while inventing hits in comments. Definitions, call sites, qualified value usages (enum variants, consts: `Encoding::O200kBase`) and import bindings of an exact name. Type definitions and imports are covered, so a struct used only through its type, or a crate pulled in for a derive, is still found. Qualified names (`serde_json::to_string`, `client.charge`, `Encoding::parse`) narrow by file, owning type and import module, and definitions that merely share the bare name are listed separately rather than passed off as the symbol. Re-exports are followed, so a crate-facade path (`mycrate::thing`, from a `pub use` in lib.rs) resolves to the definition in the module it actually lives in, and any symbol a module root republishes is reported under `published as` - renaming one is a breaking change even when every call site is local. Each hit carries its enclosing caller and test context, so production callers are distinguishable from test ones at a glance. An rpc declared in a `.proto` schema (looked up by any spelling - `CreateInvoice`, `create_invoice`, `Billing.CreateInvoice`, `acme.billing.v1.Billing/CreateInvoice`) also lists its handlers and its callers through generated stubs in every language, each with the evidence that tied it. A miss is an answer, not an error - it names the kinds searched, the nearest indexed names, and whether the qualifier is a declared dependency.", json!({"symbol": {"type": "string", "description": "exact symbol name, optionally qualified (a::b or a.b)"}}), &["symbol"], ), @@ -1798,7 +1861,7 @@ fn mcp_tools() -> Value { ), tool( "services", - "ANSWERS how the parts of this system talk to each other, and which code carries each hop. The service map and the call edges between services, with the call sites behind them. Services come from `.ccc/map.json` when present, top-level directories otherwise. An edge is `declared` if the config lists it, `detected` if calls were resolved across it - both are reported, since a declared HTTP or queue link resolves no calls by design. Edges also cross repositories: `externals` in `.ccc/map.json` names peer repos (a local checkout, or a surface they published with `ccc export`), and `ccc:calls` / `ccc:serves` comments naming the same key join a call here to its handler there, whatever language that repo is written in.", + "ANSWERS how the parts of this system talk to each other, and which code carries each hop. The service map and the call edges between services, with the call sites behind them. Services come from `.ccc/map.json` when present, top-level directories otherwise. An edge is `declared` if the config lists it, `detected` if calls were resolved across it - both are reported, since a declared HTTP or queue link resolves no calls by design. Edges also cross repositories: `externals` in `.ccc/map.json` names peer repos (a local checkout, or a surface they published with `ccc init`), and `ccc:calls` / `ccc:serves` comments naming the same key join a call here to its handler there, whatever language that repo is written in.", json!({ "service": {"type": "string", "description": "drill into one service: its definition plus every edge touching it"}, "limit": {"type": "integer", "description": "edges per page (default 25, max 500)"}, @@ -1810,7 +1873,7 @@ fn mcp_tools() -> Value { // the one tool aimed at the person rather than the agent tool( "insights", - "CALL THIS WHEN THE USER ASKS TO SEE the analysis - show me the insights, open the dashboard, what does this codebase look like. Opens the human-facing insights UI (`/insights`) in their browser: the flame graph of the static call tree, hot paths, the service map, this branch's changes, test triggers and targets, lints and per-language totals - the same analysis pass the other tools read, laid out to be looked at rather than parsed. Returns the URL and the headline totals, so you can talk about the page while they read it. Needs the server started with `ccc serve --html`; without it the tool says so, and the data is at /insights.json either way.", + "CALL THIS WHEN THE USER ASKS TO SEE the analysis - show me the insights, open the dashboard, what does this codebase look like. Opens the human-facing insights UI (`/insights`) in their browser: the flame graph of the static call tree, hot paths, the service map, this branch's changes, test triggers and targets, lints and per-language totals - the same analysis pass the other tools read, laid out to be looked at rather than parsed. Returns the URL and the headline totals, so you can talk about the page while they read it. Served by default; only `ccc run --no-html` turns it off, and then the tool says so and the data is still at /insights.json.", json!({}), &[], ), @@ -1833,10 +1896,10 @@ fn mcp_initialize(params: &Value) -> Value { "capabilities": {"tools": {}, "resources": {}}, "serverInfo": { "name": "ccc", - "title": "ContextCodeCache", + "title": "CodeCaChe", "version": env!("CARGO_PKG_VERSION"), }, - "instructions": "Code map of this project (the .ccc ContextCodeCache), held in \ + "instructions": "Code map of this project (the ccc CodeCaChe), held in \ memory and refreshed automatically about three seconds after source changes.\n\n\ 1. SEARCHING - always start here. For any question about where something is \ defined, called, imported or changed in this project, call a ccc tool before \ @@ -2294,6 +2357,31 @@ fn md_references(v: &Value) -> String { } } md_section(&mut out, "references", &hits("references")); + for rpc in jarr(v, "rpcs") { + let schema = rpc.get("schema").cloned().unwrap_or_default(); + out.push_str(&format!( + "\n## rpc {}\ndeclared at {}:{}{}\n", + jstr(&rpc, "key"), + jstr(&schema, "file"), + jnum(&schema, "line"), + if jbool(&schema, "in_project") { "" } else { " (schema from map.json contracts)" }, + )); + let row = |r: &Value, what: &str| { + format!( + "{}:{} {} [{}, {}]{}\n", + jstr(r, "file"), + jnum(r, "line"), + jstr(r, what), + jstr(r, "language"), + jstr(r, "evidence"), + if jbool(r, "test_ctx") { " (test)" } else { "" }, + ) + }; + let handlers: String = jarr(&rpc, "handlers").iter().map(|h| row(h, "function")).collect(); + let callers: String = jarr(&rpc, "callers").iter().map(|c| row(c, "caller")).collect(); + md_section(&mut out, "handlers", &handlers); + md_section(&mut out, "callers", &callers); + } let name_only = hits("name_only_matches"); if !name_only.is_empty() { out.push_str(&format!( @@ -3193,7 +3281,7 @@ fn q_insights(map: &MapState, open: impl Fn(&str) -> Result<(), String>) -> Resu let url = format!("{}/insights", map.origin); if !map.html { return Err(format!( - "the insights UI is disabled; restart the server with `ccc serve --html` to \ + "the insights UI is disabled by `--no-html`; restart the server as `ccc run` to \ serve it at {url} (the analysis itself is at {}/insights.json, and the \ `changes`, `test_triggers`, `test_targets`, `lints`, `hot` and `services` \ tools read the same pass)", @@ -3371,7 +3459,7 @@ fn mcp_resources_list(state: &RwLock) -> Value { let mut resources = vec![json!({ "uri": "ccc://index", "name": "CCC.md", - "description": "ContextCodeCache index for the whole project", + "description": "CodeCaChe index for the whole project", "mimeType": "text/markdown", })]; for c in &map.caches { @@ -3694,7 +3782,7 @@ const ENDPOINTS: &[&str] = &[ "GET /health", "GET /prompts[?base=] (which claude/copilot request produced each change)", "GET /insights.json[?base=] (the whole analysis payload)", - "GET /insights (human UI over the same data; needs --html)", + "GET /insights (human UI over the same data; off with --no-html)", "POST /refresh", "POST /mcp (Model Context Protocol, JSON-RPC)", "GET /fragment/{find,references,dependencies,health} (HTML for HTMX)", @@ -3799,13 +3887,13 @@ fn route(state: &RwLock, method: &str, url: &str, body: &[u8]) -> Repl let map = state.read().expect("map lock poisoned"); ok((*map.analysis(get("base"))).clone()) } - // human-facing insights UI; off unless `ccc serve --html` + // human-facing insights UI; on unless `ccc run --no-html` ("GET", "/insights") => { let map = state.read().expect("map lock poisoned"); if !map.html { return bad( 404, - "insights UI is disabled; restart with `ccc serve --html` \ + "insights UI is disabled by `--no-html`; restart with `ccc run` \ (the data is at /insights.json either way)", ); } @@ -3913,7 +4001,7 @@ pub fn serve(root: &Path, opts: &ServeOptions) -> Result<()> { { let map = state.read().expect("map lock poisoned"); println!( - "ccc serve: {} files mapped from {}", + "ccc run: {} files mapped from {}", map.caches.len(), root.display() ); @@ -4046,6 +4134,70 @@ mod tests { // The shapes the fixture above has nothing to say about: imports, a name // shared by two owners, type definitions, and a manifest. + // The schema lives in a sibling protos checkout, named by `contracts`; the + // lookup is spelled the way the rust caller spells it. + #[test] + fn references_to_an_rpc_list_its_handlers_and_callers_in_every_language() { + let base = std::env::temp_dir().join(format!("ccc-serve-rpc-{}", std::process::id())); + let _ = fs::remove_dir_all(&base); + let (app, protos) = (base.join("app"), base.join("protos/acme/billing/v1")); + fs::create_dir_all(app.join(".ccc")).unwrap(); + fs::create_dir_all(app.join("svc")).unwrap(); + fs::create_dir_all(app.join("client/src")).unwrap(); + fs::create_dir_all(&protos).unwrap(); + fs::write( + protos.join("billing.proto"), + "syntax = \"proto3\";\npackage acme.billing.v1;\n\ + message CreateInvoiceRequest { string customer = 1; }\n\ + message Invoice { string id = 1; }\n\ + service Billing { rpc CreateInvoice(CreateInvoiceRequest) returns (Invoice); }\n", + ) + .unwrap(); + fs::write( + app.join(".ccc/map.json"), + "{\"contracts\": [\"../protos/**/*.proto\"]}\n", + ) + .unwrap(); + fs::write( + app.join("svc/server.go"), + "package svc\nimport billingv1 \"github.com/acme/gen/acme/billing/v1\"\n\ + type server struct{}\n\ + func (s *server) CreateInvoice(ctx context.Context, req *billingv1.CreateInvoiceRequest) (*billingv1.Invoice, error) {\n\ + \treturn nil, nil\n}\n", + ) + .unwrap(); + fs::write( + app.join("client/src/lib.rs"), + "use billing::v1::billing_client::BillingClient;\n\ + pub async fn charge() {\n\ + \x20 let mut client = BillingClient::connect(\"http://x\").await.unwrap();\n\ + \x20 client.create_invoice(req).await;\n\ + }\n", + ) + .unwrap(); + let map = MapState::build(&app).unwrap(); + let _ = fs::remove_dir_all(&base); + + for symbol in ["create_invoice", "Billing.CreateInvoice", "acme.billing.v1.Billing/CreateInvoice"] { + let v = q_references(&map, symbol).unwrap(); + let rpcs = v["rpcs"].as_array().unwrap_or_else(|| panic!("{symbol}: {v}")); + assert_eq!(rpcs.len(), 1, "{symbol}"); + let rpc = &rpcs[0]; + assert_eq!(rpc["key"], "acme.billing.v1.Billing/CreateInvoice"); + assert_eq!(rpc["schema"]["in_project"], false); + assert_eq!(rpc["handlers"][0]["file"], "svc/server.go"); + assert_eq!(rpc["handlers"][0]["evidence"], "request-type"); + assert_eq!(rpc["callers"][0]["file"], "client/src/lib.rs"); + assert_eq!(rpc["callers"][0]["caller"], "charge"); + assert_eq!(rpc["callers"][0]["language"], "rust"); + assert!(v.get("miss").is_none(), "{symbol}: an rpc hit is not a miss"); + } + let md = md_references(&q_references(&map, "create_invoice").unwrap()); + assert!(md.contains("## rpc acme.billing.v1.Billing/CreateInvoice"), "{md}"); + assert!(md.contains("client/src/lib.rs:4 charge [rust, stub-type]"), "{md}"); + assert!(q_references(&map, "Ledger.CreateInvoice").unwrap().get("rpcs").is_none()); + } + fn fixture_imports() -> MapState { static SEQ: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); let n = SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -5384,11 +5536,14 @@ mod tests { } #[test] - fn insights_ui_is_opt_in() { + fn the_insights_ui_is_served_unless_it_is_refused() { let state = RwLock::new(fixture()); - assert_eq!(route(&state, "GET", "/insights", b"").status, 404); + // off only when asked: the error names the flag that turned it off and + // the JSON way in, which stays open either way + state.write().unwrap().html = false; let off = route(&state, "GET", "/insights", b""); - assert!(json_of(&off)["error"].as_str().unwrap().contains("--html")); + assert_eq!(off.status, 404); + assert!(json_of(&off)["error"].as_str().unwrap().contains("--no-html")); assert_eq!(route(&state, "GET", "/insights.json", b"").status, 200); state.write().unwrap().html = true; @@ -5429,9 +5584,10 @@ mod tests { // disabled UI: the error names the flag and the JSON way in, and no // browser is opened for a page that would 404 - let err = q_insights(&map, |_| panic!("must not open a browser with --html off")) - .expect_err("the UI is off in the fixture"); - assert!(err.contains("ccc serve --html"), "{err}"); + map.html = false; + let err = q_insights(&map, |_| panic!("must not open a browser with --no-html")) + .expect_err("the UI was turned off above"); + assert!(err.contains("--no-html"), "{err}"); assert!(err.contains("http://127.0.0.1:7788/insights.json"), "{err}"); map.html = true; diff --git a/src/tokenize.rs b/src/tokenize.rs index 16c530e..038c624 100644 --- a/src/tokenize.rs +++ b/src/tokenize.rs @@ -2,12 +2,13 @@ //! pretrained tiktoken vocabulary. Downstream consumers load raw token IDs //! (`&[u32]`) directly from `tokens.bin` - no re-tokenization at load time. //! -//! Layout written into `.ccc/`: +//! Layout written into the cache directory (`.ccc/` unless `--dir` moved it): //! - `tokens.bin` - little-endian `u32` token IDs for every cache file, concatenated //! - `tokens.json` - index: encoding, layout, and per-file `(offset, len)` in tokens use anyhow::{anyhow, ensure, Context, Result}; use serde::{Deserialize, Serialize}; +use std::collections::BTreeMap; use std::fs; use std::path::{Path, PathBuf}; use tiktoken_rs::CoreBPE; @@ -88,25 +89,28 @@ pub struct TokenizeReport { pub bin_path: PathBuf, } -// encode every `.md` cache file under `/.ccc` into `tokens.bin` + -// `tokens.json`, then verify the persisted stream decodes back to the corpus -pub fn tokenize(root: &Path, enc: Encoding) -> Result { - let ccc = root.join(".ccc"); - ensure!( - ccc.is_dir(), - "no .ccc directory at {} - run `ccc scan` first", - ccc.display() - ); +// Encode a rendered map into `tokens.bin` + `tokens.json` under `ccc`, then +// verify the persisted stream decodes back to it. +// +// The corpus is passed in rather than read back off disk: the map is built in +// memory by whoever called this, and re-reading the markdown would encode a +// round trip through the filesystem instead of the thing that was mapped. It +// also means a token stream can be produced for a project that never writes +// the markdown at all. +pub fn tokenize( + corpus: &BTreeMap, + ccc: &Path, + enc: Encoding, +) -> Result { + ensure!(!corpus.is_empty(), "nothing to encode - the map is empty"); let bpe = enc.load()?; - - let names = list_markdown(&ccc)?; - ensure!(!names.is_empty(), "no .md cache files in {}", ccc.display()); + let names = ordered_names(corpus); let mut stream: Vec = Vec::new(); let mut entries: Vec = Vec::new(); for name in &names { - let text = fs::read_to_string(ccc.join(name)).with_context(|| format!("reading {name}"))?; - let toks = bpe.encode_ordinary(&text); + let text = &corpus[name]; + let toks = bpe.encode_ordinary(text); entries.push(FileEntry { file: name.clone(), offset: stream.len(), @@ -115,6 +119,8 @@ pub fn tokenize(root: &Path, enc: Encoding) -> Result { stream.extend_from_slice(&toks); } + fs::create_dir_all(ccc).with_context(|| format!("creating {}", ccc.display()))?; + // tokens.bin - little-endian u32 stream let mut bytes = Vec::with_capacity(stream.len() * 4); for t in &stream { @@ -139,7 +145,7 @@ pub fn tokenize(root: &Path, enc: Encoding) -> Result { .with_context(|| format!("writing {}", ccc.join(TOKENS_INDEX).display()))?; // verify the persisted artifacts round-trip back to the exact corpus - verify_roundtrip(root, enc, &names)?; + verify_roundtrip(ccc, enc, corpus, &names)?; Ok(TokenizeReport { files: index.files.len(), @@ -163,13 +169,17 @@ pub fn clear(ccc: &Path) -> Result<()> { } // reload persisted tokens from disk and confirm they decode to the corpus -fn verify_roundtrip(root: &Path, enc: Encoding, names: &[String]) -> Result<()> { - let cache = TokenCache::load(root)?; +fn verify_roundtrip( + ccc: &Path, + enc: Encoding, + corpus: &BTreeMap, + names: &[String], +) -> Result<()> { + let cache = TokenCache::load(ccc)?; let bpe = enc.load()?; - let ccc = root.join(".ccc"); let mut expected = String::new(); for name in names { - expected.push_str(&fs::read_to_string(ccc.join(name))?); + expected.push_str(&corpus[name]); } let decoded = bpe .decode(cache.all()) @@ -178,21 +188,15 @@ fn verify_roundtrip(root: &Path, enc: Encoding, names: &[String]) -> Result<()> Ok(()) } -fn list_markdown(ccc: &Path) -> Result> { - let mut names = Vec::new(); - for entry in fs::read_dir(ccc)? { - let name = entry?.file_name().to_string_lossy().to_string(); - if name.ends_with(".md") { - names.push(name); - } - } - names.sort(); - // keep the index first for a stable, human-friendly layout +// the corpus in stream order: the index first, then the rest by name, so the +// layout is stable and a reader can find `CCC.md` at offset zero +fn ordered_names(corpus: &BTreeMap) -> Vec { + let mut names: Vec = corpus.keys().cloned().collect(); if let Some(pos) = names.iter().position(|n| n == "CCC.md") { let c = names.remove(pos); names.insert(0, c); } - Ok(names) + names } // loaded token cache: the raw `u32` stream plus its index @@ -204,8 +208,7 @@ pub struct TokenCache { } impl TokenCache { - pub fn load(root: &Path) -> Result { - let ccc = root.join(".ccc"); + pub fn load(ccc: &Path) -> Result { let idx_path = ccc.join(TOKENS_INDEX); let json = fs::read_to_string(&idx_path) .with_context(|| format!("reading {} (run `ccc tokenize`)", idx_path.display()))?; @@ -268,24 +271,27 @@ impl TokenCache { mod tests { use super::*; + fn corpus() -> BTreeMap { + BTreeMap::from([ + ("CCC.md".to_string(), "# CodeCaChe\n# files\n".to_string()), + ( + "src-main.rs.md".to_string(), + "# main.rs.md\n# funcs\n - L1:4@main\n".to_string(), + ), + ]) + } + #[test] fn tokenize_load_roundtrip() { let dir = std::env::temp_dir().join(format!("ccc-tok-{}", std::process::id())); let ccc = dir.join(".ccc"); - fs::create_dir_all(&ccc).unwrap(); - fs::write(ccc.join("CCC.md"), "# ContextCodeCache\n# files\n").unwrap(); - fs::write( - ccc.join("src-main.rs.md"), - "# main.rs.md\n# funcs\n - L1:4@main\n", - ) - .unwrap(); - - let report = tokenize(&dir, Encoding::O200kBase).unwrap(); + // the destination need not exist, and the corpus need never be on disk + let report = tokenize(&corpus(), &ccc, Encoding::O200kBase).unwrap(); assert_eq!(report.files, 2); assert!(report.total_tokens > 0); assert_eq!(report.bytes, report.total_tokens * 4); - let cache = TokenCache::load(&dir).unwrap(); + let cache = TokenCache::load(&ccc).unwrap(); // stream is labeled approximate / non-Claude assert!(cache.index.approximate); assert!(cache.index.note.contains("Claude")); @@ -295,6 +301,8 @@ mod tests { let toks = cache.file("src-main.rs.md").unwrap(); let decoded = cache.decode(toks).unwrap(); assert_eq!(decoded, "# main.rs.md\n# funcs\n - L1:4@main\n"); + // no markdown was written, only the stream + assert!(!ccc.join("CCC.md").exists()); fs::remove_dir_all(&dir).ok(); }