diff --git a/CHANGELOG.md b/CHANGELOG.md index 646a21e..5582cd7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,12 @@ ## Unreleased +- The plugin folder now carries its own copy of the server code under + `client-plugin/server/src`, so a directory install that receives only that folder + starts. `python scripts/build_client_package.py --sync-vendored` rewrites it from + `src/`, and a test fails when the copy drifts. The launcher no longer falls back to + the repository's `src` and reports a missing copy in one line. +- The Claude plugin icon is the 512 px PNG, which keeps every plugin file under 256 KiB. - The Claude plugin manifest carries directory listing fields: display name, keywords, homepage, documentation, support, privacy and terms links, and a 1024 px icon. - Claude Code now asks for the readable workspace and the optional state directory diff --git a/client-plugin/.claude-plugin/icon.png b/client-plugin/.claude-plugin/icon.png index 7147273..b13186f 100644 Binary files a/client-plugin/.claude-plugin/icon.png and b/client-plugin/.claude-plugin/icon.png differ diff --git a/client-plugin/README.md b/client-plugin/README.md index d0158c8..53ce98d 100644 --- a/client-plugin/README.md +++ b/client-plugin/README.md @@ -22,7 +22,9 @@ private state directory** that you can leave empty for read-only access. The Claude manifest passes these values as `${user_config.*}` launch arguments. Portable and Codex manifests keep the REPLACE_WITH_ABSOLUTE_WORKSPACE placeholder; replace it with the directory you want the client to read, and -select your installed Python executable. Keep the complete extracted bundle. The Windows x64 binary MCPB and ZIP include Python; +select your installed Python executable. The plugin folder carries its own copy +of the server code under `server/src`, so keep the complete folder or extracted +bundle together. The Windows x64 binary MCPB and ZIP include Python; open the MCPB in a compatible desktop client and choose a workspace directory, or configure the ZIP's server executable with --workspace ABSOLUTE_DIRECTORY. No model, API key, hosting account, automatic client configuration, or publisher diff --git a/client-plugin/server/serve.py b/client-plugin/server/serve.py index 47bc878..8af291c 100644 --- a/client-plugin/server/serve.py +++ b/client-plugin/server/serve.py @@ -2,9 +2,10 @@ from pathlib import Path import sys base = Path(__file__).resolve().parents[1] -source = base / "server/src" -if not source.is_dir(): - source = base.parent / "src" +source = base / "server" / "src" +if not (source / "index_graph" / "client_mcp.py").is_file(): + sys.stderr.write("index: the server code is missing from the plugin folder. Reinstall the plugin.\n") + raise SystemExit(1) sys.path.insert(0, str(source)) from index_graph.client_mcp import main if __name__ == "__main__": diff --git a/client-plugin/server/src/index_graph/__init__.py b/client-plugin/server/src/index_graph/__init__.py new file mode 100644 index 0000000..5db91ad --- /dev/null +++ b/client-plugin/server/src/index_graph/__init__.py @@ -0,0 +1,18 @@ +"""Compact JSON repository inventory maps for multi-repo workspaces.""" + +from __future__ import annotations + +__version__ = "2.15.0" + +from .classify import classify +from .config import Config, Rule, default_config, load_config +from .model import SCHEMA_VERSION, Map, RepoRow +from .route import build_route +from .scan import build_map, discover_repos, write_map + +__all__ = [ + "build_map", "write_map", "discover_repos", + "Map", "RepoRow", "SCHEMA_VERSION", + "Config", "Rule", "load_config", "default_config", + "classify", "build_route", "__version__", +] diff --git a/client-plugin/server/src/index_graph/__main__.py b/client-plugin/server/src/index_graph/__main__.py new file mode 100644 index 0000000..bfdcd0c --- /dev/null +++ b/client-plugin/server/src/index_graph/__main__.py @@ -0,0 +1,4 @@ +from .cli import main + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/client-plugin/server/src/index_graph/arch/__init__.py b/client-plugin/server/src/index_graph/arch/__init__.py new file mode 100644 index 0000000..d83b6cf --- /dev/null +++ b/client-plugin/server/src/index_graph/arch/__init__.py @@ -0,0 +1,10 @@ +"""Architecture criteria and the check that measures a graph against them.""" +from __future__ import annotations + +from .criteria import ArchitectureCriteria, ForbidRule, parse_architecture +from .check import Finding, check_graph + +__all__ = [ + "ArchitectureCriteria", "ForbidRule", "parse_architecture", + "Finding", "check_graph", +] diff --git a/client-plugin/server/src/index_graph/arch/check.py b/client-plugin/server/src/index_graph/arch/check.py new file mode 100644 index 0000000..7c8e111 --- /dev/null +++ b/client-plugin/server/src/index_graph/arch/check.py @@ -0,0 +1,128 @@ +"""Measure a graph against an ArchitectureCriteria; produce evidence-bearing findings. + +`glob_to_regex` is imported lazily inside the matcher so this module can be +imported from config.py without a circular import. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass + +from .criteria import ArchitectureCriteria + + +@dataclass(frozen=True) +class Finding: + rule: str + detail: str + edge: str | None + evidence: str | None + + +def _match(glob: str, name: str) -> bool: + from ..config import glob_to_regex + return re.match(glob_to_regex(glob), name) is not None + + +def _first_evidence(rel: dict) -> str | None: + sigs = rel.get("signals") or [] + if not sigs: + return None + s = sigs[0] + f = s.get("file") + if not f: + return None + line = s.get("line") + return f"{f}:{line}" if line is not None else f + + +def _layer_of(name: str, layers: tuple[str, ...]) -> int | None: + for i, layer in enumerate(layers): + if (name == layer or name.startswith(layer + "/") + or _match(f"{layer}/**", name) or _match(f"**/{layer}", name)): + return i + return None + + +def check_graph(pack: dict, criteria: ArchitectureCriteria) -> list[Finding]: + findings: list[Finding] = [] + relations = [r for r in pack.get("relations", []) if not r.get("external")] + + # forbidden edges + for rule in criteria.forbid: + for r in relations: + frm, to = r.get("from"), r.get("to") + if to and _match(rule.from_glob, frm) and _match(rule.to_glob, to): + findings.append(Finding( + "forbid", f"{rule.from_glob} must not depend on {rule.to_glob}", + f"{frm} -> {to}", _first_evidence(r))) + + # layers: lower index = lower layer; an edge from lower to higher is a violation + if criteria.layers: + for r in relations: + frm, to = r.get("from"), r.get("to") + if not to: + continue + li, lj = _layer_of(frm, criteria.layers), _layer_of(to, criteria.layers) + if li is not None and lj is not None and li < lj: + findings.append(Finding( + "layer", + f"{criteria.layers[li]} must not depend upward on {criteria.layers[lj]}", + f"{frm} -> {to}", _first_evidence(r))) + + # cycle ceiling + if criteria.max_cycles is not None: + n = len(pack.get("cycles", [])) + if n > criteria.max_cycles: + findings.append(Finding( + "max_cycles", + f"{n} dependency cycle(s) exceed the ceiling of {criteria.max_cycles}", + None, None)) + + # repo names present in the workspace (used by forbid, owns, require checks) + names: list[str] = [] + if criteria.owns or criteria.require or criteria.forbid: + names_src = [r.get("from") for r in pack.get("relations", [])] + names_src += [r.get("to") for r in pack.get("relations", []) if r.get("to")] + names_src += list(pack.get("roles", {}).keys()) + names = sorted({n for n in names_src if n}) + + # a forbid rule whose endpoints name no repo cannot be meaningfully checked: + # the forbidden edge is trivially absent because a glob is wrong, not + # because the code obeys the rule. Mirror require/owns: emit an UNVERIFIABLE + # criterion-quality finding, never a silent vacuous pass. + for rule in criteria.forbid: + if not (any(_match(rule.from_glob, n) for n in names) + and any(_match(rule.to_glob, n) for n in names)): + findings.append(Finding( + "forbid_unmatched", + f"forbid {rule.from_glob} -> {rule.to_glob} names a repo not in the workspace", + None, None)) + + # ownership: a declared owner glob that matches no repo is a finding + for glob, owner in criteria.owns: + if not any(_match(glob, n) for n in names): + findings.append(Finding( + "owns", f"ownership glob {glob} ({owner}) matches no repo", None, None)) + + # required edges (Reflexion conformance). An intended dependency between repos that + # both exist but are not connected is an 'absence' (a confirmed breach -> DRIFT). A + # require rule whose endpoints are not in the workspace at all is 'require_unmatched', + # a criterion-quality gap that reads UNVERIFIABLE, mirroring an unmatched layer. + for rule in criteria.require: + if not (any(_match(rule.from_glob, n) for n in names) + and any(_match(rule.to_glob, n) for n in names)): + findings.append(Finding( + "require_unmatched", + f"require {rule.from_glob} -> {rule.to_glob} names a repo not in the workspace", + None, None)) + continue + present = any( + r.get("to") and _match(rule.from_glob, r.get("from")) and _match(rule.to_glob, r.get("to")) + for r in relations) + if not present: + findings.append(Finding( + "absence", f"{rule.from_glob} should depend on {rule.to_glob} but does not", + None, None)) + + return sorted(findings, key=lambda f: (f.rule, f.edge or "", f.detail)) diff --git a/client-plugin/server/src/index_graph/arch/criteria.py b/client-plugin/server/src/index_graph/arch/criteria.py new file mode 100644 index 0000000..2746d15 --- /dev/null +++ b/client-plugin/server/src/index_graph/arch/criteria.py @@ -0,0 +1,65 @@ +"""The architecture criterion: a rule the graph is measured against. + +This module imports nothing from the rest of the package so that config.py can +import it without a cycle. The check that consumes a criterion lives in check.py. +""" +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class ForbidRule: + from_glob: str + to_glob: str + + +@dataclass(frozen=True) +class RequireRule: + """An intended dependency that must be realized. A missing one is an + 'absence' in Reflexion-model conformance (the architecture you declared is + not the one the code actually built).""" + from_glob: str + to_glob: str + + +@dataclass(frozen=True) +class ArchitectureCriteria: + layers: tuple[str, ...] = () + forbid: tuple[ForbidRule, ...] = () + max_cycles: int | None = None + owns: tuple[tuple[str, str], ...] = () + require: tuple[RequireRule, ...] = () + + @property + def declared(self) -> bool: + return bool(self.layers or self.forbid or self.require + or self.max_cycles is not None or self.owns) + + +def parse_architecture(data: dict) -> ArchitectureCriteria: + """Parse the [architecture] TOML block. Raises SystemExit on malformed input.""" + layers = tuple(str(x) for x in data.get("layers", [])) + + forbid: list[ForbidRule] = [] + for idx, item in enumerate(data.get("forbid", [])): + if not isinstance(item, dict) or "from" not in item or "to" not in item: + raise SystemExit(f"[architecture] forbid[{idx}] requires 'from' and 'to'") + forbid.append(ForbidRule(str(item["from"]), str(item["to"]))) + + require: list[RequireRule] = [] + for idx, item in enumerate(data.get("require", [])): + if not isinstance(item, dict) or "from" not in item or "to" not in item: + raise SystemExit(f"[architecture] require[{idx}] requires 'from' and 'to'") + require.append(RequireRule(str(item["from"]), str(item["to"]))) + + mc = data.get("max_cycles", None) + if mc is not None and (isinstance(mc, bool) or not isinstance(mc, int) or mc < 0): + raise SystemExit("[architecture] max_cycles must be a non-negative integer") + + owns_raw = data.get("owns", {}) + if not isinstance(owns_raw, dict): + raise SystemExit("[architecture] owns must be a table of glob = owner") + owns = tuple(sorted((str(k), str(v)) for k, v in owns_raw.items())) + + return ArchitectureCriteria(layers, tuple(forbid), mc, owns, tuple(require)) diff --git a/client-plugin/server/src/index_graph/bench/__init__.py b/client-plugin/server/src/index_graph/bench/__init__.py new file mode 100644 index 0000000..71095fd --- /dev/null +++ b/client-plugin/server/src/index_graph/bench/__init__.py @@ -0,0 +1,4 @@ +"""Token economy: index's structural pack vs the source it reads (the no-eval-gap number).""" +from .economy import SCHEMA, bench_workspace + +__all__ = ["SCHEMA", "bench_workspace"] diff --git a/client-plugin/server/src/index_graph/bench/economy.py b/client-plugin/server/src/index_graph/bench/economy.py new file mode 100644 index 0000000..798eba1 --- /dev/null +++ b/client-plugin/server/src/index_graph/bench/economy.py @@ -0,0 +1,90 @@ +"""Token economy: how much smaller is index's structural pack than the source it reads. + +The thesis the research converges on (a structural map is cheaper than letting an +agent read files) is here turned into a number anyone can reproduce on their own +workspace. index reads the manifests and sources of every ecosystem to build the +graph; it emits one compact structural pack. This measures the ratio between the two. + +Bytes are exact and model-agnostic. The token figures use the common ~4 bytes/token +approximation, but the reduction RATIO is independent of that constant (it divides out), +so the headline number does not depend on any tokenizer. +""" +from __future__ import annotations + +import json +from pathlib import Path + +from ..context.pack import to_json +from ..freshness.fingerprint import relevant_files +from ..graph.build import build_graph + +SCHEMA = "index.bench/1" +BYTES_PER_TOKEN = 4 # common ~4 bytes/token approximation; the reduction ratio does not depend on it + + +def _source_bytes(repo_paths: dict[str, Path]) -> tuple[int, int]: + """Total bytes and file count of the graph-relevant files index reads.""" + total = files = 0 + for root in repo_paths.values(): + for p in relevant_files(root): + try: + total += p.stat().st_size + except OSError: + continue + files += 1 + return total, files + + +def _compact_bytes(obj) -> int: + return len(json.dumps(obj, sort_keys=True, separators=(",", ":")).encode("utf-8")) + + +def _grounding(pack: dict) -> dict: + """Faithfulness of the reduction: what fraction of the internal dependency + edges the compact pack KEEPS are backed by real file:line source evidence. + + A byte reduction is only honest if it fabricates nothing. grep-and-truncate + or an LLM summary can drop or invent structure; index's every retained edge + cites the import that produced it. This turns 'the reduction is faithful' + into a measured number, not a promise: grounded internal edges / internal + edges. 1.0 means every structural fact kept is provably in the source.""" + internal = [r for r in pack.get("relations", []) if not r.get("external")] + grounded = [r for r in internal + if any(s.get("file") for s in r.get("signals", []))] + total = len(internal) + return { + "internal_edges": total, + "grounded_edges": len(grounded), + # a workspace with no internal edges grounded NOTHING: report the honest + # null, not a vacuous 1.0 that reads as perfect faithfulness for a + # reduction that had no structure to fabricate or preserve + "edge_grounding": (round(len(grounded) / total, 4) if total else None), + "note": ("fraction of kept dependency edges carrying file:line evidence; " + "1.0 = the reduction fabricates no structure; null = no " + "internal edges to ground"), + } + + +def bench_workspace(repo_paths: dict[str, Path], *, use_graph_cache: bool = True) -> dict: + """The bytes index reads vs the bytes of the structural pack it emits, AND + the faithfulness of that reduction (every kept edge grounded in source). + + Deterministic for a fixed workspace, so the report is re-checkable like every + other index verdict. + """ + src_bytes, n_files = _source_bytes(repo_paths) + pack = to_json(build_graph(repo_paths, use_cache=use_graph_cache)) + pack_bytes = _compact_bytes(pack) + reduction = round(src_bytes / pack_bytes, 1) if pack_bytes else None + return { + "schema": SCHEMA, + "repos": len(repo_paths), + "source_files": n_files, + "source_bytes": src_bytes, + "pack_bytes": pack_bytes, + "reduction": reduction, + "bytes_per_token": BYTES_PER_TOKEN, + "approx_tokens_source": src_bytes // BYTES_PER_TOKEN, + "approx_tokens_pack": pack_bytes // BYTES_PER_TOKEN, + "faithfulness": _grounding(pack), + } diff --git a/client-plugin/server/src/index_graph/cache.py b/client-plugin/server/src/index_graph/cache.py new file mode 100644 index 0000000..9580238 --- /dev/null +++ b/client-plugin/server/src/index_graph/cache.py @@ -0,0 +1,118 @@ +"""Small filesystem cache for expensive workspace-wide index surfaces.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +from time import time +from typing import Any, Callable + +SCHEMA = "index.cache-entry/v1" + + +def _ttl_seconds() -> float: + raw = os.environ.get("INDEX_CACHE_TTL_SECONDS", "900") + try: + return max(0.0, float(raw)) + except ValueError: + return 900.0 + + +def _cache_dir() -> Path: + raw = os.environ.get("INDEX_CACHE_DIR") + if raw: + return Path(raw) + base = os.environ.get("LOCALAPPDATA") + if base: + return Path(base) / "index_graph" / "cache" + return Path.home() / ".cache" / "index_graph" + + +def _sha256_text(value: str) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def workspace_signature(root: Path) -> str: + """Cheap workspace signature for cache invalidation. + + This is intentionally cheaper than the full freshness fingerprint. It catches + config and top-level entry name changes, while the TTL bounds nested-change + staleness. Directory mtimes are excluded so volatile runtime trees such as + .scratch do not churn interactive cache keys before a cache lookup. + """ + root = root.resolve() + parts = [str(root)] + for cfg in (root / ".index.toml", root / ".repomap.toml"): + try: + stat = cfg.stat() + except OSError: + parts.append(f"{cfg.name}:missing") + else: + parts.append(f"{cfg.name}:{stat.st_mtime_ns}:{stat.st_size}") + try: + entries = sorted(root.iterdir(), key=lambda item: item.name.lower()) + except OSError as exc: + parts.append(f"root-error:{type(exc).__name__}:{exc}") + return _sha256_text("|".join(parts)) + for entry in entries: + try: + stat = entry.stat() + except OSError: + parts.append(f"{entry.name}:unstatable") + continue + kind = "d" if entry.is_dir() else "f" + if kind == "d": + parts.append(f"{entry.name}:{kind}") + else: + parts.append(f"{entry.name}:{kind}:{stat.st_mtime_ns}:{stat.st_size}") + return _sha256_text("|".join(parts)) + + +def _key(tool: str, root: Path, args: dict[str, Any]) -> str: + payload = { + "tool": tool, + "root": str(root.resolve()), + "args": args, + "workspace_signature": workspace_signature(root), + } + body = json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str) + return _sha256_text(body) + + +def _path(key: str) -> Path: + return _cache_dir() / f"{key}.json" + + +def cached_text( + tool: str, + root: Path, + args: dict[str, Any], + build: Callable[[], str], + *, + enabled: bool = True, + cache_root: Path | None = None, +) -> str: + ttl = _ttl_seconds() + if not enabled or ttl <= 0: + return build() + key = _key(tool, root, args) + path = _path(key) if cache_root is None else cache_root / f"{key}.json" + now = time() + try: + data = json.loads(path.read_text(encoding="utf-8")) + if data.get("schema") == SCHEMA and now - float(data.get("created_at", 0.0)) <= ttl: + return str(data.get("text", "")) + except (OSError, json.JSONDecodeError, ValueError): + pass + text = build() + try: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text( + json.dumps({"schema": SCHEMA, "created_at": now, "text": text}, separators=(",", ":")), + encoding="utf-8", + ) + except OSError: + pass + return text diff --git a/client-plugin/server/src/index_graph/certify/__init__.py b/client-plugin/server/src/index_graph/certify/__init__.py new file mode 100644 index 0000000..52d1327 --- /dev/null +++ b/client-plugin/server/src/index_graph/certify/__init__.py @@ -0,0 +1,6 @@ +"""Verdict certificates for check and drift.""" +from __future__ import annotations + +from .certificate import canonical_sha, build_certificate + +__all__ = ["canonical_sha", "build_certificate"] diff --git a/client-plugin/server/src/index_graph/certify/certificate.py b/client-plugin/server/src/index_graph/certify/certificate.py new file mode 100644 index 0000000..de3db4c --- /dev/null +++ b/client-plugin/server/src/index_graph/certify/certificate.py @@ -0,0 +1,42 @@ +"""The verdict certificate: re-checkable, three answers, never a fourth. + +A consumer believes a certificate by re-running its `recheck` command, +recomputing the hashes, and confirming the verdict. There is no TRUSTED. +""" +from __future__ import annotations + +import hashlib +import json + +_VERDICTS = frozenset({"MATCH", "DRIFT", "UNVERIFIABLE"}) + + +def canonical_sha(obj) -> str: + blob = json.dumps(obj, sort_keys=True, separators=(",", ":")).encode("utf-8") + return hashlib.sha256(blob).hexdigest() + + +def build_certificate(kind: str, *, content: dict, criterion: dict | None, + verdict: str, findings: list[dict], recheck: str, + tool_version: str, coverage: dict | None = None, + freshness: dict | None = None) -> dict: + if verdict not in _VERDICTS: + raise ValueError(f"verdict must be one of {sorted(_VERDICTS)}, got {verdict!r}") + cert = { + "schema": "index.certificate/1", + "tool_version": tool_version, + "kind": kind, + "content_sha256": canonical_sha(content), + "criterion_sha256": canonical_sha(criterion) if criterion is not None else None, + "verdict": verdict, + "findings": findings, + "recheck": recheck, + } + if coverage is not None: + # soundness scope: what the verdict could and could not structurally verify + cert["coverage"] = coverage + if freshness is not None: + # the workspace content fingerprint at mint time, so a consumer can later + # ask `index freshness` whether the ground truth has moved since. + cert["freshness"] = freshness + return cert diff --git a/client-plugin/server/src/index_graph/classify.py b/client-plugin/server/src/index_graph/classify.py new file mode 100644 index 0000000..ea1a7aa --- /dev/null +++ b/client-plugin/server/src/index_graph/classify.py @@ -0,0 +1,29 @@ +"""Pure classification: ordered glob rules, then a remote-host fallback.""" + +from __future__ import annotations + +from urllib.parse import urlsplit + +from .config import PUBLIC_HOSTS, Config + + +def _remote_host(origin: str) -> str | None: + if not origin: + return None + if "://" not in origin and "@" in origin and ":" in origin: + # scp-like SSH form: git@github.com:owner/repo.git + return origin.split("@", 1)[1].split(":", 1)[0] or None + return urlsplit(origin).hostname + + +def classify(path: str, is_repo: bool, origin: str, config: Config) -> str: + for rule in config.rules: + if rule.regex.match(path): + return rule.class_ + if is_repo: + host = _remote_host(origin) + if host is None: + return "local" + return "public" if host in PUBLIC_HOSTS else "private" + name = path.rsplit("/", 1)[-1] + return "hidden" if name.startswith(".") else "entry" diff --git a/client-plugin/server/src/index_graph/cli.py b/client-plugin/server/src/index_graph/cli.py new file mode 100644 index 0000000..deb3ef6 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli.py @@ -0,0 +1,190 @@ +"""Command-line entry point: map (default) + graph + context subcommands.""" + +from __future__ import annotations + +import json +import sys +from dataclasses import replace + +from . import __version__ +from .cli_handlers import ( + cmd_atlas, + cmd_workbench, + cmd_bench, + cmd_check, + cmd_context, + cmd_context_envelope, + cmd_lens, + cmd_drift, + cmd_freshness, + cmd_graph, + cmd_internals, + cmd_internals_symbols, + cmd_lsp, + cmd_mcp, + cmd_router, + cmd_serve, + cmd_snapshot, + cmd_symbols, + cmd_verify, + cmd_watch, + cmd_viz, +) +from .cli_parser import build_parser +from .router_job_surface import cmd_router_job +from .route import cmd_route +from .config import load_config +from .context.select import cmd_select +from .flagship import cmd_demo, cmd_doctor, cmd_status +from .freshness.invalidate_cli import cmd_invalidate +from .graph.walk import GraphSourceError +from .scan import build_map, write_map +from .wiki.cli import cmd_wiki + +_SUBCOMMANDS = { + "map", + "graph", + "context", + "context-envelope", + "lens", + "select", + "viz", + "atlas", + "workbench", + "wiki", + "internals", + "internals-symbols", + "symbols", + "check", + "snapshot", + "drift", + "router", + "router-job", + "route", + "verify", + "freshness", + "watch", + "invalidate", + "bench", + "serve", + "lsp", + "mcp", + "status", + "doctor", + "demo", +} + +# Dispatch table for named subcommands. `map` is intentionally absent: it is +# the implicit default reached when no known subcommand leads the invocation. +_DISPATCH = { + "status": cmd_status, + "doctor": cmd_doctor, + "demo": cmd_demo, + "atlas": cmd_atlas, + "workbench": cmd_workbench, + "wiki": cmd_wiki, + "graph": cmd_graph, + "context": cmd_context, + "context-envelope": cmd_context_envelope, + "lens": cmd_lens, + "select": cmd_select, + "viz": cmd_viz, + "internals": cmd_internals, + "internals-symbols": cmd_internals_symbols, + "symbols": cmd_symbols, + "check": cmd_check, + "snapshot": cmd_snapshot, + "drift": cmd_drift, + "router": cmd_router, + "router-job": cmd_router_job, + "route": cmd_route, + "verify": cmd_verify, + "freshness": cmd_freshness, + "watch": cmd_watch, + "invalidate": cmd_invalidate, + "bench": cmd_bench, + "serve": cmd_serve, + "lsp": cmd_lsp, + "mcp": cmd_mcp, +} + + +def _map_summary(data) -> str: + return ( + f"repos={data.repo_count} dirty_verified={data.dirty_count} " + f"metadata_status={data.metadata_status} " + f"metadata_unknown={data.metadata_unknown_count}" + ) + + +def _configure_stdio() -> None: + for stream_name in ("stdout", "stderr"): + stream = getattr(sys, stream_name, None) + if stream is None or not hasattr(stream, "reconfigure"): + continue + try: + stream.reconfigure(encoding="utf-8", errors="replace") + except (OSError, ValueError): + pass + + +def _cmd_map(args) -> int: + root = args.root.resolve() + if not root.is_dir(): + raise SystemExit(f"root not found: {root}") + config = load_config(args.config, root) + if args.jobs is not None: + if args.jobs < 1: + raise SystemExit("--jobs must be a positive integer") + config = replace(config, jobs=args.jobs) + if args.json: + if args.dry_run: + raise SystemExit( + "map: --dry-run applies to the file-writing mode; " + "--json already writes nothing" + ) + print(json.dumps(build_map(root, config, __version__, resume_state=args.resume_state).to_json(), indent=2)) + return 0 + output = args.output.resolve() if args.output else root / "INDEX.json" + if args.dry_run: + print(f"index map: would write {output} (dry-run, nothing written)") + data = build_map(root, config, __version__, resume_state=args.resume_state) + print(_map_summary(data)) + return 0 + print(f"index map: writing {output}") + data = write_map(root, config, __version__, output, resume_state=args.resume_state) + print(f"wrote {output}") + print(_map_summary(data)) + return 0 + + +def _normalize_argv(argv: list[str] | None) -> list[str]: + raw = list(sys.argv[1:] if argv is None else argv) + # No leading subcommand: route top-level --version/--help to the root + # parser; otherwise treat the invocation as the implicit `map` command + # (preserves v0.2.0 behavior). + if not raw or raw[0] not in _SUBCOMMANDS: + if raw and raw[0] in ("--version", "-h", "--help"): + build_parser().parse_args(raw[:1]) # prints and exits + raw = ["map", *raw] + return raw + + +def main(argv: list[str] | None = None) -> int: + _configure_stdio() + args = build_parser().parse_args(_normalize_argv(argv)) + handler = _DISPATCH.get(args.cmd, _cmd_map) + try: + return handler(args) + except GraphSourceError as exc: + receipt = {"schema": "index.graph-error/v1", "status": "UNVERIFIABLE", + "error_type": type(exc).__name__, "message": str(exc)} + if getattr(args, "json", False): + print(json.dumps(receipt, indent=2, sort_keys=True)) + else: + print(f"UNVERIFIABLE: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/client-plugin/server/src/index_graph/cli_handlers/__init__.py b/client-plugin/server/src/index_graph/cli_handlers/__init__.py new file mode 100644 index 0000000..e7381ce --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/__init__.py @@ -0,0 +1,41 @@ +"""Per-subcommand handler functions for the `index` CLI. + +Split out of the former monolithic ``cli.py`` so every module stays under +the 300-line ceiling and every handler under 50 lines. Behavior is identical +to the pre-split single-file implementation. +""" + +from __future__ import annotations + +from .certify import cmd_check, cmd_drift, cmd_snapshot +from .context import cmd_context, cmd_context_envelope, cmd_lens +from .graph import cmd_graph, cmd_internals, cmd_internals_symbols, cmd_viz +from .lsp import cmd_lsp +from .maps import cmd_atlas, cmd_router, cmd_workbench +from .serve import cmd_serve +from .symbols import cmd_symbols +from .verify import cmd_bench, cmd_freshness, cmd_mcp, cmd_verify, cmd_watch + +__all__ = [ + "cmd_atlas", + "cmd_workbench", + "cmd_bench", + "cmd_check", + "cmd_context", + "cmd_context_envelope", + "cmd_lens", + "cmd_drift", + "cmd_freshness", + "cmd_graph", + "cmd_internals", + "cmd_internals_symbols", + "cmd_lsp", + "cmd_mcp", + "cmd_router", + "cmd_serve", + "cmd_snapshot", + "cmd_symbols", + "cmd_verify", + "cmd_watch", + "cmd_viz", +] diff --git a/client-plugin/server/src/index_graph/cli_handlers/_common.py b/client-plugin/server/src/index_graph/cli_handlers/_common.py new file mode 100644 index 0000000..c3ceb7c --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/_common.py @@ -0,0 +1,79 @@ +"""Shared helpers for the CLI subcommand handlers.""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +from ..config import load_config +from ..scan import ( + ScanBudget, + ScanBudgetExceeded, + discover_repos, + enforce_interactive_repo_limit, + repo_key_map, +) + + +def repo_paths(root: Path, *, skipped: list | None = None, budget_ms: int | None = None) -> dict[str, Path]: + # discover_repos requires a Config; use neutral defaults for graph/context. + # `skipped`, when given, collects directories the scan could not read, so a + # narrowed scan is a receiptable fact rather than a stderr-only warning. + config = load_config(None, root) + budget = ScanBudget(budget_ms) + repos = discover_repos( + root, + config, + skipped=skipped, + checkpoint=budget.checkpoint if budget.budget_ms > 0 else None, + ) + if budget.exhausted: + raise ScanBudgetExceeded(root=root, budget=budget, repo_count=len(repos), skipped=skipped) + keyed = repo_key_map( + root, + repos, + include_root_repo=config.include_root_repo, + ) + enforce_interactive_repo_limit(len(keyed), budget_ms=budget.budget_ms) + return keyed + + +def require_dir(root: Path) -> Path: + resolved = root.resolve() + if not resolved.is_dir(): + raise SystemExit(f"root not found: {resolved}") + return resolved + + +def rel_to_root(root: Path, p: Path) -> str: + r = p.resolve().relative_to(root).as_posix() + return "" if r == "." else r # a repo AT the root -> "" dir + + +def head_commit(root) -> str | None: + try: + out = subprocess.run( + ["git", "-C", str(root), "rev-parse", "HEAD"], + capture_output=True, + text=True, + timeout=5, + ) + return out.stdout.strip() or None + except Exception: + return None + + +def emit_cert(cert: dict, as_json: bool) -> int: + if as_json: + print(json.dumps(cert, indent=2, sort_keys=True)) + else: + print(f"verdict={cert['verdict']} findings={len(cert['findings'])}") + for f in cert["findings"]: + loc = f" ({f['evidence']})" if f.get("evidence") else "" + print(f" [{f['rule']}] {f['detail']}{loc}") + cov = cert.get("coverage") + if cov is not None and not cov.get("complete", True): + n = len(cov.get("unverifiable_repos", {})) + print(f" coverage: incomplete, {n} repo(s) with unverifiable regions") + return 0 if cert["verdict"] == "MATCH" else 1 diff --git a/client-plugin/server/src/index_graph/cli_handlers/certify.py b/client-plugin/server/src/index_graph/cli_handlers/certify.py new file mode 100644 index 0000000..373f850 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/certify.py @@ -0,0 +1,228 @@ +"""Certificate handlers: check, snapshot, drift.""" + +from __future__ import annotations + +import json + +from .. import __version__ +from ..config import load_config +from ..context.pack import to_json +from ..graph.build import build_graph +from ._common import emit_cert, repo_paths, require_dir + + +def cmd_check(args) -> int: + from ..arch.check import check_graph + from ..certify import build_certificate + from ..freshness import workspace_fingerprint + + root = require_dir(args.root) + config = load_config(args.config, root) + crit = config.architecture + scan_skipped: list = [] + paths = repo_paths(root, skipped=scan_skipped) + graph = build_graph(paths) + pack = to_json(graph) + names = set(pack.get("roles", {}).keys()) + fresh_stamp = workspace_fingerprint(paths) if args.freshness else None + fresh_flag = " --freshness" if args.freshness else "" + + if not crit.declared: + cert = build_certificate( + "check", + content=pack, + criterion=None, + verdict="UNVERIFIABLE", + findings=[ + { + "rule": "criterion", + "detail": "no [architecture] criterion declared", + "edge": None, + "evidence": None, + } + ], + recheck=f"index check --root {args.root}{fresh_flag}", + tool_version=__version__, + freshness=fresh_stamp, + ) + return emit_cert(cert, args.json) + + findings = [ + {"rule": f.rule, "detail": f.detail, "edge": f.edge, "evidence": f.evidence} + for f in check_graph(pack, crit) + ] + # a scan narrowed by unreadable directories could not see the whole tree: + # a criterion-quality gap (UNVERIFIABLE), never a MATCH over a partial scan + if scan_skipped: + findings.append({ + "rule": "scan_incomplete", + "detail": f"repo discovery skipped {len(scan_skipped)} unreadable " + f"director(y/ies): {sorted(scan_skipped)[:3]}", + "edge": None, "evidence": None}) + internal_content = _check_internals(args, crit, paths, findings) + # an internal graph the analyzer could not fully build (parse errors / + # unreadable files) must not yield a MATCH: the map would certify structure + # it could not fully see. + internal_incomplete = internal_content is not None and any( + not repo.get("coverage", {}).get("complete", True) + for repo in internal_content.values()) + # a *_unmatched rule or a scan_incomplete gap is a criterion-quality gap + # (UNVERIFIABLE), not a breach; capture real_violations BEFORE layer + # findings are appended (order matters). + _gap_rules = {"scan_incomplete"} + real_violations = any( + not f["rule"].endswith("_unmatched") and f["rule"] not in _gap_rules + for f in findings) + unmatched = _check_layers(crit, names, findings) + verdict = _check_verdict(real_violations, unmatched, findings, internal_incomplete) + cert = _check_certificate( + args, crit, pack, internal_content, findings, verdict, fresh_stamp, fresh_flag + ) + return emit_cert(cert, args.json) + + +def _check_internals(args, crit, paths, findings) -> dict | None: + # optional intra-repo module checks: internal cycles against the ceiling + if not args.internals: + return None + from ..internals import build_internals + + internal_content: dict = {} + for name, p in sorted(paths.items()): + g = build_internals(p, name) + internal_content[name] = { + "cycles": [list(c) for c in g.cycles], + "coverage": { + "complete": g.coverage.complete, + "parse_errors": list(g.coverage.parse_errors), + "dynamic_imports": [ + {"file": fpath, "line": ln} + for fpath, ln in g.coverage.dynamic_imports + ], + }, + } + if crit.max_cycles is not None and len(g.cycles) > crit.max_cycles: + findings.append( + { + "rule": "max_cycles", + "detail": f"{name}: {len(g.cycles)} internal module cycle(s) " + f"exceed the ceiling of {crit.max_cycles}", + "edge": None, + "evidence": None, + } + ) + return internal_content + + +def _check_layers(crit, names, findings) -> list[str]: + # criterion-quality warnings: layers that name no repo + unmatched = [ + layer + for layer in crit.layers + if not any( + n == layer or n.startswith(layer + "/") or n.endswith("/" + layer) + for n in names + ) + ] + for layer in unmatched: + findings.append( + { + "rule": "layer", + "detail": f"layer '{layer}' matches no repo", + "edge": None, + "evidence": None, + } + ) + return unmatched + + +def _check_verdict(real_violations, unmatched, findings, internal_incomplete=False) -> str: + # a confirmed breach outranks an unverifiable criterion + if real_violations: + return "DRIFT" + # a *_unmatched criterion gap, an unmatched layer, an internal graph the + # analyzer could not fully build, or a scan narrowed by unreadable + # directories all read UNVERIFIABLE: a MATCH must not be issued over a + # graph or scan that is not fully derivable + if (unmatched or internal_incomplete + or any(f["rule"].endswith("_unmatched") + or f["rule"] == "scan_incomplete" for f in findings)): + return "UNVERIFIABLE" + return "MATCH" + + +def _check_certificate( + args, crit, pack, internal_content, findings, verdict, fresh_stamp, fresh_flag +): + from ..certify import build_certificate + + criterion_doc = { + "layers": list(crit.layers), + "forbid": [{"from": f.from_glob, "to": f.to_glob} for f in crit.forbid], + "max_cycles": crit.max_cycles, + "owns": [list(o) for o in crit.owns], + } + if crit.require: # keep empty-require criteria byte-identical (hash stability) + criterion_doc["require"] = [ + {"from": r.from_glob, "to": r.to_glob} for r in crit.require + ] + content = ( + pack + if internal_content is None + else {"pack": pack, "internals": internal_content} + ) + coverage_doc = None + if internal_content is not None: + incomplete = { + n: internal_content[n]["coverage"] + for n in internal_content + if not internal_content[n]["coverage"]["complete"] + } + coverage_doc = {"complete": not incomplete, "unverifiable_repos": incomplete} + recheck = ( + f"index check --root {args.root}" + + (" --internals" if args.internals else "") + + fresh_flag + ) + return build_certificate( + "check", + content=content, + criterion=criterion_doc, + verdict=verdict, + findings=findings, + recheck=recheck, + tool_version=__version__, + coverage=coverage_doc, + freshness=fresh_stamp, + ) + + +def cmd_snapshot(args) -> int: + from ..drift import dumps_canonical, snapshot_pack + + root = require_dir(args.root) + graph = build_graph(repo_paths(root)) + snap = snapshot_pack(to_json(graph)) + args.out.write_text(dumps_canonical(snap), encoding="utf-8") + print(f"wrote {args.out} repos={len(snap['repos'])} edges={len(snap['edges'])}") + return 0 + + +def cmd_drift(args) -> int: + from ..drift import diff_snapshots, load_snapshot + + old = load_snapshot(args.from_snap.read_text(encoding="utf-8")) + new = load_snapshot(args.to_snap.read_text(encoding="utf-8")) + try: + report = diff_snapshots(old, new) + except ValueError as exc: + raise SystemExit(f"drift: {exc}") + if args.json: + print(json.dumps(report.to_json(), indent=2)) + else: + print(f"verdict={report.verdict}") + for e in report.edges_added: + print(f" edge added: {e}") + for e in report.edges_removed: + print(f" edge removed: {e}") + return 0 if report.verdict == "MATCH" else 1 diff --git a/client-plugin/server/src/index_graph/cli_handlers/context.py b/client-plugin/server/src/index_graph/cli_handlers/context.py new file mode 100644 index 0000000..c139632 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/context.py @@ -0,0 +1,248 @@ +"""Context handlers: context pack and budgeted context envelope.""" + +from __future__ import annotations + +import json +import sys + +from ..context.focus import FocusRejection, focus_rejection, render_rejection +from ..context.pack import closure, focus_subgraph, preservation, render_text, to_json +from ..graph.build import build_graph +from ..scan import ScanBudgetExceeded, ScanWorkloadExceeded, default_interactive_budget_ms +from ._common import repo_paths + + +def _budget_ms(args) -> int: + value = getattr(args, "budget_ms", None) + if value is None: + value = default_interactive_budget_ms() + if value < 0: + raise SystemExit("--budget-ms must be non-negative") + return value + + +def _scan_budget_payload(command: str, exc: ScanBudgetExceeded | ScanWorkloadExceeded) -> dict: + payload = { + "schema": "index.scan-budget-exceeded/v1", + "command": command, + "status": "UNVERIFIABLE", + "message": str(exc), + "budget_ms": getattr(exc, "budget_ms", None), + "partial_repos": getattr(exc, "repo_count", None), + "next_actions": [ + "Increase --budget-ms for a larger bounded interactive scan.", + "Use --budget-ms 0 for an unbounded interactive run.", + "Use index map --resume-state PATH for complete repository inventory over large workspaces.", + ], + } + if hasattr(exc, "elapsed_ms"): + payload["elapsed_ms"] = exc.elapsed_ms + if hasattr(exc, "last_path"): + payload["last_path"] = exc.last_path + if hasattr(exc, "skipped"): + payload["skipped"] = exc.skipped + if hasattr(exc, "repo_limit"): + payload["repo_limit"] = exc.repo_limit + return payload + + +def _emit_scan_budget(args, command: str, exc: ScanBudgetExceeded | ScanWorkloadExceeded) -> int: + payload = _scan_budget_payload(command, exc) + if getattr(args, "json", False): + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + print(f"{command}: UNVERIFIABLE: {exc}") + print("next: increase --budget-ms, use --budget-ms 0, or run index map --resume-state PATH") + return 2 + + +def _write_bounded_json_stdout(payload: dict) -> None: + text = json.dumps(payload, indent=2, sort_keys=True) + "\n" + data = text.encode("utf-8") + buffer = getattr(sys.stdout, "buffer", None) + if buffer is not None: + buffer.write(data) + buffer.flush() + else: + sys.stdout.write(text) + sys.stdout.flush() + + +def cmd_context(args) -> int: + if args.hops is not None and args.hops < 0: + raise SystemExit("--hops must be >= 0") + try: + graph = build_graph(repo_paths(args.root.resolve(), budget_ms=_budget_ms(args))) + except (ScanBudgetExceeded, ScanWorkloadExceeded) as exc: + return _emit_scan_budget(args, "index context", exc) + names = {n.name for n in graph.repos} + if args.audit: + return _context_audit(graph) + preserved = None + if args.focus: + if args.focus not in names: + receipt = focus_rejection(args.focus, names) + print( + json.dumps(receipt, indent=2, sort_keys=True) + if args.json + else render_rejection(receipt) + ) + return 2 + keep = closure(list(graph.edges), args.focus, hops=args.hops) + preserved = preservation(list(graph.edges), keep, args.focus, args.hops) + graph = focus_subgraph(graph, keep) + title = f"focus={args.focus}" + ( + f" hops={args.hops}" if args.hops is not None else "" + ) + else: + title = "workstation context" + return _context_emit(args, graph, title, preserved) + + +def _context_audit(graph) -> int: + data = to_json(graph) + print(f"salience-faithfulness warnings: {len(data['salience_audit'])}") + for w in data["salience_audit"]: + print(f" [{w['kind']}] {w['node']} (in={w['in_degree']}): {w['note']}") + return 0 + + +def _context_emit(args, graph, title, preserved) -> int: + if args.json: + pack = to_json(graph) + if preserved is not None: + pack["preserved"] = preserved + print(json.dumps(pack, indent=2, sort_keys=True)) + else: + text = render_text(graph, title) + if preserved is not None: + b = preserved["boundary"] + text += ( + f"\n## Preserved\n- focus: {', '.join(preserved['focus'])}; " + f"hops: {preserved['hops']}; kept: {preserved['kept_nodes']} nodes\n" + f"- boundary dropped: {len(b['dropped_edges'])} edge(s) to " + f"{len(b['dropped_nodes'])} node(s)" + ) + print(text) + return 0 + + +def cmd_context_envelope(args) -> int: + if getattr(args, "verify", None) is not None: + return _verify_envelope(args) + if args.budget < 1: + raise SystemExit("--budget must be a positive integer") + if args.hops is not None and args.hops < 0: + raise SystemExit("--hops must be >= 0") + from ..context.envelope import build_context_envelope + + try: + graph = build_graph(repo_paths(args.root.resolve(), budget_ms=_budget_ms(args))) + except (ScanBudgetExceeded, ScanWorkloadExceeded) as exc: + return _emit_scan_budget(args, "index context-envelope", exc) + try: + env = build_context_envelope( + graph, + root=args.root.resolve(), + token_budget=args.budget, + focus=args.focus, + hops=args.hops, + bounded_output=args.bounded_output, + bounded_output_transport=( + "cli_json_stdout" if args.bounded_output and args.json else "canonical_json" + ), + ) + except FocusRejection as exc: + print( + json.dumps(exc.receipt, indent=2, sort_keys=True) + if args.json + else render_rejection(exc.receipt) + ) + return 2 + except ValueError as exc: + print(str(exc)) + return 2 + if args.json: + if args.bounded_output: + _write_bounded_json_stdout(env) + else: + print(json.dumps(env, indent=2, sort_keys=True)) + else: + print( + f"context-envelope verdict={env['verification_verdict']} " + f"tokens={env['budget']['approx_tokens']}/{env['budget']['token_budget']}" + ) + print(f"retained={len(env['retained'])} omitted={len(env['omitted'])}") + return 0 + + +def _verify_envelope(args) -> int: + from ..context.envelope import verify_envelope_freshness + + try: + envelope = json.loads(args.verify.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + print(f"could not read envelope {args.verify}: {exc}") + return 2 + try: + graph = build_graph(repo_paths(args.root.resolve(), budget_ms=_budget_ms(args))) + except (ScanBudgetExceeded, ScanWorkloadExceeded) as exc: + return _emit_scan_budget(args, "index context-envelope --verify", exc) + verdict = verify_envelope_freshness(envelope, graph) + if args.json: + print(json.dumps(verdict, indent=2, sort_keys=True)) + else: + print(f"envelope-freshness verdict={verdict['verdict']} " + f"root_ok={verdict['workspace_root_ok']}") + if verdict["drifted_repos"]: + print(f"drifted: {', '.join(verdict['drifted_repos'])}") + if verdict["missing_repos"]: + print(f"missing: {', '.join(verdict['missing_repos'])}") + return 0 if verdict["fresh"] else 1 + + +def cmd_lens(args) -> int: + if args.budget < 1: + raise SystemExit("--budget must be a positive integer") + if args.hops is not None and args.hops < 0: + raise SystemExit("--hops must be >= 0") + from ..context.lens import build_lens_pack + from ..viz.lens_html import render_lens_html + + graph = build_graph(repo_paths(args.root.resolve())) + try: + lens = build_lens_pack( + graph, + root=args.root.resolve(), + token_budget=args.budget, + focus=args.focus, + hops=args.hops, + ) + except FocusRejection as exc: + print( + json.dumps(exc.receipt, indent=2, sort_keys=True) + if args.json + else render_rejection(exc.receipt) + ) + return 2 + except ValueError as exc: + print(str(exc)) + return 2 + if args.json: + print(json.dumps(lens, indent=2, sort_keys=True)) + return 0 + out = getattr(args, "out", None) + if out: + from pathlib import Path + + Path(out).write_text(render_lens_html(lens), encoding="utf-8") + env = lens["envelope"] + print( + f"context lens -> {out} " + f"(verdict={env['verification_verdict']}, " + f"{len(env['retained'])} retained / {len(env['omitted'])} omitted " + f"at budget {env['budget']['token_budget']})" + ) + else: + print(render_lens_html(lens)) + return 0 diff --git a/client-plugin/server/src/index_graph/cli_handlers/graph.py b/client-plugin/server/src/index_graph/cli_handlers/graph.py new file mode 100644 index 0000000..b311325 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/graph.py @@ -0,0 +1,237 @@ +"""Graph-shaped handlers: graph, internals, viz.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from .. import __version__ +from ..context.focus import focus_rejection, render_rejection +from ..context.pack import closure, focus_subgraph, render_text, to_json +from ..graph.build import build_graph +from ..graph.progress import stderr_progress +from ..scan import ScanBudgetExceeded, ScanWorkloadExceeded, default_interactive_budget_ms +from ._common import head_commit, repo_paths, require_dir + + +def _budget_ms(args) -> int: + value = getattr(args, "budget_ms", None) + if value is None: + value = default_interactive_budget_ms() + if value < 0: + raise SystemExit("--budget-ms must be non-negative") + return value + + +def _emit_scan_budget(args, command: str, exc: ScanBudgetExceeded) -> int: + if getattr(args, "json", False): + payload = { + "schema": "index.scan-budget-exceeded/v1", + "command": command, + "status": "UNVERIFIABLE", + "message": str(exc), + "budget_ms": getattr(exc, "budget_ms", None), + "partial_repos": getattr(exc, "repo_count", None), + } + if hasattr(exc, "elapsed_ms"): + payload["elapsed_ms"] = exc.elapsed_ms + if hasattr(exc, "last_path"): + payload["last_path"] = exc.last_path + if hasattr(exc, "skipped"): + payload["skipped"] = exc.skipped + if hasattr(exc, "repo_limit"): + payload["repo_limit"] = exc.repo_limit + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + print(f"{command}: UNVERIFIABLE: {exc}") + return 2 + + +def cmd_graph(args) -> int: + try: + graph = build_graph(repo_paths(args.root.resolve(), budget_ms=_budget_ms(args)), + executor="process", on_progress=stderr_progress()) + except (ScanBudgetExceeded, ScanWorkloadExceeded) as exc: + return _emit_scan_budget(args, "index graph", exc) + if getattr(args, "cycles", False): + from ..graph.cycles import find_cycles + + cycles = find_cycles(graph.edges) + if args.json: + print(json.dumps({"cycles": [list(c) for c in cycles]}, indent=2)) + elif not cycles: + print("no cycles, a clean DAG") + else: + print(f"{len(cycles)} cycle(s):") + for c in cycles: + print(f" - {' -> '.join(c)} -> {c[0]}") + return 0 + if args.json: + print(json.dumps(to_json(graph), indent=2)) + else: + print(render_text(graph, "dependency graph")) + return 0 + + +def cmd_internals(args) -> int: + from ..internals import build_internals + + root = require_dir(args.root) + g = build_internals(root) + if getattr(args, "cycles", False): + return _internals_cycles(args, g) + payload = _internals_payload(g) + if args.json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + cov = ( + "complete" + if g.coverage.complete + else f"{len(g.coverage.parse_errors)} unparsed, " + f"{len(g.coverage.dynamic_imports)} dynamic" + ) + print( + f"modules={len(g.modules)} edges={len(g.edges)} " + f"cycles={len(g.cycles)} coverage={cov}" + ) + return 0 + + +def _internals_cycles(args, g) -> int: + if args.json: + print(json.dumps({"cycles": [list(c) for c in g.cycles]}, indent=2)) + elif not g.cycles: + print("no internal cycles - clean DAG") + else: + print(f"{len(g.cycles)} internal cycle(s):") + for c in g.cycles: + print(f" - {' -> '.join(c)}") + return 0 + + +def _internals_payload(g) -> dict: + return { + "repo": g.repo, + "modules": [ + {"id": m.id, "path": m.path, "language": m.language} for m in g.modules + ], + "edges": [ + { + "from": e.from_id, + "to": e.to_id, + "file": e.evidence_file, + "line": e.evidence_line, + "raw": e.raw, + } + for e in g.edges + ], + "cycles": [list(c) for c in g.cycles], + "fan_in": g.fan_in, + "fan_out": g.fan_out, + "coverage": { + "complete": g.coverage.complete, + "modules": g.coverage.modules, + "internal_edges": g.coverage.internal_edges, + "parse_errors": list(g.coverage.parse_errors), + "dynamic_imports": [ + {"file": fpath, "line": ln} for fpath, ln in g.coverage.dynamic_imports + ], + }, + } + + +def cmd_internals_symbols(args) -> int: + from ..symbols import build_symbol_graph, symbol_graph_to_payload + + root = require_dir(args.root) + g = build_symbol_graph(root) + payload = symbol_graph_to_payload(g) + if getattr(args, "coverage", False): + cov = payload["coverage"] + if args.json: + print(json.dumps(cov, indent=2, sort_keys=True)) + else: + print( + f"symbols={cov['symbols']} resolved={cov['resolved_calls']} " + f"unresolved={cov['unresolved_calls']} " + f"parse_errors={len(cov['parse_errors'])} " + f"dynamic={len(cov['dynamic_calls'])}" + ) + return 0 + if args.json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + cov = payload["coverage"] + print( + f"symbols={len(payload['symbols'])} calls={len(payload['calls'])} " + f"resolved={cov['resolved_calls']} unresolved={cov['unresolved_calls']}" + ) + return 0 + + +def cmd_viz(args) -> int: + from .. import viz + + graph = build_graph(repo_paths(args.root.resolve())) + names = {n.name for n in graph.repos} + if args.focus: + if args.focus not in names: + print(render_rejection(focus_rejection(args.focus, names))) + return 2 + graph = focus_subgraph(graph, closure(list(graph.edges), args.focus)) + pack = to_json(graph) + include_external = not args.no_external + + def _svg() -> str: + return viz.render_svg(viz.build_layout(pack, include_external=include_external)) + + def _html() -> str: + return viz.render_html( + pack, + svg=_svg(), + charts=viz.render_charts(pack, include_external=include_external), + ) + + if args.format == "all": + return _viz_all(args, viz, pack, include_external, _svg, _html) + text = { + "svg": _svg, + "mermaid": lambda: viz.render_mermaid(pack, include_external=include_external), + "html": _html, + }[args.format]() + if args.out: + Path(args.out).write_text(text, encoding="utf-8") + else: + print(text) + return 0 + + +def _viz_all(args, viz, pack, include_external, _svg, _html) -> int: + out_dir = Path(args.out_dir or ".") + out_dir.mkdir(parents=True, exist_ok=True) + files = { + "graph.mmd": viz.render_mermaid(pack, include_external=include_external).encode( + "utf-8" + ), + "graph.svg": _svg().encode("utf-8"), + "graph.html": _html().encode("utf-8"), + "context.json": json.dumps(pack, indent=2).encode("utf-8"), + } + for name, data in files.items(): + (out_dir / name).write_bytes(data) + artifacts = { + "mermaid": ("graph.mmd", files["graph.mmd"]), + "svg": ("graph.svg", files["graph.svg"]), + "html": ("graph.html", files["graph.html"]), + "context": ("context.json", files["context.json"]), + } + meta = { + "version": __version__, + "commit": head_commit(args.root.resolve()), + "root": str(args.root), + } + manifest = viz.render_manifest(pack, artifacts=artifacts, meta=meta) + (out_dir / "context-manifest.json").write_text( + json.dumps(manifest, indent=2), encoding="utf-8" + ) + return 0 diff --git a/client-plugin/server/src/index_graph/cli_handlers/lsp.py b/client-plugin/server/src/index_graph/cli_handlers/lsp.py new file mode 100644 index 0000000..e8a8c38 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/lsp.py @@ -0,0 +1,16 @@ +"""Handler for `index lsp`: start the stdio LSP server on a workspace root. + +A thin adapter over ``index_graph.lsp.LSPServer``. It binds no socket and holds +no model; it reads Content-Length-framed JSON-RPC from stdin and writes answers +to stdout, so an IDE (VSCode/Neovim/JetBrains) can consume index's verified +symbol graph natively. Tests may inject in-memory binary streams via the private +``_stdin``/``_stdout`` attributes on the args object. +""" +from __future__ import annotations + + +def cmd_lsp(args) -> int: + from ..lsp import LSPServer + + server = LSPServer(root=args.root, trace=getattr(args, "trace", "off")) + return server.serve(getattr(args, "_stdin", None), getattr(args, "_stdout", None)) diff --git a/client-plugin/server/src/index_graph/cli_handlers/maps.py b/client-plugin/server/src/index_graph/cli_handlers/maps.py new file mode 100644 index 0000000..469b678 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/maps.py @@ -0,0 +1,131 @@ +"""Knowledge-map handlers: atlas and router.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from ..cache import cached_text +from ..scan import ScanBudgetExceeded, ScanWorkloadExceeded, default_interactive_budget_ms +from ..graph.build import build_graph +from ..graph.progress import stderr_progress +from ._common import rel_to_root, repo_paths, require_dir + + +def cmd_atlas(args) -> int: + from ..knowledge.atlas import build_atlas_pack + from ..knowledge.docs import discover_docs + + root = require_dir(args.root) + paths = repo_paths(root) + repo_dirs = {name: rel_to_root(root, p) for name, p in paths.items()} + graph = build_graph(paths) + docs = discover_docs(root) + pack = build_atlas_pack(graph, docs, repo_dirs) + if args.format == "html": + return _atlas_html(args, pack, docs) + if args.json: + print(json.dumps(pack, indent=2)) + else: + print( + f"repos={len(pack['repos'])} docs={len(pack['docs'])} " + f"knowledge_edges={len(pack['knowledge_edges'])}" + ) + return 0 + + +def cmd_workbench(args) -> int: + from ..knowledge.docs import discover_docs + from ..viz.workbench_html import render_workbench_html + from ..workbench import build_workbench_pack + + if args.budget < 1: + raise SystemExit("--budget must be a positive integer") + root = require_dir(args.root) + paths = repo_paths(root) + repo_dirs = {name: rel_to_root(root, p) for name, p in paths.items()} + graph = build_graph(paths) + docs = discover_docs(root) + wb = build_workbench_pack( + graph, docs, repo_dirs, + root=root, token_budget=args.budget, spine_dir=args.spine_dir, + max_doc_bodies=args.max_doc_bodies) + if args.json: + print(json.dumps({k: v for k, v in wb.items() if k != "svg"}, + indent=2, sort_keys=True)) + return 0 + page = render_workbench_html(wb) + if args.out: + Path(args.out).write_text(page, encoding="utf-8") + s = wb["summary"] + print(f"workbench -> {args.out} ({s['repos']} repos, {s['docs']} docs, " + f"{len(wb['spine']['tools'])} spine envelopes, " + f"receipt {wb['receipt_sha256'][:16]}…)") + else: + print(page) + return 0 + + +def _atlas_html(args, pack, docs) -> int: + from .. import viz + + include_external = not args.no_external + svg = viz.render_atlas_svg( + viz.build_atlas_layout(pack, include_external=include_external) + ) + html = viz.render_atlas_html(pack, docs, svg=svg, include_external=include_external) + if args.out: + Path(args.out).write_text(html, encoding="utf-8") + print(f"wrote {args.out}") + else: + print(html) + return 0 + + +def cmd_router(args) -> int: + from ..knowledge.atlas import build_router_pack + from ..knowledge.docs import discover_router_docs + from ..router import render_router + from ..router_inventory import build_router_inventory + + root = require_dir(args.root) + max_docs = max(0, int(getattr(args, "max_docs", 500))) + budget_ms = getattr(args, "budget_ms", None) + if budget_ms is None: + budget_ms = default_interactive_budget_ms() + if budget_ms < 0: + raise SystemExit("--budget-ms must be non-negative") + + def _build() -> str: + inventory = build_router_inventory(root, budget_ms=budget_ms) + pack = build_router_pack( + build_graph( + inventory.repo_paths, + executor="process", + use_cache=not getattr(args, "no_cache", False), + on_progress=stderr_progress(), + file_lists=inventory.repo_file_lists, + ), + discover_router_docs(root, paths=inventory.router_doc_paths), + inventory.repo_dirs, + ) + return render_router(pack, max_docs=max_docs) + + try: + text = cached_text( + "router", + root, + {"max_docs": max_docs, "budget_ms": budget_ms}, + _build, + enabled=not getattr(args, "no_cache", False), + ) + except (ScanBudgetExceeded, ScanWorkloadExceeded) as exc: + print(f"index router: UNVERIFIABLE: {exc}") + print("next: increase --budget-ms, use --budget-ms 0, or run index map --resume-state PATH") + return 2 + if args.out: + Path(args.out).write_text(text, encoding="utf-8") + print(f"wrote {args.out}") + else: + print(text) + return 0 diff --git a/client-plugin/server/src/index_graph/cli_handlers/serve.py b/client-plugin/server/src/index_graph/cli_handlers/serve.py new file mode 100644 index 0000000..f5bf36a --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/serve.py @@ -0,0 +1,14 @@ +"""Handler for `index serve`: run the on-demand verified-wiki HTTP server. + +Delegates to the wiki serve module so the CLI handler stays a thin adapter and +the server logic lives with the rest of the wiki surface. Binds loopback by +default; a --host/--port override is allowed but defaults keep it local. +""" + +from __future__ import annotations + + +def cmd_serve(args) -> int: + from ..wiki.serve import serve_forever + + return serve_forever(host=args.host, port=args.port) diff --git a/client-plugin/server/src/index_graph/cli_handlers/symbols.py b/client-plugin/server/src/index_graph/cli_handlers/symbols.py new file mode 100644 index 0000000..7426113 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/symbols.py @@ -0,0 +1,106 @@ +"""Handler for `index symbols `: symbol-granular navigation. + +Go-to-definition, find-references, and find-implementations over the wave-1 +symbol graph, each hop carrying file:line evidence. The query is a symbol id +(``module::name`` or ``Class::method``) or a bare name. With no mode flag, all +three sections are reported; ``--def`` / ``--refs`` / ``--impls`` select one. + +Every row defers to the graph's own resolution, so nothing is guessed: an +unresolved reference is listed separately and never as a caller, and an +external base class yields no implementation edge. +""" +from __future__ import annotations + +import json + +from ..symbols import (build_symbol_navigator, find_definitions, + find_implementations, find_references) +from ._common import require_dir + + +def cmd_symbols(args) -> int: + root = require_dir(args.root) + query = args.query.strip() + graph, edges = build_symbol_navigator(root) + want_def, want_refs, want_impls = _modes(args) + + result: dict = {"repo": graph.repo, "query": query} + if want_def: + result["definitions"] = find_definitions(graph, query) + if want_refs: + result["references"] = find_references(graph, query) + if want_impls: + result["implementations"] = find_implementations(graph, edges, query) + + if args.json: + print(json.dumps(result, indent=2, sort_keys=True)) + return 0 + _render_text(result, want_def, want_refs, want_impls) + return _exit_code(result, want_def, want_refs, want_impls) + + +def _modes(args) -> tuple[bool, bool, bool]: + """Selected sections; no flag means all three.""" + d, r, i = args.definition, args.references, args.implementations + if not (d or r or i): + return True, True, True + return d, r, i + + +def _exit_code(result: dict, want_def, want_refs, want_impls) -> int: + """0 when the query matched something; 2 when every requested section was empty. + + A non-zero code lets scripts distinguish "no such symbol / no evidence" from + a successful lookup, without ever fabricating a match. + """ + found = False + if want_def: + found = found or bool(result["definitions"]) + if want_refs: + refs = result["references"] + found = found or bool(refs["references"]) or bool(refs["unresolved"]) + if want_impls: + impls = result["implementations"] + found = found or bool(impls["subclasses"]) or bool(impls["overrides"]) + return 0 if found else 2 + + +def _render_text(result: dict, want_def, want_refs, want_impls) -> None: + print(f"symbol query: {result['query']} (repo {result['repo']})") + if want_def: + _render_definitions(result["definitions"]) + if want_refs: + _render_references(result["references"]) + if want_impls: + _render_implementations(result["implementations"]) + + +def _render_definitions(defs: list) -> None: + print(f"definitions ({len(defs)}):") + if not defs: + print(" (none: no matching definition in this repo)") + for d in defs: + vis = "public" if d["is_public"] else "internal" + print(f" {d['id']} [{d['kind']}, {vis}] {d['file']}:{d['line']}") + + +def _render_references(refs: dict) -> None: + resolved = refs["references"] + unresolved = refs["unresolved"] + print(f"references ({len(resolved)} resolved, {len(unresolved)} unresolved):") + for r in resolved: + print(f" {r['from_symbol']} {r['file']}:{r['line']} " + f"[{r['resolution']}, {r['confidence']}]") + for u in unresolved: + print(f" ? {u['from_symbol']} {u['file']}:{u['line']} " + f"(unresolved same-name reference, never a caller)") + + +def _render_implementations(impls: dict) -> None: + subs = impls["subclasses"] + overs = impls["overrides"] + print(f"implementations ({len(subs)} subclasses, {len(overs)} overrides):") + for s in subs: + print(f" subclass {s['child']} {s['file']}:{s['line']} [{s['resolution']}]") + for o in overs: + print(f" override {o['child']} {o['file']}:{o['line']} [{o['resolution']}]") diff --git a/client-plugin/server/src/index_graph/cli_handlers/verify.py b/client-plugin/server/src/index_graph/cli_handlers/verify.py new file mode 100644 index 0000000..1f7d459 --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_handlers/verify.py @@ -0,0 +1,235 @@ +"""Grounding handlers: verify, freshness, bench, mcp.""" + +from __future__ import annotations + +import json + +from .. import __version__ +from ..cache import cached_text +from ..context.pack import to_json +from ..graph.build import build_graph +from ._common import repo_paths, require_dir + + +def cmd_watch(args) -> int: + """Auto-resync: hold the prior fingerprint, recompute on each tick, emit a + live FRESH/STALE receipt on change, and (optionally) regenerate an artifact.""" + import time as _time + + from ..freshness.watch import watch_iter + from ._common import rel_to_root + + if args.interval <= 0: + raise SystemExit("watch: --interval must be positive") + root = require_dir(args.root) + + def _paths(): + return repo_paths(root) + + regen = None + if args.regen: + regen = _make_regen(args, root, rel_to_root) + + max_ticks = args.max_ticks if args.max_ticks and args.max_ticks > 0 else None + changed = 0 + try: + for report in watch_iter(_paths(), interval=args.interval, + max_ticks=max_ticks): + if args.json: + print(json.dumps(report), flush=True) + elif report["tick"] == 0: + print(f"watching {root} — baseline {report['curr_root'][:12]}… " + f"(every {args.interval}s; Ctrl-C to stop)", flush=True) + else: + deltas = (report.get("repos_changed", []) + + ["+" + a for a in report.get("repos_added", [])] + + ["-" + r for r in report.get("repos_removed", [])]) + print(f"[tick {report['tick']}] {report['verdict']}: " + f"{', '.join(deltas) or report.get('error', '—')}", flush=True) + if regen and report["tick"] > 0 and report["verdict"] == "STALE": + out = regen() + if not args.json: + print(f" regenerated {out}", flush=True) + if report["tick"] > 0 and report["verdict"] == "STALE": + changed += 1 + except KeyboardInterrupt: + if not args.json: + print(f"\nstopped — {changed} resync(s) detected", flush=True) + return 0 + + +def _make_regen(args, root, rel_to_root): + """Return a zero-arg callable that regenerates the requested artifact and + returns its path. Reuses the same builders the one-shot subcommands use.""" + from pathlib import Path + + from ..knowledge.docs import discover_docs + + out = Path(args.out or f"index-{args.regen}.html") + + def _run(): + paths = repo_paths(root) + graph = build_graph(paths) + if args.regen == "workbench": + from ..viz.workbench_html import render_workbench_html + from ..workbench import build_workbench_pack + repo_dirs = {n: rel_to_root(root, p) for n, p in paths.items()} + wb = build_workbench_pack(graph, discover_docs(root), repo_dirs, root=root) + out.write_text(render_workbench_html(wb), encoding="utf-8") + elif args.regen == "atlas": + from .. import viz + from ..knowledge.atlas import build_atlas_pack + docs = discover_docs(root) + repo_dirs = {n: rel_to_root(root, p) for n, p in paths.items()} + pack = build_atlas_pack(graph, docs, repo_dirs) + svg = viz.render_atlas_svg(viz.build_atlas_layout(pack)) + out.write_text(viz.render_atlas_html(pack, docs, svg=svg), encoding="utf-8") + return str(out) + + return _run + + +def cmd_verify(args) -> int: + from ..verify import build_verification + + root = require_dir(args.root) + if (args.depends is None) == (args.exists is None): + raise SystemExit( + "verify: pass exactly one of --depends 'A -> B' or --exists NAME" + ) + if args.exists is not None and not args.exists.strip(): + raise SystemExit("verify: --exists NAME must be non-empty") + if args.depends: + if "->" not in args.depends: + raise SystemExit("verify: --depends must be 'A -> B'") + frm, to = (s.strip() for s in args.depends.split("->", 1)) + claim = {"kind": "depends", "from": frm, "to": to} + recheck = f'index verify --root "{args.root}" --depends "{args.depends}"' + else: + claim = {"kind": "exists", "name": args.exists.strip()} + recheck = f'index verify --root "{args.root}" --exists "{args.exists}"' + pack = to_json(build_graph(repo_paths(root))) + rec = build_verification(pack, claim, tool_version=__version__, recheck=recheck) + if args.json: + print(json.dumps(rec, indent=2, sort_keys=True)) + else: + loc = f" ({rec['evidence']})" if rec["evidence"] else "" + print(f"verdict={rec['verdict']}: {rec['detail']}{loc}") + return {"MATCH": 0, "REFUTED": 1, "UNVERIFIABLE": 2}[rec["verdict"]] + + +def cmd_freshness(args) -> int: + from ..freshness import REPORT_SCHEMA, compare_freshness, workspace_fingerprint + + root = require_dir(args.root) + try: + cert = json.loads(args.cert.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise SystemExit(f"freshness: cannot read certificate {args.cert}: {exc}") + stamp = cert.get("freshness") if isinstance(cert, dict) else None + if not stamp: + report = { + "schema": REPORT_SCHEMA, + "verdict": "UNVERIFIABLE", + "detail": "certificate carries no freshness stamp " + "(mint it with index check --freshness)", + } + else: + try: + report = compare_freshness(stamp, workspace_fingerprint(repo_paths(root))) + except ValueError as exc: + report = { + "schema": REPORT_SCHEMA, + "verdict": "UNVERIFIABLE", + "detail": str(exc), + } + report["recheck"] = f'index freshness --cert "{args.cert}" --root "{args.root}"' + return _freshness_emit(args, report) + + +def _freshness_emit(args, report) -> int: + if args.json: + print(json.dumps(report, indent=2, sort_keys=True)) + else: + line = f"verdict={report['verdict']}" + if report.get("detail"): + line += f": {report['detail']}" + print(line) + for n in report.get("repos_changed", []): + print(f" changed: {n}") + for n in report.get("repos_added", []): + print(f" added: {n}") + for n in report.get("repos_removed", []): + print(f" removed: {n}") + return {"FRESH": 0, "STALE": 1, "UNVERIFIABLE": 2}[report["verdict"]] + + +def cmd_bench(args) -> int: + root = require_dir(args.root) + text = cached_text( + "bench", + root, + {"json": bool(args.json)}, + lambda: _bench_output(args, root), + enabled=not getattr(args, "no_cache", False), + ) + print(text) + return 0 + + +def _bench_output(args, root) -> str: + from ..bench import bench_workspace + + report = bench_workspace( + repo_paths(root), + use_graph_cache=not getattr(args, "no_cache", False), + ) + report["recheck"] = f"index bench --root {args.root}" + if args.json: + return json.dumps(report, indent=2, sort_keys=True) + return _bench_human_text(report) + + +def _bench_human_text(report: dict) -> str: + st, pk = report["source_bytes"], report["pack_bytes"] + red = report["reduction"] + red_txt = f" {red}x smaller" if red else "" + lines = [ + "token economy: index's structural pack vs the source it reads", + ( + f" source read {st:>11,} bytes " + f"(~{report['approx_tokens_source']:,} tokens) " + f"{report['source_files']} files in {report['repos']} repos" + ), + ( + f" index pack {pk:>11,} bytes " + f"(~{report['approx_tokens_pack']:,} tokens){red_txt}" + ), + ] + f = report["faithfulness"] + if f["edge_grounding"] is None: + lines.append( + f" faithfulness n/a: {f['internal_edges']} internal edges " + "(nothing to ground)" + ) + else: + lines.append( + f" faithfulness {f['edge_grounding'] * 100:.0f}% of " + f"{f['internal_edges']} kept edges grounded in file:line source " + "(the reduction fabricates nothing)" + ) + lines.append( + f" note: ~{report['bytes_per_token']} bytes/token is an approximation; " + "the reduction ratio does not depend on it." + ) + lines.append( + " the pack answers structural questions (depends-on, roles, " + "cycles); reading the code is still needed for behavior." + ) + return "\n".join(lines) + + +def cmd_mcp(args) -> int: + from ..mcp import serve + + return serve() diff --git a/client-plugin/server/src/index_graph/cli_parser.py b/client-plugin/server/src/index_graph/cli_parser.py new file mode 100644 index 0000000..b133ddb --- /dev/null +++ b/client-plugin/server/src/index_graph/cli_parser.py @@ -0,0 +1,474 @@ +"""Argument-parser construction for the `index` CLI. + +``build_parser`` orchestrates per-subcommand builder helpers so no single +function exceeds the 50-line ceiling. The set of subcommands, flags, and +help text is identical to the pre-split single-file parser. +""" + +from __future__ import annotations + +import argparse +from pathlib import Path + +from . import __version__ +from .wiki.cli import add_wiki_parser + + +def _add_budget_ms_arg(p: argparse.ArgumentParser) -> None: + p.add_argument( + "--budget-ms", + type=int, + default=None, + help="repository-discovery time budget in milliseconds; 0 means unbounded", + ) + + +def _add_map_args(p: argparse.ArgumentParser) -> None: + p.add_argument("--root", type=Path, default=Path.cwd()) + p.add_argument("--output", type=Path, default=None) + p.add_argument("--json", action="store_true") + p.add_argument( + "--dry-run", + action="store_true", + help="report the write path and repo counts without writing anything", + ) + p.add_argument("--config", type=Path, default=None) + p.add_argument("--jobs", type=int, default=None) + p.add_argument( + "--resume-state", + type=Path, + default=None, + help="JSONL checkpoint file that lets complete map builds resume completed repo rows", + ) + + +def _add_telos_parser(sub, name: str, help_text: str) -> None: + p = sub.add_parser(name, help=help_text) + p.add_argument( + "--json", action="store_true", help="emit a Project Telos action envelope" + ) + + +def _add_graph_parser(sub) -> None: + g = sub.add_parser("graph", help="Derive the repo-level dependency graph.") + g.add_argument("--root", type=Path, default=Path.cwd()) + g.add_argument("--json", action="store_true") + _add_budget_ms_arg(g) + g.add_argument( + "--cycles", + action="store_true", + help="Report dependency cycles instead of the full graph.", + ) + + +def _add_context_parser(sub) -> None: + c = sub.add_parser("context", help="Render the synthesis context pack.") + c.add_argument("--root", type=Path, default=Path.cwd()) + c.add_argument("--json", action="store_true") + c.add_argument("--focus", default=None) + c.add_argument("--hops", type=int, default=None) + c.add_argument("--audit", action="store_true") + _add_budget_ms_arg(c) + + +def _add_context_envelope_parser(sub) -> None: + ce = sub.add_parser( + "context-envelope", + help="Emit a budgeted, receipt-backed context envelope.", + ) + ce.add_argument("--root", type=Path, default=Path.cwd()) + ce.add_argument( + "--budget", + type=int, + default=1200, + help="approximate token budget for retained context entries", + ) + ce.add_argument("--focus", default=None) + ce.add_argument("--hops", type=int, default=None) + ce.add_argument( + "--bounded-output", + action="store_true", + help="opt in to bounding the serialized JSON response, omitting source refs with expansion handles when needed", + ) + ce.add_argument("--json", action="store_true") + _add_budget_ms_arg(ce) + ce.add_argument( + "--verify", + type=Path, + default=None, + metavar="ENVELOPE_JSON", + help="re-derive a saved envelope's freshness against the current workspace " + "(exit 0 if still fresh, 1 if it drifted); other build flags are ignored", + ) + + +def _add_route_parser(sub) -> None: + route = sub.add_parser( + "route", + help="Build a bounded context envelope from explicit repository paths.", + ) + route.add_argument("--root", type=Path, default=Path.cwd()) + route.add_argument( + "--path", + dest="paths", + action="append", + required=True, + help="repository path under --root; repeat for multiple repositories", + ) + route.add_argument("--budget", type=int, default=1200, help="context token budget") + route.add_argument("--hops", type=int, default=None) + route.add_argument("--json", action="store_true") + + +def _add_lens_parser(sub) -> None: + ln = sub.add_parser( + "lens", + help="Render the Context Lens: a live, self-contained page showing " + "what a token budget retains and drops, with failure codes.", + ) + ln.add_argument("--root", type=Path, default=Path.cwd()) + ln.add_argument( + "--budget", + type=int, + default=1200, + help="approximate token budget the lens opens at (slider varies it live)", + ) + ln.add_argument("--focus", default=None) + ln.add_argument("--hops", type=int, default=None) + ln.add_argument("--out", default=None, help="write the HTML here instead of stdout") + ln.add_argument("--json", action="store_true", help="emit the lens pack JSON instead of HTML") + + +def _add_workbench_parser(sub) -> None: + wbp = sub.add_parser( + "workbench", + help="Render the unified workbench: map, docs, context lens, health, " + "and the flagship spine in one self-contained page.", + ) + wbp.add_argument("--root", type=Path, default=Path.cwd()) + wbp.add_argument("--budget", type=int, default=6000, + help="token budget the context lens opens at") + wbp.add_argument("--max-doc-bodies", type=int, default=200, + help="rendered doc bodies embedded in the page (page-weight budget; the doc list and search always cover all docs)") + wbp.add_argument("--spine-dir", default=None, + help="directory of captured flagship-action envelopes (*.json)") + wbp.add_argument("--out", default=None, help="write the HTML here instead of stdout") + wbp.add_argument("--json", action="store_true", + help="emit the workbench pack JSON (minus svg) instead of HTML") + + +def _add_watch_parser(sub) -> None: + w = sub.add_parser( + "watch", + help="Auto-resync on file change: hold the prior fingerprint, emit a " + "live FRESH/STALE receipt per change, optionally regenerate an artifact.", + ) + w.add_argument("--root", type=Path, default=Path.cwd()) + w.add_argument("--interval", type=float, default=2.0, + help="poll seconds (latency floor; the fingerprint is authoritative)") + w.add_argument("--max-ticks", type=int, default=0, + help="stop after N ticks (0 = run until Ctrl-C)") + w.add_argument("--regen", choices=["workbench", "atlas"], default=None, + help="regenerate this artifact on every detected change") + w.add_argument("--out", default=None, help="artifact output path for --regen") + w.add_argument("--json", action="store_true", + help="emit one freshness-sync receipt (JSON) per line") + + +def _add_select_parser(sub) -> None: + se = sub.add_parser( + "select", + help="Select files under a root; every rejection carries a typed receipt.", + ) + se.add_argument("--root", type=Path, default=Path.cwd()) + se.add_argument( + "--suffix", + dest="suffixes", + action="append", + default=None, + help="keep only files with this suffix (repeatable, e.g. --suffix .md)", + ) + se.add_argument( + "--max-files", + type=int, + default=None, + help="file budget; paths beyond it get over-budget receipts", + ) + se.add_argument("--json", action="store_true") + + +def _add_viz_parser(sub) -> None: + v = sub.add_parser("viz", help="Render the dependency graph (html/svg/mermaid).") + v.add_argument("--root", type=Path, default=Path.cwd()) + v.add_argument( + "--format", choices=["html", "svg", "mermaid", "all"], default="html" + ) + v.add_argument("--focus", default=None) + v.add_argument("--no-external", action="store_true") + v.add_argument("--out", default=None) + v.add_argument("--out-dir", default=None) + + +def _add_atlas_parser(sub) -> None: + a = sub.add_parser("atlas", help="Two-layer code + knowledge map (repos + docs).") + a.add_argument("--root", type=Path, default=Path.cwd()) + a.add_argument("--json", action="store_true") + a.add_argument("--format", choices=["html"], default=None) + a.add_argument("--out", default=None) + a.add_argument("--no-external", action="store_true") + + +def _add_internals_parser(sub) -> None: + i = sub.add_parser("internals", help="Intra-repo module dependency graph.") + i.add_argument("--root", type=Path, default=Path.cwd()) + i.add_argument("--json", action="store_true") + i.add_argument("--cycles", action="store_true") + + +def _add_internals_symbols_parser(sub) -> None: + s = sub.add_parser( + "internals-symbols", + help="Symbol-level call/reference graph for one repo (Python AST-exact " + "within-module; best-effort, honestly-labeled cross-module).", + ) + s.add_argument("--root", type=Path, default=Path.cwd()) + s.add_argument("--json", action="store_true") + s.add_argument( + "--coverage", + action="store_true", + help="Report only the coverage summary (symbols, resolved/unresolved " + "calls, parse errors, dynamic dispatch).", + ) + + +def _add_symbols_parser(sub) -> None: + s = sub.add_parser( + "symbols", + help="Navigate the symbol graph: go-to-definition, find-references, and " + "find-implementations for a symbol, each hop with file:line evidence.", + ) + s.add_argument( + "query", + help="symbol id (module::name or Class::method) or a bare name", + ) + s.add_argument("--root", type=Path, default=Path.cwd()) + s.add_argument("--json", action="store_true") + s.add_argument( + "--def", dest="definition", action="store_true", + help="Only go-to-definition (matching definitions with file:line).", + ) + s.add_argument( + "--refs", dest="references", action="store_true", + help="Only find-references (resolved callers; unresolved refs listed " + "separately, never as callers).", + ) + s.add_argument( + "--impls", dest="implementations", action="store_true", + help="Only find-implementations (in-repo subclasses of a class or " + "overrides of a method). No flag reports all three sections.", + ) + + +def _add_check_parser(sub) -> None: + ck = sub.add_parser( + "check", + help="Check structure against the declared [architecture] criterion.", + ) + ck.add_argument("--root", type=Path, default=Path.cwd()) + ck.add_argument( + "--internals", action="store_true", help="Include intra-repo module checks." + ) + ck.add_argument( + "--freshness", + action="store_true", + help="Stamp the certificate with a workspace content fingerprint.", + ) + ck.add_argument("--json", action="store_true") + ck.add_argument("--config", type=Path, default=None) + + +def _add_snapshot_parser(sub) -> None: + sn = sub.add_parser( + "snapshot", help="Write a canonical graph snapshot for drift diffing." + ) + sn.add_argument("--root", type=Path, default=Path.cwd()) + sn.add_argument("--out", type=Path, required=True) + + +def _add_drift_parser(sub) -> None: + dr = sub.add_parser("drift", help="Diff two snapshots into a drift report.") + dr.add_argument("--from", dest="from_snap", type=Path, required=True) + dr.add_argument("--to", dest="to_snap", type=Path, required=True) + dr.add_argument("--json", action="store_true") + + +def _add_router_parser(sub) -> None: + rt = sub.add_parser( + "router", + help="Emit a workspace map (CLAUDE.md/AGENTS.md) from the graph and docs.", + ) + rt.add_argument("--root", type=Path, default=Path.cwd()) + rt.add_argument("--out", default=None) + rt.add_argument("--max-docs", type=int, default=500, + help="maximum doc-to-repo edges rendered in the router markdown") + rt.add_argument("--no-cache", action="store_true", + help="disable router text and per-repository graph caches for this run") + _add_budget_ms_arg(rt) + + +def _add_verify_parser(sub) -> None: + vf = sub.add_parser( + "verify", + help="Ground a structural claim against the graph " + "(MATCH/REFUTED/UNVERIFIABLE).", + ) + vf.add_argument("--root", type=Path, default=Path.cwd()) + vf.add_argument("--depends", default=None, help="claim 'A -> B' (A depends on B)") + vf.add_argument("--exists", default=None, help="claim that repo NAME exists") + vf.add_argument("--json", action="store_true") + + +def _add_freshness_parser(sub) -> None: + fr = sub.add_parser( + "freshness", + help="Has the workspace changed since a certificate was minted? (FRESH/STALE).", + ) + fr.add_argument( + "--cert", + type=Path, + required=True, + help="A certificate JSON carrying a freshness stamp (index check --freshness).", + ) + fr.add_argument("--root", type=Path, default=Path.cwd()) + fr.add_argument("--json", action="store_true") + + +def _add_invalidate_parser(sub) -> None: + inv = sub.add_parser( + "invalidate", + help="Diff the tree against a pinned fingerprint and name " + "exactly what the changes invalidate (FRESH/STALE).", + ) + inv.add_argument("--root", type=Path, default=Path.cwd()) + inv.add_argument( + "--pin", + type=Path, + default=None, + help="A pin JSON minted earlier with --out; emits the " + "index.invalidation/1 report against it.", + ) + inv.add_argument( + "--out", + type=Path, + default=None, + help="Mint a pin of the current tree to this file.", + ) + inv.add_argument("--json", action="store_true") + + +def _add_bench_parser(sub) -> None: + bn = sub.add_parser( + "bench", + help="Token economy: index's structural pack vs reading the source " + "it distills.", + ) + bn.add_argument("--root", type=Path, default=Path.cwd()) + bn.add_argument("--json", action="store_true") + bn.add_argument("--no-cache", action="store_true", + help="disable the workspace bench filesystem cache for this run") + + +def _add_serve_parser(sub) -> None: + sv = sub.add_parser( + "serve", + help="Local http.server that derives a repo's verified wiki on demand " + "from its forge path (consent-clean; robots.txt disallows indexing).", + ) + sv.add_argument( + "--host", + default="127.0.0.1", + help="interface to bind (default 127.0.0.1, loopback only)", + ) + sv.add_argument( + "--port", + type=int, + default=8000, + help="port to bind (default 8000; 0 picks an ephemeral port)", + ) + + +def _add_lsp_parser(sub) -> None: + lsp = sub.add_parser( + "lsp", + help="Start a stdio LSP server exposing go-to-definition and " + "find-references over the symbol graph (VSCode/Neovim/JetBrains). " + "Answers are evidence-backed file:line or honestly empty; a stale " + "workspace is detected, never silently answered.", + ) + lsp.add_argument("--root", type=Path, default=Path.cwd()) + lsp.add_argument( + "--trace", + choices=["off", "messages", "verbose"], + default="off", + help="LSP trace verbosity (reserved; currently a no-op placeholder).", + ) + + +def _add_router_job_parser(sub) -> None: + parser = sub.add_parser("router-job", help="Start and recover durable local router builds.") + actions = parser.add_subparsers(dest="action", required=True) + start = actions.add_parser("start", help="Start a complete background router build.") + start.add_argument("--root", type=Path, default=Path.cwd()) + start.add_argument("--max-docs", type=int, default=500) + start.add_argument("--budget-ms", type=int, default=0) + start.add_argument("--no-cache", action="store_true") + for action in ("status", "result", "cancel", "resume"): + actions.add_parser(action).add_argument("job_id") + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + prog="index", + description="Repository inventory maps + dependency graph + context packs.", + ) + parser.add_argument( + "--version", action="version", version=f"%(prog)s {__version__}" + ) + sub = parser.add_subparsers(dest="cmd") + + _add_telos_parser(sub, "status", "emit Index's Project Telos operator-spine status") + _add_telos_parser(sub, "doctor", "check Index's operator-spine readiness") + _add_telos_parser(sub, "demo", "show Index's operator-spine demo command") + _add_map_args( + sub.add_parser("map", help="Write the repository inventory map (default).") + ) + _add_graph_parser(sub) + _add_context_parser(sub) + _add_context_envelope_parser(sub) + _add_route_parser(sub) + _add_watch_parser(sub) + _add_lens_parser(sub) + _add_select_parser(sub) + _add_viz_parser(sub) + _add_atlas_parser(sub) + _add_workbench_parser(sub) + add_wiki_parser(sub) + _add_internals_parser(sub) + _add_internals_symbols_parser(sub) + _add_symbols_parser(sub) + _add_check_parser(sub) + _add_snapshot_parser(sub) + _add_drift_parser(sub) + _add_router_parser(sub) + _add_router_job_parser(sub) + _add_verify_parser(sub) + _add_freshness_parser(sub) + _add_invalidate_parser(sub) + _add_bench_parser(sub) + _add_serve_parser(sub) + _add_lsp_parser(sub) + sub.add_parser( + "mcp", + help="Serve the MCP-shaped stdio protocol face (JSON-RPC over stdin/stdout).", + ) + return parser diff --git a/client-plugin/server/src/index_graph/client_mcp.py b/client-plugin/server/src/index_graph/client_mcp.py new file mode 100644 index 0000000..8a2ad2f --- /dev/null +++ b/client-plugin/server/src/index_graph/client_mcp.py @@ -0,0 +1,184 @@ +"""Local client profile: launch-root confinement and explicit tool allowlist. + +This is a read-only convenience boundary, not an OS sandbox. Concurrent local +filesystem mutation is outside its threat model. Full CLI/MCP remains separate. +""" +from __future__ import annotations + +import argparse +import json +import os +import stat +import sys +from pathlib import Path + +MAX_FILE = 8_000_000 +MAX_ENTRIES = 20_000 +NAME = "index" +PACKAGE = "index_graph" + + +class ClientRefusal(ValueError): + def __init__(self, detail, code="PATH_DENIED"): + super().__init__(detail) + self.code = code + + +def confined(root: Path, value: object, *, tree=False) -> Path: + if not isinstance(value, str) or not value or chr(0) in value: + raise ClientRefusal("path must be a non-empty string") + # Reject Windows network/device/ADS paths before resolving or accessing them. + text = value.replace(chr(92), "/") + if text.startswith("//") or ":" in text[2:] or (":" in text and not (len(text) > 2 and text[1:3] == ":/")): + raise ClientRefusal("network, device and alternate-stream paths are not permitted") + candidate = Path(value) + if not candidate.is_absolute(): + candidate = root / candidate + path = candidate.resolve(strict=True) + if path != root and root not in path.parents: + raise ClientRefusal("path is outside the launch workspace") + for part in [candidate, *candidate.parents]: + if part == root.parent: + break + info = part.lstat() + if stat.S_ISLNK(info.st_mode) or getattr(info, "st_file_attributes", 0) & 0x400: + raise ClientRefusal("links and reparse points are not permitted") + if path.is_file() and path.stat().st_size > MAX_FILE: + raise ClientRefusal("file exceeds local client limit") + if tree and path.is_dir(): + count = 0 + for folder, dirs, files in os.walk(path, followlinks=False): + dirs[:] = [d for d in dirs if d not in {".git", ".venv", "node_modules", "__pycache__"}] + for name in [*dirs, *files]: + count += 1 + if count > MAX_ENTRIES: + raise ClientRefusal("workspace exceeds local client entry limit") + confined(root, str(Path(folder) / name)) + return path + + +def handle(req, root, state=None): + if not isinstance(req, dict): + return {"jsonrpc": "2.0", "id": None, "error": {"code": -32600, "message": "invalid request"}} + if "id" not in req: + return None + response = {"jsonrpc": "2.0", "id": req["id"]} + method = req.get("method") + if method == "initialize": + from index_graph import __version__ + response["result"] = {"protocolVersion": "2025-06-18", "capabilities": {"tools": {}}, + "serverInfo": {"name": NAME + "-local", "version": __version__}} + elif method == "ping": + response["result"] = {} + elif method == "tools/list": + response["result"] = {"tools": definitions(state)} + elif method == "tools/call": + try: + params = req.get("params") or {} + args = params.get("arguments") or {} + name = params.get("name") + definition = next((d for d in definitions(state) if d["name"] == name), None) + if definition is None: + raise ClientRefusal("tool requires the separately configured full MCP surface", "TOOL_NOT_GRANTED") + if not isinstance(args, dict) or set(args) - set(definition["inputSchema"]["properties"]): + raise ClientRefusal("unsupported arguments cannot grant permissions", "ARGUMENTS_DENIED") + missing = set(definition["inputSchema"].get("required", [])) - set(args) + if missing: + raise ClientRefusal("missing required arguments", "ARGUMENTS_DENIED") + data = invoke(name, dict(args), root, state) + response["result"] = {"content": [{"type": "text", "text": data}], "isError": False} + except Exception as exc: # noqa: BLE001 - MCP returns typed engine errors. + response["result"] = {"content": [{"type": "text", "text": json.dumps( + {"code": exc.code if isinstance(exc, ClientRefusal) else "LOCAL_PROFILE_ERROR", + "detail": str(exc)})}], "isError": True} + else: + response["error"] = {"code": -32601, "message": "method not found"} + return response + + +def install_process_boundary(): + """Deny process and network actions in this stdlib profile, including Git.""" + def audit(event, args): + if event.startswith("socket.") or event in {"subprocess.Popen", "os.system", "os.posix_spawn", "os.posix_spawnp", + "os.exec", "os.spawn", "socket.connect", "socket.connect_ex", + "socket.bind", "socket.getaddrinfo"}: + raise PermissionError("local client profile does not grant processes or network") + sys.addaudithook(audit) + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--workspace", required=True, help="explicit local directory this client may read") + parser.add_argument("--state-directory", help="existing private directory for cache and owned jobs") + args = parser.parse_args(argv) + root_arg = Path(args.workspace).absolute() + root = confined(root_arg, str(root_arg)) + if not root.is_dir(): + parser.error("workspace must be a directory") + state = None + if args.state_directory: + from .client_state import validate_state + if not Path(args.state_directory).is_absolute(): + parser.error("state directory must be an absolute path or empty") + state = validate_state(Path(args.state_directory).absolute()) + install_process_boundary() + # The local profile does not read permission grants from the environment. + for line in sys.stdin: + if len(line) > MAX_FILE: + return 2 + try: + response = handle(json.loads(line), root, state) + except json.JSONDecodeError: + response = {"jsonrpc": "2.0", "id": None, + "error": {"code": -32700, "message": "parse error"}} + if response is not None: + sys.stdout.write(json.dumps(response) + "\n") + sys.stdout.flush() + return 0 + +def definitions(state=None): + from index_graph.mcp import _tool_defs + result = [d for d in _tool_defs() if d["name"] in { + "index.map", "index.select", "index.symbol-graph", + "index.symbol-definition", "index.symbol-references", "index.symbol-implementations"}] + for tool in result: + if tool["name"] == "index.map": + tool["inputSchema"] = {"type": "object", "properties": {"root": {"type": "string"}}, "required": ["root"]} + if state is not None: + tool["inputSchema"]["properties"]["no_cache"] = {"type": "boolean"} + else: + # Without a state directory this profile's map takes no resume_state + # and writes nothing. + tool["annotations"] = {**tool["annotations"], "readOnlyHint": True} + if state is not None: + from .mcp import annotate + from .router_job_surface import tool_definitions + result += [annotate(d) for d in tool_definitions() + if d["name"].rsplit(".", 1)[-1] in {"status", "result", "cancel"}] + return result + + +def invoke(name, args, root, state=None): + if state is not None: + from .client_state import invoke_state, validate_state + validate_state(state) + if name.startswith("index.router.job."): + return invoke_state(name, args, root, state) + path = confined(root, args["root"], tree=True) + if not path.is_dir(): + raise ClientRefusal("root must be a directory") + args["root"] = str(path) + if name == "index.map": + from index_graph import __version__ + from index_graph.config import default_config + from index_graph.scan import build_map + # No workspace-provided configuration, resume state or persistent caches. + if state is not None: + return invoke_state(name, args, root, state) + return json.dumps(build_map(path, default_config(), __version__).to_json()) + from index_graph.mcp import call_tool + return call_tool(name, args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/client-plugin/server/src/index_graph/client_state.py b/client-plugin/server/src/index_graph/client_state.py new file mode 100644 index 0000000..1a1d6b2 --- /dev/null +++ b/client-plugin/server/src/index_graph/client_state.py @@ -0,0 +1,56 @@ +"""Explicit client-owned persistence using the existing cache/map/job engines.""" +from __future__ import annotations + +import json +import os +import uuid +from pathlib import Path + +from .client_mcp import MAX_ENTRIES, ClientRefusal, confined + + +def validate_state(state: Path) -> Path: + state = confined(state, str(state), tree=True) + if not state.is_dir(): + raise ClientRefusal("state directory must exist") + # Persistence is private. A hard-linked output could mutate a file elsewhere. + count = 0 + for folder, dirs, files in os.walk(state, followlinks=False): + for name in [*dirs, *files]: + count += 1 + if count > MAX_ENTRIES: + raise ClientRefusal("state directory exceeds local client entry limit") + path = confined(state, str(Path(folder) / name)) + if path.is_file() and path.stat().st_nlink != 1: + raise ClientRefusal("state files cannot have hard links") + return state + + +def invoke_state(name, args, root, state): + if name == "index.map": + from . import __version__ + from .cache import cached_text + from .config import default_config + from .scan import build_map + for key in ("no_cache",): + if key in args and type(args[key]) is not bool: + raise ClientRefusal(f"{key} must be boolean", "ARGUMENTS_DENIED") + path = confined(root, args["root"], tree=True) + return cached_text("index.map", path, {"version": __version__}, + lambda: json.dumps(build_map(path, default_config(), __version__).to_json()), + enabled=not args.get("no_cache", False), cache_root=state / "cache") + from . import router_jobs + action = name.rsplit(".", 1)[-1] + operations = {"status": router_jobs.read_router_job_status, + "result": router_jobs.read_router_job_result, + "cancel": router_jobs.cancel_router_job} + if action not in operations: + raise ClientRefusal("starting or resuming workers requires a separate process grant", "TOOL_NOT_GRANTED") + job_id = str(uuid.UUID(args["job_id"])) + job = confined(state, str(state / "jobs" / job_id), tree=True) + request_path = confined(state, str(job / "request.json")) + request = json.loads(request_path.read_text(encoding="utf-8")) + if request.get("schema") != "index.router-job-request/v1" or request.get("job_id") != job_id: + raise ClientRefusal("invalid owned job request") + confined(root, request.get("root"), tree=True) + return json.dumps(operations[action](job_id, job_root=state / "jobs")) diff --git a/client-plugin/server/src/index_graph/config.py b/client-plugin/server/src/index_graph/config.py new file mode 100644 index 0000000..634cfa0 --- /dev/null +++ b/client-plugin/server/src/index_graph/config.py @@ -0,0 +1,157 @@ +"""Configuration: .index.toml parsing, neutral defaults, glob translation.""" + +from __future__ import annotations + +import os +import re +import sys +import tomllib +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +from .arch.criteria import ArchitectureCriteria, parse_architecture + +DEFAULT_PRUNE_DIRS = frozenset({ + ".git", ".mypy_cache", ".pytest_cache", ".ruff_cache", + "__pycache__", ".venv", "venv", "venvs", "env", + "node_modules", "site-packages", "lib64", ".tox", ".eggs", + "build", "dist", ".cache", ".playwright-mcp", + ".warden-safe-cache", ".next", ".turbo", + "target", "coverage", ".coverage", ".nyc_output", + ".parcel-cache", ".svelte-kit", ".angular", ".expo", + ".gradle", ".idea", ".vscode", ".yarn", ".pnpm-store", + ".terraform", "out", +}) +DEFAULT_MARKERS = ( + "README.md", "AGENTS.md", "CLAUDE.md", "pyproject.toml", "package.json", + "Cargo.toml", "CMakeLists.txt", "Makefile", "requirements.txt", +) +PUBLIC_HOSTS = frozenset({ + "github.com", "gitlab.com", "bitbucket.org", "codeberg.org", "git.sr.ht", +}) +_KNOWN_TOP = frozenset({"rule", "scan", "privacy", "output", "architecture"}) + + +def _default_jobs() -> int: + return min(32, (os.cpu_count() or 4) * 5) + + +def glob_to_regex(pattern: str) -> str: + """Translate a path glob to an anchored regex. + + `*` matches within a segment, `**` across segments, `/**` makes the + separator optional so `public/**` also matches `public`. + """ + out: list[str] = [] + i, n = 0, len(pattern) + while i < n: + if pattern.startswith("/**", i): + out.append("(/.*)?") + i += 3 + elif pattern.startswith("**", i): + out.append(".*") + i += 2 + elif pattern[i] == "*": + out.append("[^/]*") + i += 1 + else: + out.append(re.escape(pattern[i])) + i += 1 + return "^" + "".join(out) + "$" + + +@dataclass(frozen=True) +class Rule: + pattern: str + class_: str + regex: re.Pattern = field(init=False, compare=False, repr=False) + + def __post_init__(self) -> None: + object.__setattr__(self, "regex", re.compile(glob_to_regex(self.pattern))) + + +@dataclass(frozen=True) +class Config: + rules: tuple[Rule, ...] = () + extra_prune: frozenset[str] = frozenset() + markers: tuple[str, ...] = DEFAULT_MARKERS + jobs: int = field(default_factory=_default_jobs) + descend_into_repos: bool = False + include_root_repo: bool = False + omit_origin_classes: frozenset[str] = frozenset() + portable: bool = True + annotations: dict[str, Any] = field(default_factory=dict) + architecture: ArchitectureCriteria = field(default_factory=ArchitectureCriteria) + + @property + def prune(self) -> frozenset[str]: + return DEFAULT_PRUNE_DIRS | self.extra_prune + + +def default_config() -> Config: + return Config() + + +def load_config(path: Path | None, root: Path) -> Config: + if path is None: + candidates = (root / ".index.toml", root / ".repomap.toml") + path = next((candidate for candidate in candidates if candidate.exists()), None) + if path is None: + return default_config() + elif not path.exists(): + raise SystemExit(f"config not found: {path}") + try: + text = path.read_text(encoding="utf-8-sig", errors="replace") # tolerate BOM and legacy bytes + data = tomllib.loads(text) + except tomllib.TOMLDecodeError as exc: + raise SystemExit(f"{path}: invalid TOML: {exc}") from exc + except OSError as exc: + raise SystemExit(f"{path}: cannot read: {exc}") from exc + return _build_config(data, path) + + +def _build_config(data: dict[str, Any], path: Path) -> Config: + rules: list[Rule] = [] + for idx, item in enumerate(data.get("rule", [])): + if "pattern" not in item or "class" not in item: + raise SystemExit(f"{path}: rule[{idx}] requires 'pattern' and 'class'") + rules.append(Rule(str(item["pattern"]), str(item["class"]))) + + scan = data.get("scan", {}) + jobs = scan.get("jobs", _default_jobs()) + if not isinstance(jobs, int) or jobs < 1: + raise SystemExit(f"{path}: [scan] jobs must be a positive integer") + extra_prune = frozenset(str(d) for d in scan.get("prune", [])) + markers = tuple(scan["markers"]) if "markers" in scan else DEFAULT_MARKERS + descend_into_repos = scan.get("descend_into_repos", False) + if not isinstance(descend_into_repos, bool): + raise SystemExit(f"{path}: [scan] descend_into_repos must be a boolean") + include_root_repo = scan.get("include_root_repo", False) + if not isinstance(include_root_repo, bool): + raise SystemExit(f"{path}: [scan] include_root_repo must be a boolean") + + omit = frozenset(str(c) for c in data.get("privacy", {}).get("omit_origin_classes", [])) + + output = data.get("output", {}) + portable = bool(output.get("portable", True)) + annotations = dict(output.get("annotations", {})) + + architecture = parse_architecture(data.get("architecture", {})) + + for key in data: + if key not in _KNOWN_TOP: + print(f"{path}: warning: unknown config key '{key}'", file=sys.stderr) + + return Config( + rules=tuple(rules), + extra_prune=extra_prune, + markers=markers, + jobs=jobs, + descend_into_repos=descend_into_repos, + include_root_repo=include_root_repo, + omit_origin_classes=omit, + portable=portable, + annotations=annotations, + architecture=architecture, + ) diff --git a/client-plugin/server/src/index_graph/context/__init__.py b/client-plugin/server/src/index_graph/context/__init__.py new file mode 100644 index 0000000..a4b9e40 --- /dev/null +++ b/client-plugin/server/src/index_graph/context/__init__.py @@ -0,0 +1 @@ +"""Synthesis context-pack and context-envelope renderers.""" diff --git a/client-plugin/server/src/index_graph/context/envelope.py b/client-plugin/server/src/index_graph/context/envelope.py new file mode 100644 index 0000000..a673593 --- /dev/null +++ b/client-plugin/server/src/index_graph/context/envelope.py @@ -0,0 +1,677 @@ +"""Budgeted context envelopes for large-codebase agent work.""" +from __future__ import annotations + +import hashlib +import json +import math +from pathlib import Path + +from ..freshness import SCHEMA as FRESHNESS_SCHEMA, workspace_fingerprint +from ..graph.build import DependencyGraph +from .focus import resolve_focus +from .pack import closure, focus_subgraph, preservation, to_json + +SCHEMA = "project-telos.context-envelope/v1" +TOOL = "index.context.envelope" +BYTES_PER_TOKEN = 4 +FRESHNESS_ENVELOPE_SCHEMA = "index.context-envelope-freshness/v1" + + +def build_context_envelope( + graph: DependencyGraph, + *, + root: Path | str, + token_budget: int, + focus: str | None = None, + hops: int | None = None, + browser_evidence_refs: list[dict] | None = None, + bounded_output: bool = False, + bounded_output_transport: str = "canonical_json", + mcp_response_id: object | None = None, +) -> dict: + """Return a deterministic, receipt-backed context packet within ``token_budget``.""" + if token_budget < 1: + raise ValueError("token_budget must be positive") + source_graph = graph + preserved = None + candidate_repo_count = len(source_graph.repos) + if focus: + names = {node.name for node in graph.repos} + # an unresolvable selector raises FocusRejection (a ValueError) whose + # .receipt is the typed index.focus-rejection/v1 contract + focus = resolve_focus(focus, names) + keep = closure(list(graph.edges), focus, hops=hops) + preserved = preservation(list(graph.edges), keep, focus, hops) + graph = focus_subgraph(graph, keep) + pack = to_json(graph) + pack_hash = _sha(pack) + retained: list[dict] = [] + omitted: list[dict] = [] + approx_tokens = _base_tokens(pack) + source_refs = _source_refs(graph, Path(root).resolve()) + for repo in _ranked_repos(pack, focus): + item = _repo_item(repo, pack, source_refs.get(repo["name"], [])) + cost = _approx_tokens(item) + if approx_tokens + cost <= token_budget or not retained: + retained.append(item) + approx_tokens += cost + else: + omitted.append(_omitted(repo["name"], "budget_exceeded", cost)) + omitted.extend(_focus_omissions(source_graph, graph)) + omitted = _dedupe_omitted(omitted) + failure_codes = ["budget_exceeded"] if any( + item["reason"] == "budget_exceeded" for item in omitted + ) else [] + # if what was KEPT still exceeds the budget (a forced-first item larger + # than the whole budget, retained because dropping it would return + # nothing), that is a distinct honest fact from budget_exceeded omissions: + # the request could not be met even after omitting everything omittable. + # Name it rather than hiding it behind a min() cap on the reported cost. + if approx_tokens > token_budget: + failure_codes.append("budget_overflow") + verdict = "UNVERIFIABLE" if failure_codes else "MATCH" + fresh = _freshness(source_graph, retained) + envelope = { + "schema": SCHEMA, + "tool": TOOL, + "verification_verdict": verdict, + "failure_codes": failure_codes, + "root": str(Path(root)), + "focus": {"repo": focus, "hops": hops}, + "budget": { + "token_budget": token_budget, + # approx_tokens bounds the RETAINED selection (base + kept items): + # that is what the budget gates. It is NOT the whole emitted packet, + # which also carries omitted/preserved/freshness/selection metadata; + # packet_approx_tokens (set below) reports the full serialized cost + # so neither number is read as the other. + "approx_tokens": approx_tokens, + "bytes_per_token": BYTES_PER_TOKEN, + }, + "selection": _selection( + mode="focused" if focus else "workspace", + candidate_repo_count=candidate_repo_count, + selected_repo_count=len(graph.repos), + retained=retained, + omitted=omitted, + ), + "context_policy": { + "mode": "lossless_by_reference", + "raw_payload_policy": "source_refs_only", + "omission_policy": "explicit_failure_codes", + }, + "browser_evidence_refs": _browser_evidence_refs(browser_evidence_refs or []), + "retained": retained, + "omitted": omitted, + "preserved": preserved, + "freshness": fresh, + "receipts": [{"kind": "graph-pack", "sha256": pack_hash, "schema": "index.context/graph-pack"}], + "privacy": {"raw_source_included": False, "source_refs_only": True}, + "recheck": { + "command": "index context-envelope --json", + "graph_pack_sha256": pack_hash, + "freshness_root_sha256": fresh["workspace_root_sha256"], + }, + } + # the whole emitted packet's approximate token cost (retained content plus + # all the metadata the caller receives), re-derivable from the dict itself + packet_bytes = len(json.dumps(envelope, sort_keys=True).encode("utf-8")) + envelope["budget"]["packet_approx_tokens"] = packet_bytes // BYTES_PER_TOKEN + if bounded_output: + envelope = _apply_bounded_output( + envelope, + token_budget=token_budget, + root=Path(root), + focus=focus, + hops=hops, + transport=bounded_output_transport, + mcp_response_id=mcp_response_id, + ) + return envelope + + +def _apply_bounded_output( + envelope: dict, + *, + token_budget: int, + root: Path, + focus: str | None, + hops: int | None, + transport: str, + mcp_response_id: object | None, +) -> dict: + """Compact an envelope so the serialized tool response fits the budget. + + The default contract budgets the retained selection. This opt-in contract + budgets the emitted JSON packet itself, measured by serialized UTF-8 bytes + divided by ``BYTES_PER_TOKEN`` and rounded up. That is a deterministic + transport-size heuristic, not model-tokenizer output. + """ + if token_budget < 2: + raise ValueError("bounded_output budget too small to emit overflow receipt") + envelope["context_policy"]["output_policy"] = "bounded_packet" + envelope["budget"]["output_policy"] = "bounded_packet" + envelope["budget"]["pre_compaction_packet_approx_tokens"] = _packet_tokens( + envelope, ceiling=True, transport=transport, mcp_response_id=mcp_response_id) + envelope["source_ref_omissions"] = [] + _set_packet_measurement(envelope, transport=transport, mcp_response_id=mcp_response_id) + if envelope["budget"]["packet_approx_tokens"] <= token_budget: + return envelope + + _compact_source_refs( + envelope, + root=root, + focus=focus, + hops=hops, + token_budget=token_budget, + ) + _add_failure_code(envelope, "output_budget_exceeded") + envelope["verification_verdict"] = "UNVERIFIABLE" + _refresh_selection(envelope) + _set_packet_measurement(envelope, transport=transport, mcp_response_id=mcp_response_id) + if envelope["budget"]["packet_approx_tokens"] <= token_budget: + return envelope + + minimal = _minimal_bounded_overflow_receipt( + envelope, + root=root, + focus=focus, + hops=hops, + token_budget=token_budget, + ) + _set_packet_measurement(minimal, transport=transport, mcp_response_id=mcp_response_id) + if minimal["budget"]["packet_approx_tokens"] > token_budget: + raise ValueError("bounded_output budget too small to emit overflow receipt") + return minimal + + +def _compact_source_refs( + envelope: dict, + *, + root: Path, + focus: str | None, + hops: int | None, + token_budget: int, +) -> None: + retained_hashes = envelope.get("freshness", {}).get("retained_repo_sha256", {}) + omissions = [] + pre_compaction = envelope["budget"]["pre_compaction_packet_approx_tokens"] + for item in envelope.get("retained", []): + refs = list(item.get("source_refs") or []) + if not refs: + continue + item["source_refs"] = [] + omissions.append(_source_ref_omission( + item, + original_source_refs=refs, + included_source_refs=[], + root=root, + focus=focus, + hops=hops, + token_budget=token_budget, + pre_compaction_packet_approx_tokens=pre_compaction, + retained_repo_sha256=retained_hashes.get(item.get("name")), + )) + envelope["source_ref_omissions"] = omissions + + +def _source_ref_omission( + item: dict, + *, + original_source_refs: list[dict], + included_source_refs: list[dict], + root: Path, + focus: str | None, + hops: int | None, + token_budget: int, + pre_compaction_packet_approx_tokens: int, + retained_repo_sha256: str | None, +) -> dict: + omitted = len(original_source_refs) - len(included_source_refs) + args = { + "root": str(root), + "budget": max(token_budget, pre_compaction_packet_approx_tokens), + "bounded_output": False, + } + if focus is not None: + args["focus"] = focus + if hops is not None: + args["hops"] = hops + reissue_args = { + "root": str(root), + "budget": token_budget, + "bounded_output": True, + } + if focus is not None: + reissue_args["focus"] = focus + if hops is not None: + reissue_args["hops"] = hops + return { + "repo": item["name"], + "reason": "source_refs_omitted_to_bound_serialized_packet", + "failure_code": "output_budget_exceeded", + "total_source_refs": len(original_source_refs), + "included_source_refs": len(included_source_refs), + "omitted_source_refs": omitted, + "source_refs_sha256": _sha(original_source_refs), + "included_source_refs_sha256": _sha(included_source_refs), + "approx_tokens": _approx_tokens(original_source_refs), + "expand": {"tool": TOOL, "arguments": args}, + "reissue": {"tool": TOOL, "arguments": reissue_args}, + "provenance": { + "source_ref_schema": "project-telos.source-ref/v1", + "retained_repo_sha256": retained_repo_sha256, + }, + } + + +def _minimal_bounded_overflow_receipt( + envelope: dict, + *, + root: Path, + focus: str | None, + hops: int | None, + token_budget: int, +) -> dict: + pre_compaction = envelope["budget"]["pre_compaction_packet_approx_tokens"] + omitted_names = [item["name"] for item in envelope.get("retained", [])] + args = {"root": str(root), "budget": max(token_budget, pre_compaction), + "bounded_output": False} + if focus is not None: + args["focus"] = focus + if hops is not None: + args["hops"] = hops + preexisting_omitted = [dict(item) for item in envelope.get("omitted", [])] + retained_omitted = [ + { + "name": name, + "reason": "bounded_output_minimal_receipt", + "failure_code": "output_budget_exceeded", + "approx_tokens": 0, + "expand": {"tool": TOOL, "arguments": args}, + } + for name in omitted_names + ] + omitted = _dedupe_omitted(preexisting_omitted + retained_omitted) + omitted_failure_codes = sorted( + {"budget_overflow", "output_budget_exceeded"} + | {str(item.get("failure_code", "unknown")) for item in omitted} + ) + return { + "schema": SCHEMA, + "tool": TOOL, + "verification_verdict": "UNVERIFIABLE", + "failure_codes": ["budget_overflow", "output_budget_exceeded"], + "root": str(root), + "focus": {"repo": focus, "hops": hops}, + "budget": { + "token_budget": token_budget, + "approx_tokens": 0, + "bytes_per_token": BYTES_PER_TOKEN, + "output_policy": "bounded_packet", + "pre_compaction_packet_approx_tokens": pre_compaction, + }, + "selection": { + "mode": "focused" if focus else "workspace", + "candidate_repos": envelope.get("selection", {}).get("candidate_repos", 0), + "selected_repos": envelope.get("selection", {}).get("selected_repos", 0), + "retained_repos": 0, + "omitted_repos": len(omitted), + "retained_names": [], + "omitted_failure_codes": omitted_failure_codes, + }, + "retained": [], + "omitted": omitted, + "source_ref_omissions": envelope.get("source_ref_omissions", []), + } + + +def _refresh_selection(envelope: dict) -> None: + retained = envelope.get("retained", []) + omitted = envelope.get("omitted", []) + envelope["selection"] = _selection( + mode=envelope["selection"]["mode"], + candidate_repo_count=envelope["selection"]["candidate_repos"], + selected_repo_count=envelope["selection"]["selected_repos"], + retained=retained, + omitted=omitted, + ) + codes = set(envelope["selection"].get("omitted_failure_codes", [])) + codes.update(item["failure_code"] for item in envelope.get("source_ref_omissions", [])) + envelope["selection"]["omitted_failure_codes"] = sorted(codes) + + +def _add_failure_code(envelope: dict, code: str) -> None: + codes = envelope.setdefault("failure_codes", []) + if code not in codes: + codes.append(code) + + +def _set_packet_measurement( + envelope: dict, + *, + transport: str = "canonical_json", + mcp_response_id: object | None = None, +) -> None: + budget = envelope["budget"] + prior: tuple[int | None, int | None] = (None, None) + for _ in range(8): + serialized_bytes = _packet_bytes( + envelope, transport=transport, mcp_response_id=mcp_response_id) + approx = max(1, math.ceil(serialized_bytes / BYTES_PER_TOKEN)) + budget["packet_measurement"] = { + "scope": _transport_scope(transport), + "method": _transport_method(transport), + "unit": "serialized_utf8_json_bytes_ceiling_div_4", + "boundary": _transport_boundary(transport), + "serialized_bytes": serialized_bytes, + } + budget["packet_approx_tokens"] = approx + current = (serialized_bytes, approx) + if current == prior: + return + prior = current + + +def _packet_bytes( + envelope: dict, + *, + transport: str = "canonical_json", + mcp_response_id: object | None = None, +) -> int: + if transport == "canonical_json": + text = json.dumps(envelope, sort_keys=True) + elif transport == "cli_json_stdout": + text = json.dumps(envelope, indent=2, sort_keys=True) + "\n" + elif transport == "mcp_jsonrpc_tool_response": + tool_text = json.dumps(envelope, indent=2, sort_keys=True) + response = { + "jsonrpc": "2.0", + "id": mcp_response_id, + "result": { + "content": [{"type": "text", "text": tool_text}], + "isError": False, + }, + } + text = json.dumps(response, sort_keys=True) + else: + raise ValueError(f"unknown bounded_output transport: {transport}") + return len(text.encode("utf-8")) + + +def _transport_scope(transport: str) -> str: + if transport == "canonical_json": + return "entire_serialized_context_envelope" + if transport == "cli_json_stdout": + return "cli_json_stdout" + if transport == "mcp_jsonrpc_tool_response": + return "mcp_jsonrpc_tool_response" + raise ValueError(f"unknown bounded_output transport: {transport}") + + +def _transport_method(transport: str) -> str: + if transport == "canonical_json": + return "json.dumps(sort_keys=True).encode('utf-8')" + if transport == "cli_json_stdout": + return ( + "json.dumps(indent=2, sort_keys=True) + LF newline, " + "written as exact utf-8 bytes" + ) + if transport == "mcp_jsonrpc_tool_response": + return "json.dumps(JSON-RPC tool response, sort_keys=True).encode('utf-8')" + raise ValueError(f"unknown bounded_output transport: {transport}") + + +def _transport_boundary(transport: str) -> str: + if transport == "canonical_json": + return ( + "Approximate serialized-byte budget for the canonical Python envelope JSON; " + "not model-tokenizer output." + ) + if transport == "cli_json_stdout": + return ( + "Approximate serialized-byte budget for the emitted CLI --json stdout, " + "including pretty JSON and trailing LF newline written as exact utf-8 bytes; " + "not model-tokenizer output." + ) + if transport == "mcp_jsonrpc_tool_response": + return ( + "Approximate serialized-byte budget for the MCP JSON-RPC tool response wrapper; " + "not model-tokenizer output." + ) + raise ValueError(f"unknown bounded_output transport: {transport}") + + +def _packet_tokens( + envelope: dict, + *, + ceiling: bool, + transport: str = "canonical_json", + mcp_response_id: object | None = None, +) -> int: + raw = _packet_bytes(envelope, transport=transport, mcp_response_id=mcp_response_id) + if ceiling: + return max(1, math.ceil(raw / BYTES_PER_TOKEN)) + return max(1, raw // BYTES_PER_TOKEN) + + +def _ranked_repos(pack: dict, focus: str | None = None) -> list[dict]: + sal = pack.get("salience", {}) + return sorted( + pack.get("repos", []), + key=lambda repo: ( + repo["name"] != focus, + -sal.get(repo["name"], {}).get("in_degree", 0), + -sal.get(repo["name"], {}).get("out_degree", 0), + repo["name"], + ), + ) + + +def _repo_item(repo: dict, pack: dict, source_refs: list[dict]) -> dict: + sal = pack.get("salience", {}).get(repo["name"], {"in_degree": 0, "out_degree": 0}) + return { + "name": repo["name"], + "roles": pack.get("roles", {}).get(repo["name"], []), + "ecosystems": repo.get("ecosystems", []), + "description": repo.get("description", ""), + "salience": {"in_degree": sal.get("in_degree", 0), "out_degree": sal.get("out_degree", 0)}, + "source_refs": source_refs, + } + + +def _source_refs(graph: DependencyGraph, root: Path) -> dict[str, list[dict]]: + repo_paths = {node.name: Path(node.path) for node in graph.repos} + refs: dict[str, dict[tuple[str, int | None, str], dict]] = { + node.name: {} for node in graph.repos + } + for edge in graph.edges: + for signal in edge.signals: + if signal.evidence_file and edge.from_repo in repo_paths: + ref = _source_ref( + edge.from_repo, + repo_paths[edge.from_repo], + root, + signal.evidence_file, + signal.evidence_line, + signal.kind, + ) + refs[edge.from_repo].setdefault( + (ref["path"], ref["line"], ref["kind"]), ref) + for repo, path in repo_paths.items(): + if not refs[repo]: + fallback = _repo_ref(repo, path, root) + if fallback is not None: + refs[repo][(fallback["path"], fallback["line"], fallback["kind"])] = fallback + return { + repo: sorted(values.values(), key=lambda ref: (ref["path"], ref["line"] or 0, ref["kind"])) + for repo, values in refs.items() + } + + +def _repo_ref(repo: str, repo_path: Path, root: Path) -> dict | None: + for name in ("pyproject.toml", "package.json", "README.md", "README.rst", "README.txt"): + if (repo_path / name).is_file(): + return _source_ref(repo, repo_path, root, name, None, "repo") + return None + + +def _source_ref( + repo: str, + repo_path: Path, + root: Path, + evidence_file: str, + line: int | None, + kind: str, +) -> dict: + abs_path = (repo_path / evidence_file).resolve() + return { + "schema": "project-telos.source-ref/v1", + "repo": repo, + "repo_path": _rel(repo_path.resolve(), root), + "path": _rel(abs_path, root), + "kind": kind, + "line": line, + "sha256": _file_sha(abs_path), + "expand": { + "tool": "gather.docs", + "arguments": { + "path": _rel(abs_path, root), + "scope": TOOL, + }, + }, + } + + +def _focus_omissions(source: DependencyGraph, focused: DependencyGraph) -> list[dict]: + kept = {node.name for node in focused.repos} + return [_omitted(node.name, "outside_focus_or_budget", 0) + for node in source.repos if node.name not in kept] + + +def _omitted(name: str, reason: str, approx_tokens: int) -> dict: + return { + "name": name, + "reason": reason, + "failure_code": reason, + "approx_tokens": approx_tokens, + } + + +def _dedupe_omitted(items: list[dict]) -> list[dict]: + out: dict[str, dict] = {} + for item in items: + out.setdefault(item["name"], item) + return sorted(out.values(), key=lambda item: item["name"]) + + +def _selection( + *, + mode: str, + candidate_repo_count: int, + selected_repo_count: int, + retained: list[dict], + omitted: list[dict], +) -> dict: + return { + "mode": mode, + "candidate_repos": candidate_repo_count, + "selected_repos": selected_repo_count, + "retained_repos": len(retained), + "omitted_repos": len(omitted), + "retained_names": sorted(item["name"] for item in retained), + "omitted_failure_codes": sorted({item["failure_code"] for item in omitted}), + } + + +FRESHNESS_VERDICT_SCHEMA = "index.context-envelope-freshness-verdict/v1" + + +def verify_envelope_freshness(envelope: dict, graph: DependencyGraph) -> dict: + """Re-derive an envelope's freshness against the current workspace and return a verdict. + + The envelope binds itself to a workspace fingerprint (a root hash plus a per-repo hash + for every retained repo). This re-fingerprints the workspace from ``graph`` and confirms + those hashes still hold, so a cached envelope that went stale (the workspace changed + under it) is caught and the drifted repos are named -- the staleness failure mode of + cached context, made a check instead of a hope. MATCH when nothing moved, else DRIFT. + Re-derived from the workspace, never trusted from the envelope. Read-only.""" + fr = envelope.get("freshness") or {} + repo_paths = {node.name: Path(node.path) for node in graph.repos} + stamp = workspace_fingerprint(repo_paths) + root_ok = stamp["root"] == fr.get("workspace_root_sha256") + retained = fr.get("retained_repo_sha256") or {} + drifted = sorted(n for n, sha in retained.items() if stamp["repos"].get(n) != sha) + missing = sorted(n for n in retained if n not in stamp["repos"]) + fresh = root_ok and not drifted and not missing + return { + "schema": FRESHNESS_VERDICT_SCHEMA, + "verdict": "MATCH" if fresh else "DRIFT", + "fresh": fresh, + "workspace_root_ok": root_ok, + "drifted_repos": drifted, + "missing_repos": missing, + "expected_root_sha256": fr.get("workspace_root_sha256"), + "actual_root_sha256": stamp["root"], + "checked_repos": sorted(retained), + } + + +def _freshness(source_graph: DependencyGraph, retained: list[dict]) -> dict: + repo_paths = {node.name: Path(node.path) for node in source_graph.repos} + stamp = workspace_fingerprint(repo_paths) + retained_names = {item["name"] for item in retained} + retained_repo_sha256 = { + name: stamp["repos"][name] + for name in sorted(retained_names) + if name in stamp["repos"] + } + return { + "schema": FRESHNESS_ENVELOPE_SCHEMA, + "source_schema": FRESHNESS_SCHEMA, + "workspace_root_sha256": stamp["root"], + "repo_count": len(stamp["repos"]), + "retained_repo_sha256": retained_repo_sha256, + "recheck": { + "tool": "index.freshness", + "command": "index check --freshness; index freshness --cert CERT --root ROOT", + }, + } + + +def _browser_evidence_refs(refs: list[dict]) -> list[dict]: + allowed = ("ref", "schema", "mode", "hash", "verification", "side_effect") + return [ + {key: ref[key] for key in allowed if key in ref} + for ref in refs + if isinstance(ref, dict) + ] + + +def _base_tokens(pack: dict) -> int: + return _approx_tokens({ + "schema": SCHEMA, + "relations": len(pack.get("relations", [])), + "cycles": pack.get("cycles", []), + "warnings": pack.get("warnings", []), + }) + + +def _approx_tokens(value: object) -> int: + return max(1, len(json.dumps(value, sort_keys=True, separators=(",", ":"))) // BYTES_PER_TOKEN) + + +def _sha(value: object) -> str: + data = json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8") + return hashlib.sha256(data).hexdigest() + + +def _file_sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _rel(path: Path, root: Path) -> str: + try: + return path.relative_to(root).as_posix() + except ValueError: + return path.as_posix() diff --git a/client-plugin/server/src/index_graph/context/focus.py b/client-plugin/server/src/index_graph/context/focus.py new file mode 100644 index 0000000..9bbc619 --- /dev/null +++ b/client-plugin/server/src/index_graph/context/focus.py @@ -0,0 +1,78 @@ +"""Typed focus rejection receipts (index.focus-rejection/v1). + +A free-form focus selector that resolves to no repo fails TYPED, following +the index.path-selection/v1 precedent: the receipt names the unresolved +selector, a reason code from a closed set, and a bounded candidate list, +so a host can recover (pick a candidate, widen the scope) instead of +parsing a bare error string. +""" +from __future__ import annotations + +from collections.abc import Iterable + +REJECTION_SCHEMA = "index.focus-rejection/v1" + +# The closed set of rejection reason codes. A new code is added here, +# never invented at a call site. +REASON_CODES = frozenset({ + "unresolved-focus", # the selector matches no repo in the workspace + "empty-workspace", # there are no repos to match against at all +}) + +CANDIDATE_LIMIT = 8 + + +def focus_rejection(selector: str, names: Iterable[str], + limit: int = CANDIDATE_LIMIT) -> dict: + """Build the typed rejection receipt for an unresolvable ``selector``. + + Candidates are bounded to ``limit``: case-insensitive near matches + first, then the remaining repo names in sorted order. ``truncated`` + declares when the bound cut the list, and ``candidate_count`` always + carries the full population, so the receipt never implies the shown + candidates are all there is. + """ + pool = sorted(set(names)) + lowered = selector.lower() + near = [n for n in pool if lowered in n.lower() or n.lower() in lowered] + rest = [n for n in pool if n not in near] + candidates = (near + rest)[:max(limit, 0)] + return { + "schema": REJECTION_SCHEMA, + "selector": selector, + "reason_code": "empty-workspace" if not pool else "unresolved-focus", + "candidates": candidates, + "candidate_count": len(pool), + "truncated": len(pool) > len(candidates), + } + + +class FocusRejection(ValueError): + """An unresolvable focus selector, carrying its typed receipt. + + Subclasses ValueError so existing callers that catch ValueError keep + working; new callers read ``.receipt`` for the typed contract. + """ + + def __init__(self, receipt: dict): + super().__init__(f"unknown focus repo: {receipt['selector']}") + self.receipt = receipt + + +def resolve_focus(selector: str, names: Iterable[str]) -> str: + """Return ``selector`` when it names a repo; raise FocusRejection otherwise.""" + pool = set(names) + if selector in pool: + return selector + raise FocusRejection(focus_rejection(selector, pool)) + + +def render_rejection(receipt: dict) -> str: + """One human-readable line for the CLI faces; the JSON receipt is the contract.""" + line = f"focus rejected: {receipt['selector']!r} ({receipt['reason_code']})" + if receipt["candidates"]: + line += "; candidates: " + ", ".join(receipt["candidates"]) + hidden = receipt["candidate_count"] - len(receipt["candidates"]) + if receipt["truncated"] and hidden > 0: + line += f" (+{hidden} more)" + return line diff --git a/client-plugin/server/src/index_graph/context/lens.py b/client-plugin/server/src/index_graph/context/lens.py new file mode 100644 index 0000000..fadccf3 --- /dev/null +++ b/client-plugin/server/src/index_graph/context/lens.py @@ -0,0 +1,120 @@ +"""Lens pack: the context envelope plus the data needed to REPLAY it live. + +The envelope is a receipt; the lens makes it visible. To let the page vary the +token budget without re-running Python, the pack carries exactly what the +envelope algorithm consumes: the ranked repo order, each item's approximate +token cost, and the base cost. The page replays the same greedy rule +(`envelope.py`: retain while base + costs fits, first item always retained) +over the same numbers, so what the slider shows is what the CLI would emit, +never an approximation of it. + +Honest scope: the replay varies BUDGET only. Focus/hops change the candidate +set itself and require a re-run; the pack records the focus it was built with. +""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from ..graph.build import DependencyGraph +from .envelope import ( + _approx_tokens, + _base_tokens, + _ranked_repos, + _repo_item, + _source_refs, + build_context_envelope, +) +from .focus import resolve_focus +from .pack import closure, focus_subgraph, to_json + +SCHEMA = "project-telos.context-lens/v1" +TOOL = "index.context.lens" + + +def replay_retained(order: list[dict], base_tokens: int, token_budget: int) -> list[str]: + """The SAME greedy rule as build_context_envelope, over the pack's numbers. + This function is the contract the page's JS mirrors; the lens test holds + both against the real envelope output.""" + retained: list[str] = [] + approx = base_tokens + for item in order: + if approx + item["cost"] <= token_budget or not retained: + retained.append(item["name"]) + approx += item["cost"] + return retained + + +def replay_verdict(order: list[dict], base_tokens: int, token_budget: int) -> str: + """The verdict the envelope would emit at this budget, from the same numbers. + `order` is the focus-scoped candidate set, so any candidate NOT retained is a + budget drop (failure_code budget_exceeded) -> UNVERIFIABLE; else MATCH. The + page's JS mirrors exactly this so the pill never disagrees with the CLI.""" + kept = set(replay_retained(order, base_tokens, token_budget)) + return "UNVERIFIABLE" if len(kept) < len(order) else "MATCH" + + +def _plain(text: str) -> str: + """Reduce a repo's description (often raw README head: HTML tags, markdown + image/link syntax, badges) to a clean one-line plain string for display.""" + import re + + text = re.sub(r"<[^>]+>", " ", text) # HTML tags + text = re.sub(r"!\[[^\]]*\]\([^)]*\)", " ", text) # md images + text = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", text) # md links -> label + text = re.sub(r"[`*#>_|]+", " ", text) # md punctuation + return re.sub(r"\s+", " ", text).strip() + + +def build_lens_pack( + graph: DependencyGraph, + *, + root: Path | str, + token_budget: int, + focus: str | None = None, + hops: int | None = None, +) -> dict: + """Envelope + replay data + node metadata, one deterministic pack.""" + envelope = build_context_envelope( + graph, root=root, token_budget=token_budget, focus=focus, hops=hops) + scoped = graph + if focus: + names = {node.name for node in graph.repos} + resolved = resolve_focus(focus, names) + scoped = focus_subgraph( + graph, closure(list(graph.edges), resolved, hops=hops)) + pack = to_json(scoped) + refs = _source_refs(scoped, Path(root).resolve()) + order = [] + for repo in _ranked_repos(pack, envelope["focus"]["repo"]): + item = _repo_item(repo, pack, refs.get(repo["name"], [])) + order.append({ + "name": item["name"], + "cost": _approx_tokens(item), # cost from the FULL item (refs incl.) + "roles": item["roles"], + "description": _plain(item["description"])[:160], + "salience": item["salience"], + "source_refs": item["source_refs"], + }) + lens = { + "schema": SCHEMA, + "tool": TOOL, + "envelope": envelope, + "replay": { + "rule": "greedy-in-rank-order; retain while base+costs fit; first item always retained", + "base_tokens": _base_tokens(pack), + "order": order, + }, + "edges": [ + {"from": e["from"], "to": e["to"]} + for e in pack.get("relations", []) if not e.get("external") + ], + } + lens["receipt_sha256"] = _sha(lens) + return lens + + +def _sha(value: object) -> str: + data = json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8") + return hashlib.sha256(data).hexdigest() diff --git a/client-plugin/server/src/index_graph/context/pack.py b/client-plugin/server/src/index_graph/context/pack.py new file mode 100644 index 0000000..1ea0b9f --- /dev/null +++ b/client-plugin/server/src/index_graph/context/pack.py @@ -0,0 +1,138 @@ +"""Render a DependencyGraph as the synthesis context pack (relations+roles+prose). + +No editorializing: every line traces to a data field or an evidence record. +""" +from __future__ import annotations + +from ..graph.build import DependencyGraph, RepoNode +from ..graph.edges import Edge +from ..graph.cycles import find_cycles, cycle_edge_keys +from ..graph.roles import salience_audit, structural_salience + + +def _marker_list(node: RepoNode) -> list[str]: + out = [] + if "entry" in node.markers: + out.append("entry") + if "published" in node.markers: + out.append("published") + return out + + +def render_text(graph: DependencyGraph, title: str) -> str: + L = [f"# Context pack: {title}", ""] + L.append("## Roles (project: roles, in/out degree)") + sal = structural_salience(list(graph.edges)) + for node in sorted(graph.repos, key=lambda n: n.name): + rs = ", ".join(graph.roles.get(node.name, ())) or "(none)" + s = sal.get(node.name, {"in_degree": 0, "out_degree": 0}) + L.append(f"- {node.name}: {rs} (in={s['in_degree']} out={s['out_degree']})") + L.append("") + L.append("## Relations (A -> B: signals [confidence])") + for e in graph.edges: + if e.external: + continue + kinds = "+".join(sorted({s.kind for s in e.signals})) + L.append(f"- {e.from_repo} -> {e.to_repo}: {kinds} [{e.confidence}]") + L.append("") + L.append("## External dependencies (A -> name)") + for e in graph.edges: + if e.external: + L.append(f"- {e.from_repo} -> {e.target_name}") + L.append("") + L.append("## Inventory (all projects, extracted description)") + for node in sorted(graph.repos, key=lambda n: n.name): + eco = "/".join(node.ecosystems) or "none" + L.append(f"- {node.name} [{eco}]: {node.description}") + L.append("") + if graph.warnings: + L.append(f"## Warnings ({len(graph.warnings)})") + for w in graph.warnings: + L.append(f"- {w}") + return "\n".join(L) + + +def to_json(graph: DependencyGraph) -> dict: + sal = structural_salience(list(graph.edges)) + marked = {n.name: _marker_list(n) for n in graph.repos if _marker_list(n)} + cycles = find_cycles(graph.edges) + cyc_keys = cycle_edge_keys(graph.edges, cycles) + relations = [{ + "from": e.from_repo, "to": e.to_repo, "target_name": e.target_name, + "external": e.external, "confidence": e.confidence, + "in_cycle": (e.from_repo, e.to_repo) in cyc_keys, + "signals": [{"kind": s.kind, "file": s.evidence_file, "line": s.evidence_line, + "raw": s.raw_spec} for s in e.signals], + } for e in graph.edges] + return { + "roles": {n.name: list(graph.roles.get(n.name, ())) for n in graph.repos}, + "relations": relations, + "cycles": [list(c) for c in cycles], + "salience": sal, + "salience_audit": salience_audit(sal, marked), + "repos": [{"name": n.name, "ecosystems": list(n.ecosystems), + "description": n.description, "markers": sorted(n.markers)} + for n in graph.repos], + "warnings": list(graph.warnings), + } + + +def closure(edges: list[Edge], focus: str, hops: int | None = None) -> set[str]: + """Nodes reachable from focus (bidirectionally). With hops=N, bounded to the + N-hop neighborhood; with hops=None, the whole connected component.""" + adj: dict[str, set[str]] = {} + for e in edges: + if e.external or e.to_repo is None: + continue + adj.setdefault(e.from_repo, set()).add(e.to_repo) + adj.setdefault(e.to_repo, set()).add(e.from_repo) + seen = {focus} + frontier = {focus} + depth = 0 + while frontier and (hops is None or depth < hops): + nxt: set[str] = set() + for n in frontier: + for m in adj.get(n, ()): + if m not in seen: + seen.add(m) + nxt.add(m) + frontier = nxt + depth += 1 + return seen + + +def preservation(edges: list[Edge], keep: set[str], focus: str, + hops: int | None) -> dict: + """State what a focused pack preserves and what it drops at the boundary. + + An internal edge with exactly one endpoint inside `keep` crosses the boundary + and is not carried. Stating this is the information-bottleneck discipline: the + pack declares what structural information it guarantees to preserve and what it + knowingly drops, so a compact pack never reads as if it were complete. + """ + dropped_nodes: set[str] = set() + dropped_edges: set[str] = set() + for e in edges: + if e.external or e.to_repo is None: + continue + a_in, b_in = e.from_repo in keep, e.to_repo in keep + if a_in != b_in: + dropped_nodes.add(e.to_repo if a_in else e.from_repo) + dropped_edges.add(f"{e.from_repo} -> {e.to_repo}") + return { + "focus": [focus], + "hops": hops, + "kept_nodes": len(keep), + "boundary": { + "dropped_nodes": sorted(dropped_nodes), + "dropped_edges": sorted(dropped_edges), + }, + } + + +def focus_subgraph(graph: DependencyGraph, keep: set[str]) -> DependencyGraph: + repos = tuple(n for n in graph.repos if n.name in keep) + edges = tuple(e for e in graph.edges + if e.from_repo in keep and (e.external or e.to_repo in keep)) + roles = {k: v for k, v in graph.roles.items() if k in keep} + return DependencyGraph(repos, edges, roles, graph.warnings) diff --git a/client-plugin/server/src/index_graph/context/select.py b/client-plugin/server/src/index_graph/context/select.py new file mode 100644 index 0000000..686896e --- /dev/null +++ b/client-plugin/server/src/index_graph/context/select.py @@ -0,0 +1,251 @@ +"""Path selection with typed rejection receipts (index.path-selection/v1). + +Every path the selector considers lands in exactly one of two buckets: +selected, or rejected with a typed receipt naming a reason code from a +closed set and the rule that dropped it. Counts must reconcile +(candidates == selected + rejected), and ``reconcile_selection`` +re-derives that ledger from the lists themselves, so a silent drop, a +forged count, or an untyped rejection is machine-detectable. A selector +that cannot show what it dropped is not auditable. +""" +from __future__ import annotations + +import copy +import json +import os +from collections import Counter +from pathlib import Path + +from ..graph.walk import EXCLUDE_DIRS + +RECEIPT_SCHEMA = "index.path-selection/v1" +RESULT_SCHEMA = "index.path-selection-result/v1" +RECONCILIATION_SCHEMA = "index.path-selection-reconciliation/v1" + +# The closed set of rejection reason codes. The validator refuses anything +# outside it; a new code is added here, never invented at a call site. +REASON_CODES = frozenset({ + "excluded-by-rule", # a directory pruned by the shared EXCLUDE_DIRS rule + "suffix-mismatch", # a file whose suffix is outside the requested set + "over-budget", # a file beyond the max_files budget + "not-found", # the selection root does not exist + "unreadable", # a selected file that failed the read probe +}) + +_RECEIPT_FIELDS = ("schema", "path", "reason_code", "rule_ref") +_EXCLUDE_RULE_REF = "index_graph.graph.walk.EXCLUDE_DIRS" + + +def _receipt(path: str, reason_code: str, rule_ref: str) -> dict: + return {"schema": RECEIPT_SCHEMA, "path": path, + "reason_code": reason_code, "rule_ref": rule_ref} + + +def _result(rules: dict, selected: list[str], rejected: list[dict]) -> dict: + return { + "schema": RESULT_SCHEMA, + "rules": rules, + "selected": selected, + "rejected": rejected, + "counts": {"candidates": len(selected) + len(rejected), + "selected": len(selected), "rejected": len(rejected)}, + } + + +def _walk_candidates(root: Path): + """Yield (relative_posix_path, pruned) for every candidate under root. + + A pruned entry is a directory dropped by EXCLUDE_DIRS. It counts as one + candidate and carries one receipt, and nothing beneath it is walked, so + a buried file can never bypass the directory's receipt. + """ + for dirpath, dirnames, filenames in os.walk(root, onerror=lambda _e: None): + base = Path(dirpath) + for name in sorted(d for d in dirnames if d in EXCLUDE_DIRS): + yield (base / name).relative_to(root).as_posix(), True + dirnames[:] = sorted(d for d in dirnames if d not in EXCLUDE_DIRS) + for name in sorted(filenames): + yield (base / name).relative_to(root).as_posix(), False + + +def select_paths(root: Path | str, suffixes: tuple[str, ...] | None = None, + max_files: int | None = None) -> dict: + """Split every candidate under ``root`` into selected or rejected. + + ``suffixes`` keeps only files ending in one of the given suffixes + (None keeps every file). ``max_files`` is a budget: files beyond it + are rejected with over-budget receipts, never silently dropped. + """ + if max_files is not None and max_files < 0: + raise ValueError("max_files must be >= 0") + root = Path(root) + rules = {"suffixes": list(suffixes) if suffixes else None, + "max_files": max_files} + if not root.is_dir(): + return _result(rules, [], [_receipt(".", "not-found", "select.root")]) + selected: list[str] = [] + rejected: list[dict] = [] + for rel, pruned in _walk_candidates(root): + if pruned: + rejected.append(_receipt(rel, "excluded-by-rule", _EXCLUDE_RULE_REF)) + elif suffixes is not None and not rel.endswith(tuple(suffixes)): + rejected.append(_receipt(rel, "suffix-mismatch", + "select.suffixes:" + ",".join(suffixes))) + else: + selected.append(rel) + selected.sort() + if max_files is not None and len(selected) > max_files: + for rel in selected[max_files:]: + rejected.append(_receipt(rel, "over-budget", + f"select.max_files:{max_files}")) + selected = selected[:max_files] + rejected.sort(key=lambda receipt: receipt["path"]) + return _result(rules, selected, rejected) + + +def validate_receipt(receipt: object) -> list[str]: + """Return every reason ``receipt`` is not a valid rejection receipt. + + An empty list means valid. The check is closed-world: the exact schema + id, exactly the four typed fields, and a reason code from REASON_CODES. + """ + if not isinstance(receipt, dict): + return [f"receipt must be an object, got {type(receipt).__name__}"] + errors: list[str] = [] + unknown = sorted(set(receipt) - set(_RECEIPT_FIELDS)) + if unknown: + errors.append("unknown fields: " + ", ".join(unknown)) + for field in _RECEIPT_FIELDS: + if field not in receipt: + errors.append(f"missing field: {field}") + elif not isinstance(receipt[field], str) or not receipt[field]: + errors.append(f"{field} must be a non-empty string") + schema = receipt.get("schema") + if isinstance(schema, str) and schema and schema != RECEIPT_SCHEMA: + errors.append(f"schema must be {RECEIPT_SCHEMA}, got {schema!r}") + code = receipt.get("reason_code") + if isinstance(code, str) and code and code not in REASON_CODES: + errors.append(f"unknown reason_code: {code!r} " + f"(closed set: {', '.join(sorted(REASON_CODES))})") + return errors + + +def reconcile_selection(selection: object) -> dict: + """Re-derive the selection ledger from its own lists; DRIFT on any gap. + + MATCH is earned, never declared: the declared counts must equal the + list lengths, candidates must equal selected + rejected, every receipt + must validate against the closed reason-code set, and no path may be + booked twice across selected and rejected. + """ + if not isinstance(selection, dict) or selection.get("schema") != RESULT_SCHEMA: + return _report([{"code": "result-schema", + "detail": f"selection schema must be {RESULT_SCHEMA}"}]) + failures: list[dict] = [] + selected = selection.get("selected") or [] + rejected = selection.get("rejected") or [] + counts = selection.get("counts") or {} + if counts.get("candidates") != counts.get("selected", 0) + counts.get("rejected", 0): + failures.append({"code": "counts-mismatch", + "detail": (f"declared candidates={counts.get('candidates')} != " + f"selected={counts.get('selected')} + " + f"rejected={counts.get('rejected')}")}) + if counts.get("selected") != len(selected): + failures.append({"code": "selected-count-mismatch", + "detail": (f"declared selected={counts.get('selected')} " + f"but the list holds {len(selected)}")}) + if counts.get("rejected") != len(rejected): + failures.append({"code": "rejected-count-mismatch", + "detail": (f"declared rejected={counts.get('rejected')} but the " + f"list holds {len(rejected)} receipt(s); a rejection " + "without a receipt is a silent drop")}) + for receipt in rejected: + errors = validate_receipt(receipt) + if errors: + path = receipt.get("path", "?") if isinstance(receipt, dict) else "?" + failures.append({"code": "invalid-receipt", + "detail": f"{path}: " + "; ".join(errors)}) + booked = Counter(list(selected) + + [r.get("path") for r in rejected if isinstance(r, dict)]) + for path, times in sorted(booked.items(), key=lambda item: str(item[0])): + if times > 1 and path is not None: + failures.append({"code": "duplicate-path", + "detail": f"path booked {times} times: {path}"}) + return _report(failures) + + +def _report(failures: list[dict]) -> dict: + return {"schema": RECONCILIATION_SCHEMA, + "verdict": "MATCH" if not failures else "DRIFT", + "failures": failures} + + +def reject_selected(selection: dict, path: str, reason_code: str, + rule_ref: str) -> dict: + """Move ``path`` from selected to rejected, leaving a typed receipt. + + Refuses reason codes outside the closed set and paths that are not + currently selected, so a caller cannot mint receipts for paths the + selection never held. Candidates are conserved. + """ + if reason_code not in REASON_CODES: + raise ValueError(f"unknown reason_code: {reason_code!r}") + if path not in selection["selected"]: + raise ValueError(f"path is not selected: {path!r}") + selection["selected"].remove(path) + receipt = _receipt(path, reason_code, rule_ref) + selection["rejected"].append(receipt) + selection["counts"]["selected"] -= 1 + selection["counts"]["rejected"] += 1 + return receipt + + +def _read_probe(path: Path) -> None: + """Raise OSError if ``path`` cannot actually be opened and read.""" + with path.open("rb") as handle: + handle.read(1) + + +def probe_readable(selection: dict, root: Path | str) -> dict: + """Return a copy of ``selection`` where files failing the read probe + have moved to rejected with unreadable receipts. Probing reclassifies; + it never drops, so the candidate count is conserved.""" + root = Path(root) + probed = copy.deepcopy(selection) + for rel in list(probed["selected"]): + try: + _read_probe(root / rel) + except OSError: + reject_selected(probed, rel, "unreadable", "select.read_check") + return probed + + +def run_select(root: Path | str, suffixes: tuple[str, ...] | None, + max_files: int | None) -> dict: + """Select, probe readability, reconcile: the shared CLI/MCP payload.""" + selection = probe_readable( + select_paths(root, suffixes=suffixes, max_files=max_files), root) + return {"selection": selection, + "reconciliation": reconcile_selection(selection)} + + +def cmd_select(args) -> int: + """CLI face: print the selection and its reconciliation report.""" + if args.max_files is not None and args.max_files < 0: + raise SystemExit("--max-files must be >= 0") + suffixes = tuple(args.suffixes) if args.suffixes else None + payload = run_select(args.root.resolve(), suffixes, args.max_files) + selection, report = payload["selection"], payload["reconciliation"] + if args.json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + counts = selection["counts"] + print(f"select verdict={report['verdict']} " + f"candidates={counts['candidates']} " + f"selected={counts['selected']} rejected={counts['rejected']}") + reasons = Counter(r["reason_code"] for r in selection["rejected"]) + for code, count in sorted(reasons.items()): + print(f" {code}: {count}") + for failure in report["failures"]: + print(f" [{failure['code']}] {failure['detail']}") + return 0 if report["verdict"] == "MATCH" else 1 diff --git a/client-plugin/server/src/index_graph/drift/__init__.py b/client-plugin/server/src/index_graph/drift/__init__.py new file mode 100644 index 0000000..d7761b1 --- /dev/null +++ b/client-plugin/server/src/index_graph/drift/__init__.py @@ -0,0 +1,7 @@ +"""Snapshots over time and the drift between them.""" +from __future__ import annotations + +from .snapshot import snapshot_pack, dumps_canonical, load_snapshot +from .diff import DriftReport, diff_snapshots + +__all__ = ["snapshot_pack", "dumps_canonical", "load_snapshot", "DriftReport", "diff_snapshots"] diff --git a/client-plugin/server/src/index_graph/drift/diff.py b/client-plugin/server/src/index_graph/drift/diff.py new file mode 100644 index 0000000..3458863 --- /dev/null +++ b/client-plugin/server/src/index_graph/drift/diff.py @@ -0,0 +1,65 @@ +"""Diff two snapshots into a DriftReport with a MATCH/DRIFT verdict.""" +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class DriftReport: + repos_added: tuple[str, ...] + repos_removed: tuple[str, ...] + edges_added: tuple[str, ...] + edges_removed: tuple[str, ...] + cycles_introduced: tuple[tuple[str, ...], ...] + cycles_cleared: tuple[tuple[str, ...], ...] + roles_changed: tuple[tuple[str, str, str], ...] + + @property + def verdict(self) -> str: + changed = any([ + self.repos_added, self.repos_removed, self.edges_added, + self.edges_removed, self.cycles_introduced, self.cycles_cleared, + self.roles_changed, + ]) + return "DRIFT" if changed else "MATCH" + + def to_json(self) -> dict: + return { + "verdict": self.verdict, + "repos_added": list(self.repos_added), + "repos_removed": list(self.repos_removed), + "edges_added": list(self.edges_added), + "edges_removed": list(self.edges_removed), + "cycles_introduced": [list(c) for c in self.cycles_introduced], + "cycles_cleared": [list(c) for c in self.cycles_cleared], + "roles_changed": [list(t) for t in self.roles_changed], + } + + +def _cycle_set(snap: dict) -> set[tuple[str, ...]]: + # a cycle is an unordered node set; normalize so diffs are order-insensitive + return {tuple(sorted(c)) for c in snap.get("cycles", [])} + + +def diff_snapshots(old: dict, new: dict) -> DriftReport: + for snap in (old, new): + if not isinstance(snap, dict) or snap.get("schema") != "index.snapshot/1": + raise ValueError("not an index.snapshot/1 document; refusing to diff") + o_repos, n_repos = set(old.get("repos", [])), set(new.get("repos", [])) + o_edges, n_edges = set(old.get("edges", [])), set(new.get("edges", [])) + o_cyc, n_cyc = _cycle_set(old), _cycle_set(new) + o_roles, n_roles = old.get("roles", {}), new.get("roles", {}) + roles_changed: list[tuple[str, str, str]] = [] + for repo in sorted(set(o_roles) & set(n_roles)): + a, b = o_roles.get(repo, []), n_roles.get(repo, []) + if a != b: + roles_changed.append((repo, ",".join(a), ",".join(b))) + return DriftReport( + repos_added=tuple(sorted(n_repos - o_repos)), + repos_removed=tuple(sorted(o_repos - n_repos)), + edges_added=tuple(sorted(n_edges - o_edges)), + edges_removed=tuple(sorted(o_edges - n_edges)), + cycles_introduced=tuple(sorted(n_cyc - o_cyc)), + cycles_cleared=tuple(sorted(o_cyc - n_cyc)), + roles_changed=tuple(roles_changed), + ) diff --git a/client-plugin/server/src/index_graph/drift/snapshot.py b/client-plugin/server/src/index_graph/drift/snapshot.py new file mode 100644 index 0000000..d685af8 --- /dev/null +++ b/client-plugin/server/src/index_graph/drift/snapshot.py @@ -0,0 +1,30 @@ +"""Canonical, minimal, sorted projection of a context pack for drift diffing.""" +from __future__ import annotations + +import json + + +def snapshot_pack(pack: dict) -> dict: + edges = sorted( + f"{r['from']} -> {r['to']}" + for r in pack.get("relations", []) + if not r.get("external") and r.get("to") + ) + roles = {k: list(v) for k, v in sorted(pack.get("roles", {}).items())} + cycles = sorted(tuple(sorted(c)) for c in pack.get("cycles", [])) + repos = sorted(roles.keys()) + return { + "schema": "index.snapshot/1", + "repos": repos, + "edges": edges, + "roles": roles, + "cycles": [list(c) for c in cycles], + } + + +def dumps_canonical(obj: dict) -> str: + return json.dumps(obj, sort_keys=True, separators=(",", ":")) + + +def load_snapshot(text: str) -> dict: + return json.loads(text) diff --git a/client-plugin/server/src/index_graph/flagship.py b/client-plugin/server/src/index_graph/flagship.py new file mode 100644 index 0000000..99712a4 --- /dev/null +++ b/client-plugin/server/src/index_graph/flagship.py @@ -0,0 +1,160 @@ +from __future__ import annotations + +import json +import tempfile +import time +from pathlib import Path +from typing import Any + +from index_graph import __version__ + +SCHEMA = "project-telos.flagship-action/v1" +TOOL = "index" +TELOS_CONTRACTS = { + "host_surfaces": ["CLI JSON", "MCP stdio", "plugins", "IDEs", "TUIs", "apps"], + "schemas": [ + "project-telos.flagship-action/v1", + "project-telos.context-envelope/v1", + "project-telos.action-receipt/v1", + ], + "workflow_domains": ["enterprise", "research", "creative", "scientific", "education"], + "second_brain_role": ( + "map codebases, assets, docs, and dormant engine parts into compact reusable context " + "with selection summaries, freshness roots, and source-ref expansion handles" + ), + "privacy_boundary": "hosts receive receipts, hashes, redacted refs, and verdicts; raw private payloads stay in local adapters", +} + + +def envelope(command: str, *, status: str = "MATCH", native: dict | None = None, + next_actions: list[dict] | None = None, + diagnostics: list[dict] | None = None) -> dict: + return { + "schema": SCHEMA, + "tool": TOOL, + "tool_version": __version__, + "command": command, + "status": status, + "inputs": [], + "outputs": [], + "receipts": [], + "native": native or {}, + "next_actions": next_actions or [], + "diagnostics": diagnostics or [], + } + + +def _next(tool: str, action: str, reason: str) -> dict: + return {"tool": tool, "action": action, "reason": reason, "inputs": [], "priority": "normal"} + + +def _mcp_map_probe() -> dict[str, Any]: + start = time.perf_counter() + try: + with tempfile.TemporaryDirectory(prefix="index-doctor-") as tmp: + root = Path(tmp) + repo = root / "solo" + repo.mkdir() + (repo / ".git").mkdir() + (repo / "README.md").write_text("# Solo\n", encoding="utf-8") + + from .mcp import call_tool + + payload = json.loads(call_tool("index.map", {"root": str(root)})) + except Exception as exc: + return { + "name": "mcp_map_probe", + "status": "DRIFT", + "error": type(exc).__name__, + "elapsed_ms": round((time.perf_counter() - start) * 1000), + "side_effect": "temporary_workspace", + } + + status = ( + "MATCH" + if payload.get("repo_count") == 1 + and payload.get("absolute_paths_included") is False + else "DRIFT" + ) + return { + "name": "mcp_map_probe", + "status": status, + "repo_count": payload.get("repo_count"), + "metadata_status": payload.get("metadata_status"), + "metadata_unknown_count": payload.get("metadata_unknown_count"), + "absolute_paths_included": payload.get("absolute_paths_included"), + "elapsed_ms": round((time.perf_counter() - start) * 1000), + "side_effect": "temporary_workspace", + } + + +def status_payload() -> dict: + from .mcp import _tool_defs + + return envelope( + "status", + native={ + "role": "structure-context", + "commands": ["map", "graph", "context", "context-envelope", "route", "select", + "atlas", "wiki", "serve", "verify", "invalidate", "router", "router-job"], + "operator_commands": ["status", "doctor", "demo", "mcp"], + "mcp_tools": [tool["name"] for tool in _tool_defs()], + "current_status": ( + f"{__version__} workspace atlas, certificates, freshness, benchmarking, " + "selection-aware context envelopes, and MCP parity" + ), + "telos_contracts": TELOS_CONTRACTS, + }, + next_actions=[_next("gather", "docs", "gather docs backing structural decisions")], + ) + + +def doctor_payload() -> dict: + checks: list[dict[str, Any]] = [ + {"name": "workspace_map", "status": "MATCH"}, + {"name": "context_pack", "status": "MATCH"}, + {"name": "structural_verification", "status": "MATCH"}, + _mcp_map_probe(), + ] + status = "MATCH" if all(check["status"] == "MATCH" for check in checks) else "DRIFT" + diagnostics = [] if status == "MATCH" else [{ + "code": "mcp_map_probe_drift", + "message": "MCP map probe failed on a bounded temporary workspace", + }] + return envelope( + "doctor", + status=status, + native={"checks": checks, "filesystem_writes_performed": True}, + next_actions=[_next("forum", "route", "route the next workspace action")], + diagnostics=diagnostics, + ) + + +def demo_payload() -> dict: + return envelope( + "demo", + native={"command": "index map --root --json"}, + next_actions=[_next("telos", "workflow", "render workspace structure into the shared room")], + ) + + +def emit(payload: dict, as_json: bool) -> int: + if as_json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + print(f"status={payload['status']} tool={payload['tool']} command={payload['command']}") + for action in payload["next_actions"]: + print(f"next: {action['tool']} {action['action']} - {action['reason']}") + return 0 + + +def cmd_status(args) -> int: + return emit(status_payload(), args.json) + + +def cmd_doctor(args) -> int: + return emit(doctor_payload(), args.json) + + +def cmd_demo(args) -> int: + return emit(demo_payload(), args.json) diff --git a/client-plugin/server/src/index_graph/freshness/__init__.py b/client-plugin/server/src/index_graph/freshness/__init__.py new file mode 100644 index 0000000..3142034 --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/__init__.py @@ -0,0 +1,8 @@ +"""Content fingerprints and the freshness comparison (the 'has ground truth moved?' check).""" +from .compare import REPORT_SCHEMA, compare_freshness +from .fingerprint import SCHEMA, repo_fingerprint, workspace_fingerprint + +__all__ = [ + "SCHEMA", "REPORT_SCHEMA", + "repo_fingerprint", "workspace_fingerprint", "compare_freshness", +] diff --git a/client-plugin/server/src/index_graph/freshness/compare.py b/client-plugin/server/src/index_graph/freshness/compare.py new file mode 100644 index 0000000..1b57a5a --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/compare.py @@ -0,0 +1,33 @@ +"""Compare a recorded freshness stamp against the live workspace fingerprint.""" +from __future__ import annotations + +from .fingerprint import SCHEMA + +REPORT_SCHEMA = "index.freshness-report/1" + + +def compare_freshness(stamp: dict, current: dict) -> dict: + """FRESH if every repo fingerprint matches; STALE with named deltas otherwise. + + Pure (no I/O), so the report is re-checkable: a consumer recomputes the + current fingerprint and runs this again. Raises ValueError if either side + is not an index.freshness/1 document. + """ + for doc, who in ((stamp, "stamp"), (current, "current")): + if not isinstance(doc, dict) or doc.get("schema") != SCHEMA: + raise ValueError(f"{who} is not an {SCHEMA} document") + s_repos = stamp.get("repos", {}) + c_repos = current.get("repos", {}) + added = sorted(set(c_repos) - set(s_repos)) + removed = sorted(set(s_repos) - set(c_repos)) + changed = sorted(n for n in (set(s_repos) & set(c_repos)) if s_repos[n] != c_repos[n]) + fresh = not (added or removed or changed) + return { + "schema": REPORT_SCHEMA, + "verdict": "FRESH" if fresh else "STALE", + "stamp_root": stamp.get("root"), + "current_root": current.get("root"), + "repos_added": added, + "repos_removed": removed, + "repos_changed": changed, + } diff --git a/client-plugin/server/src/index_graph/freshness/fingerprint.py b/client-plugin/server/src/index_graph/freshness/fingerprint.py new file mode 100644 index 0000000..019bef5 --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/fingerprint.py @@ -0,0 +1,131 @@ +"""Deterministic content fingerprints: detect when graph-relevant content changed. + +A repo fingerprint is a SHA-256 over the sorted relative path and content SHA-256 +for every file a resolver could read (manifests and sources across all nine +ecosystems), so identical content yields an identical fingerprint on any machine, +and any graph-relevant edit, addition, or removal changes it. Working bytes are +always read: Git index flags and clean filters can hide resolver-visible edits. +The workspace fingerprint folds the per-repo fingerprints under their names. + +It is conservative on purpose. It may report a change to a file that does not +alter the resolved graph (so STALE can be a false alarm), but it never misses +a change to a file that could (so FRESH is never a false assurance). The set +of relevant files is declared by the resolvers themselves +(`fingerprint_names`, `fingerprint_suffixes`, `fingerprint_globs`), so a new +resolver is covered without touching this module. +""" +from __future__ import annotations + +import hashlib +import fnmatch +from collections.abc import Iterator +from pathlib import Path +from time import perf_counter + +from ..graph.resolvers import ALL_RESOLVERS +from ..graph.walk import read_source_bytes, walk_files + +SCHEMA = "index.freshness/1" + + +def _matchers(resolvers): + names: set[str] = set() + suffixes: set[str] = set() + globs: set[str] = set() + for r in resolvers: + names |= set(getattr(r, "fingerprint_names", ())) + suffixes |= set(getattr(r, "fingerprint_suffixes", ())) + globs |= set(getattr(r, "fingerprint_globs", ())) + return names, tuple(sorted(suffixes)), tuple(sorted(globs)) + + +def _is_relevant(filename: str, names, suffixes, globs) -> bool: + if filename in names: + return True + if suffixes and filename.endswith(suffixes): + return True + # fnmatchcase, not fnmatch: fnmatch lowercases via os.path.normcase on + # Windows, which would make the fingerprint platform-dependent. + return any(fnmatch.fnmatchcase(filename, g) for g in globs) + + +def _add_stat(stats: dict[str, int] | None, key: str, value: int) -> None: + if stats is not None: + stats[key] = int(stats.get(key, 0)) + int(value) + + +def _add_timing(stats: dict[str, int] | None, key: str, started: float) -> None: + _add_stat(stats, key, int((perf_counter() - started) * 1000)) + + +def relevant_files(repo_root: Path, resolvers=ALL_RESOLVERS, *, checkpoint=None, + stop_at_nested_repos: bool = False) -> Iterator[Path]: + """Yield every graph-relevant file under repo_root (the manifests and source + suffixes the resolvers read, across all ecosystems), pruning EXCLUDE_DIRS. + Fail-closed: a missing or unreadable tree yields nothing rather than raising. + When `stop_at_nested_repos` is true, child repositories are excluded so a + parent repo fingerprint matches the graph builder's repository boundary. + """ + names, suffixes, globs = _matchers(resolvers) + yield from walk_files( + Path(repo_root), + names=tuple(sorted(names)) or None, + suffixes=suffixes or None, + globs=globs or None, + checkpoint=checkpoint, + stop_at_nested_repos=stop_at_nested_repos, + ) + + +def repo_fingerprint(repo_root: Path, resolvers=ALL_RESOLVERS, *, stats: dict[str, int] | None = None) -> str: + """A SHA-256 over the sorted (relpath, file-sha256) of every relevant file. + + Fail-closed: an unreadable file contributes a fixed marker rather than + raising, and a missing or unreadable tree yields the empty-set hash. + """ + root = Path(repo_root) + entries = [] + started = perf_counter() + files = list(relevant_files(root, resolvers, stop_at_nested_repos=True)) + _add_timing(stats, "fingerprint_walk_ms", started) + _add_stat(stats, "fingerprint_files", len(files)) + started = perf_counter() + byte_count = 0 + unreadable = 0 + for p in files: + try: + data = read_source_bytes(p) + byte_count += len(data) + digest = hashlib.sha256(data).hexdigest() + except OSError: + unreadable += 1 + digest = "unreadable" + try: + rel = p.relative_to(root).as_posix() + except ValueError: + rel = p.as_posix() + entries.append((rel, digest)) + _add_stat(stats, "fingerprint_bytes", byte_count) + _add_stat(stats, "fingerprint_unreadable", unreadable) + _add_timing(stats, "fingerprint_read_hash_ms", started) + entries.sort() + h = hashlib.sha256() + for rel, digest in entries: + h.update(rel.encode("utf-8")) + h.update(b"\0") + h.update(digest.encode("ascii")) + h.update(b"\n") + return h.hexdigest() + + +def workspace_fingerprint(repo_paths: dict[str, Path], resolvers=ALL_RESOLVERS) -> dict: + """An index.freshness/1 stamp: a per-repo fingerprint map plus a root fold.""" + repos = {name: repo_fingerprint(root, resolvers) + for name, root in sorted(repo_paths.items())} + h = hashlib.sha256() + for name in sorted(repos): + h.update(name.encode("utf-8")) + h.update(b"\0") + h.update(repos[name].encode("ascii")) + h.update(b"\n") + return {"schema": SCHEMA, "root": h.hexdigest(), "repos": repos} diff --git a/client-plugin/server/src/index_graph/freshness/invalidate.py b/client-plugin/server/src/index_graph/freshness/invalidate.py new file mode 100644 index 0000000..8c321af --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/invalidate.py @@ -0,0 +1,273 @@ +"""Typed invalidation reports: the diff that names what it invalidates. + +`index freshness` says THAT the workspace moved; `index.invalidation/1` +says WHAT that movement invalidates. A pin (`index.invalidation-pin/1`) +records per-file hashes plus the structural snapshot at mint time. +Comparing it against the current tree lands every fingerprinted artifact +or scope in exactly one of two buckets: invalidated, with a reason code +from a closed set, or still valid. Counts must reconcile (invalidated + +still_valid == fingerprinted scope), and `reconcile_invalidation` +re-derives that ledger from the report itself, so a forged count, an +unknown reason code, or a double-booked scope is machine-detectable. +""" +from __future__ import annotations + +import hashlib +from pathlib import Path + +from ..certify import canonical_sha +from .fingerprint import relevant_files + +PIN_SCHEMA = "index.invalidation-pin/1" +REPORT_SCHEMA = "index.invalidation/1" +RECONCILIATION_SCHEMA = "index.invalidation-reconciliation/1" + +# The closed set of invalidation reason codes. A new code is added here, +# never invented at a call site. +REASON_CODES = frozenset({ + "file-changed", # a pinned graph-relevant file's content moved + "file-removed", # a pinned graph-relevant file is gone + "dependency-edge-changed", # the structural snapshot (edges/roles/cycles) moved + "doc-changed", # a pinned doc the context pack reads moved + "unversioned", # content now in scope that the pin never versioned +}) + +# The derived artifacts index fingerprints, invalidated as workspace-level scopes. +ARTIFACTS = ("certificate", "context-pack", "graph-snapshot") + +# The docs the context pack reads (repo descriptions come from these). +DOC_NAMES = ("README.md", "README.rst", "README.txt", "readme.md") + +EVIDENCE_LIMIT = 5 + +# Most severe first: aggregation picks the first code its inputs contain. +_REASON_PRIORITY = ("file-removed", "file-changed", "dependency-edge-changed", + "doc-changed", "unversioned") + + +def _hash_file(path: Path) -> str: + try: + return hashlib.sha256(path.read_bytes()).hexdigest() + except OSError: + return "unreadable" + + +def _repo_state(repo_root: Path) -> dict: + """Per-file hashes for one repo: the graph-relevant files the resolvers + read, and the root docs the context pack reads. Fail-closed: a missing + tree yields empty maps rather than raising.""" + root = Path(repo_root) + graph_files: dict[str, str] = {} + for p in relevant_files(root): + try: + rel = p.relative_to(root).as_posix() + except ValueError: + rel = p.as_posix() + graph_files[rel] = _hash_file(p) + try: + entries = {p.name for p in root.iterdir() if p.is_file()} + except OSError: + entries = set() + doc_files = {name: _hash_file(root / name) for name in DOC_NAMES if name in entries} + return {"graph_files": graph_files, "doc_files": doc_files} + + +def _pin_state(repo_paths: dict[str, Path], snapshot: dict) -> dict: + return {"repos": {name: _repo_state(path) + for name, path in sorted(repo_paths.items())}, + "snapshot": snapshot} + + +def _workspace_inputs(root: Path) -> tuple[dict[str, Path], dict]: + from ..config import load_config + from ..context.pack import to_json + from ..drift import snapshot_pack + from ..graph.build import build_graph + from ..scan import discover_repos, repo_key_map + config = load_config(None, Path(root)) + repo_paths = repo_key_map( + Path(root), + discover_repos(Path(root), config), + include_root_repo=config.include_root_repo, + ) + return repo_paths, snapshot_pack(to_json(build_graph(repo_paths))) + + +def mint_pin(root: Path | str) -> dict: + """Pin the current tree: per-file hashes plus the structural snapshot, + content-addressed by ``pinned_ref`` (the canonical SHA-256 of the state, + per docs/PROTOCOL.md hashing).""" + from .. import __version__ + state = _pin_state(*_workspace_inputs(Path(root))) + return {"schema": PIN_SCHEMA, "tool_version": __version__, + "pinned_ref": canonical_sha(state), **state} + + +def _split(pinned: dict, current: dict) -> tuple[list[str], list[str], list[str]]: + """(removed, changed, added) keys between two {path: sha} maps.""" + removed = sorted(set(pinned) - set(current)) + added = sorted(set(current) - set(pinned)) + changed = sorted(k for k in set(pinned) & set(current) if pinned[k] != current[k]) + return removed, changed, added + + +def _repo_reason(pinned: dict, current: dict) -> tuple[str | None, list[str]]: + """The single most severe reason a pinned repo scope moved, with evidence.""" + removed, changed, added = _split(pinned.get("graph_files", {}), + current.get("graph_files", {})) + doc_removed, doc_changed, doc_added = _split(pinned.get("doc_files", {}), + current.get("doc_files", {})) + if removed: + return "file-removed", removed[:EVIDENCE_LIMIT] + if changed: + return "file-changed", changed[:EVIDENCE_LIMIT] + if added: + return "unversioned", added[:EVIDENCE_LIMIT] + docs = sorted(set(doc_removed) | set(doc_changed) | set(doc_added)) + if docs: + return "doc-changed", docs[:EVIDENCE_LIMIT] + return None, [] + + +def _snapshot_reason(pinned: dict, current: dict) -> tuple[str | None, list[str]]: + """Why the structural snapshot no longer holds, or None when it does.""" + old_edges, new_edges = set(pinned.get("edges", [])), set(current.get("edges", [])) + if old_edges != new_edges: + diff = sorted(f"- {e}" for e in old_edges - new_edges) + diff += sorted(f"+ {e}" for e in new_edges - old_edges) + return "dependency-edge-changed", diff[:EVIDENCE_LIMIT] + old_repos, new_repos = set(pinned.get("repos", [])), set(current.get("repos", [])) + if old_repos - new_repos: + return "file-removed", sorted(old_repos - new_repos)[:EVIDENCE_LIMIT] + if new_repos - old_repos: + return "unversioned", sorted(new_repos - old_repos)[:EVIDENCE_LIMIT] + if pinned != current: # roles or cycles moved with identical edge strings + return "dependency-edge-changed", [] + return None, [] + + +def _aggregate(reasons: dict[str, list[str]]) -> tuple[str, list[str]]: + """Fold contributing reasons into the single most severe one.""" + for code in _REASON_PRIORITY: + if code in reasons: + return code, list(dict.fromkeys(reasons[code]))[:EVIDENCE_LIMIT] + raise ValueError("aggregate called with no reasons") # pragma: no cover + + +def _item(scope: str, reason: str, evidence: list[str]) -> dict: + return {"artifact_or_scope": scope, "reason_code": reason, "evidence": evidence} + + +def invalidation_report(pin: dict, root: Path | str, *, recheck: str | None = None) -> dict: + """Diff the pinned state against the current tree and name what it invalidates. + + A tampered pin hash simply reads as a moved file (STALE, file-changed); + only a document that is not a pin at all raises ValueError. + """ + if (not isinstance(pin, dict) or pin.get("schema") != PIN_SCHEMA + or not isinstance(pin.get("repos"), dict) + or not isinstance(pin.get("snapshot"), dict)): + raise ValueError(f"pin is not an {PIN_SCHEMA} document") + repo_paths, current_snapshot = _workspace_inputs(Path(root)) + current = _pin_state(repo_paths, current_snapshot) + invalidated: list[dict] = [] + derived: dict[str, list[str]] = {} # reasons feeding certificate + context-pack + for name in sorted(pin["repos"]): + reason, evidence = _repo_reason(pin["repos"][name], + current["repos"].get(name, {})) + if reason: + invalidated.append(_item(f"repo:{name}", reason, evidence)) + derived.setdefault(reason, []).extend(evidence or [f"repo:{name}"]) + snap_reason, snap_evidence = _snapshot_reason(pin["snapshot"], current_snapshot) + if snap_reason: + invalidated.append(_item("graph-snapshot", snap_reason, snap_evidence)) + derived.setdefault(snap_reason, []).extend(snap_evidence or ["graph-snapshot"]) + new_repos = sorted(set(current["repos"]) - set(pin["repos"])) + if new_repos: + derived.setdefault("unversioned", []).extend(new_repos) + if derived: + reason, evidence = _aggregate(derived) + invalidated.append(_item("certificate", reason, evidence)) + invalidated.append(_item("context-pack", reason, evidence)) + invalidated.sort(key=lambda item: item["artifact_or_scope"]) + scope = sorted(list(ARTIFACTS) + [f"repo:{name}" for name in pin["repos"]]) + named = {item["artifact_or_scope"] for item in invalidated} + still_valid = [entry for entry in scope if entry not in named] + return { + "schema": REPORT_SCHEMA, + "pinned_ref": pin.get("pinned_ref"), + "current_ref": canonical_sha(current), + "verdict": "FRESH" if not invalidated else "STALE", + "invalidated": invalidated, + "still_valid": still_valid, + "counts": {"scope": len(scope), "invalidated": len(invalidated), + "still_valid": len(still_valid)}, + "recheck": recheck or "index invalidate --root ROOT --pin PIN --json", + } + + +def reconcile_invalidation(report: object) -> dict: + """Re-derive the invalidation ledger from the report itself; DRIFT on any gap. + + MATCH is earned, never declared: declared counts must equal the list + lengths, invalidated + still_valid must equal the fingerprinted scope, + every reason code must come from the closed set, no scope may be booked + twice, and the verdict must agree with the lists it sits over. + """ + if not isinstance(report, dict) or report.get("schema") != REPORT_SCHEMA: + return _reconciliation([{"code": "report-schema", + "detail": f"report schema must be {REPORT_SCHEMA}"}]) + invalidated = report.get("invalidated") or [] + still_valid = report.get("still_valid") or [] + failures = _count_failures(report.get("counts") or {}, invalidated, still_valid) + booked: dict[str, int] = {} + for item in invalidated: + scope_name = item.get("artifact_or_scope") if isinstance(item, dict) else None + code = item.get("reason_code") if isinstance(item, dict) else None + if code not in REASON_CODES: + failures.append({"code": "unknown-reason-code", + "detail": (f"{scope_name or '?'}: {code!r} (closed set: " + f"{', '.join(sorted(REASON_CODES))})")}) + if scope_name is not None: + booked[scope_name] = booked.get(scope_name, 0) + 1 + for scope_name in still_valid: + booked[scope_name] = booked.get(scope_name, 0) + 1 + for scope_name, times in sorted(booked.items(), key=lambda kv: str(kv[0])): + if times > 1: + failures.append({"code": "duplicate-scope", + "detail": f"scope booked {times} times: {scope_name}"}) + verdict = report.get("verdict") + if verdict == "UNVERIFIABLE": + if invalidated or still_valid: + failures.append({"code": "verdict-mismatch", + "detail": "an UNVERIFIABLE report must carry no scope entries"}) + elif verdict != ("FRESH" if not invalidated else "STALE"): + failures.append({"code": "verdict-mismatch", + "detail": (f"verdict {verdict!r} over " + f"{len(invalidated)} invalidation(s)")}) + return _reconciliation(failures) + + +def _count_failures(counts: dict, invalidated: list, still_valid: list) -> list[dict]: + """The declared-count checks: the ledger must add up before anything else.""" + failures: list[dict] = [] + if counts.get("scope") != counts.get("invalidated", 0) + counts.get("still_valid", 0): + failures.append({"code": "counts-mismatch", + "detail": (f"declared scope={counts.get('scope')} != " + f"invalidated={counts.get('invalidated')} + " + f"still_valid={counts.get('still_valid')}")}) + if counts.get("invalidated") != len(invalidated): + failures.append({"code": "invalidated-count-mismatch", + "detail": (f"declared invalidated={counts.get('invalidated')} " + f"but the list holds {len(invalidated)}")}) + if counts.get("still_valid") != len(still_valid): + failures.append({"code": "still-valid-count-mismatch", + "detail": (f"declared still_valid={counts.get('still_valid')} " + f"but the list holds {len(still_valid)}")}) + return failures + + +def _reconciliation(failures: list[dict]) -> dict: + return {"schema": RECONCILIATION_SCHEMA, + "verdict": "MATCH" if not failures else "DRIFT", + "failures": failures} diff --git a/client-plugin/server/src/index_graph/freshness/invalidate_cli.py b/client-plugin/server/src/index_graph/freshness/invalidate_cli.py new file mode 100644 index 0000000..c4e37a5 --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/invalidate_cli.py @@ -0,0 +1,80 @@ +"""CLI and shared-payload face for the typed invalidation report. + +`index invalidate --out PIN` mints a pin of the current tree; +`index invalidate --pin PIN` diffs it against the tree and emits the +`index.invalidation/1` report plus its reconciliation. The MCP tool +reuses `run_invalidate` and `mint_pin`, so the protocol face never +disagrees with the CLI. +""" +from __future__ import annotations + +import json +from pathlib import Path + +from .invalidate import ( + REPORT_SCHEMA, + invalidation_report, + mint_pin, + reconcile_invalidation, +) + +_EXIT = {"FRESH": 0, "STALE": 1, "UNVERIFIABLE": 2} + + +def _unverifiable(detail: str, recheck: str | None) -> dict: + return {"schema": REPORT_SCHEMA, "pinned_ref": None, "current_ref": None, + "verdict": "UNVERIFIABLE", "detail": detail, + "invalidated": [], "still_valid": [], + "counts": {"scope": 0, "invalidated": 0, "still_valid": 0}, + "recheck": recheck or "index invalidate --root ROOT --pin PIN --json"} + + +def run_invalidate(root: Path | str, pin: dict, *, recheck: str | None = None) -> dict: + """Report + reconciliation: the shared CLI/MCP payload. + + A document that is not a pin yields an UNVERIFIABLE report, not a crash; + a tampered pin yields STALE with file-changed reasons. + """ + try: + report = invalidation_report(pin, root, recheck=recheck) + except ValueError as exc: + report = _unverifiable(str(exc), recheck) + return {"report": report, "reconciliation": reconcile_invalidation(report)} + + +def cmd_invalidate(args) -> int: + """CLI face: mint a pin with --out, or report against one with --pin.""" + root = args.root.resolve() + if not root.is_dir(): + raise SystemExit(f"root not found: {root}") + if (args.out is None) == (args.pin is None): + raise SystemExit( + "invalidate: pass exactly one of --out PIN (mint) or --pin PIN (report)") + if args.out is not None: + pin = mint_pin(root) + args.out.write_text(json.dumps(pin, indent=2, sort_keys=True), encoding="utf-8") + print(f"wrote {args.out} repos={len(pin['repos'])} " + f"pinned_ref={pin['pinned_ref'][:12]}") + return 0 + try: + pin = json.loads(args.pin.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise SystemExit(f"invalidate: cannot read pin {args.pin}: {exc}") + recheck = f'index invalidate --root "{args.root}" --pin "{args.pin}" --json' + payload = run_invalidate(root, pin, recheck=recheck) + report = payload["report"] + if args.json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + counts = report["counts"] + line = (f"invalidation verdict={report['verdict']} scope={counts['scope']} " + f"invalidated={counts['invalidated']} " + f"still_valid={counts['still_valid']}") + if report.get("detail"): + line += f": {report['detail']}" + print(line) + for item in report["invalidated"]: + print(f" {item['artifact_or_scope']}: {item['reason_code']}") + for failure in payload["reconciliation"]["failures"]: + print(f" [{failure['code']}] {failure['detail']}") + return _EXIT[report["verdict"]] diff --git a/client-plugin/server/src/index_graph/freshness/watch.py b/client-plugin/server/src/index_graph/freshness/watch.py new file mode 100644 index 0000000..6adcc0b --- /dev/null +++ b/client-plugin/server/src/index_graph/freshness/watch.py @@ -0,0 +1,118 @@ +"""watch.py — incremental auto-resync: the live drift verdict competitors have, +plus the receipt they don't. + +Every codebase-map tool in the category auto-syncs on file change; index did +not, so its freshness was baseline-only (the workbench Health view says as much +honestly). This closes that gap by COMPOSING the freshness machinery rather +than adding new state: a watcher holds the prior workspace fingerprint, and on +each tick recomputes the current one and runs the pure `compare_freshness`. The +retained prior snapshot is exactly what turns "current fingerprint only" into a +real FRESH/STALE verdict — the thing a single --root pass cannot emit. + +Every resync emits a re-checkable sync receipt (schema below): the two +fingerprint roots, the verdict, and the named repo deltas. A consumer recomputes +the fingerprint and re-runs compare_freshness to confirm it — the watcher never +asks to be trusted. + +Zero dependencies: stdlib polling (mtime + the content fingerprint that already +exists), no OS file-event API, no watchdog package. Poll cadence is a floor on +latency, not a correctness property — the fingerprint is authoritative, so a +missed poll only delays a verdict, never fabricates one. + +Robustness: a rescan that raises (a file vanishing mid-walk, a permission flip) +is caught and reported as an errored tick, never crashing the loop; the prior +good fingerprint is retained so the next tick recovers. +""" +from __future__ import annotations + +import time +from collections.abc import Callable, Iterator +from pathlib import Path + +from .compare import compare_freshness +from .fingerprint import workspace_fingerprint + +SYNC_SCHEMA = "index.freshness-sync/1" + + +def _now(clock: Callable[[], float] | None) -> float: + return (clock or time.time)() + + +def sync_report(prev: dict, curr: dict, *, tick: int, at: float, + changed_paths: list[str] | None = None) -> dict: + """One re-checkable resync receipt. `prev`/`curr` are index.freshness/1 + fingerprints; the verdict is the pure compare over them.""" + cmp = compare_freshness(prev, curr) + return { + "schema": SYNC_SCHEMA, + "tick": tick, + "at": round(at, 3), + "verdict": cmp["verdict"], # FRESH | STALE + "prev_root": cmp["stamp_root"], + "curr_root": cmp["current_root"], + "repos_added": cmp["repos_added"], + "repos_removed": cmp["repos_removed"], + "repos_changed": cmp["repos_changed"], + "changed_paths": sorted(changed_paths) if changed_paths else [], + "recheck": "index freshness --root ROOT (recompute the fingerprint and re-compare)", + } + + +def diff_changed_paths(prev_paths: dict[str, str], curr_paths: dict[str, str]) -> list[str]: + """Which relevant files changed between two path->sha maps (for the receipt's + detail line). Optional; the verdict never depends on it.""" + out = [] + for p in set(prev_paths) | set(curr_paths): + if prev_paths.get(p) != curr_paths.get(p): + out.append(p) + return out + + +def watch_iter( + repo_paths: dict[str, Path], + *, + interval: float = 2.0, + max_ticks: int | None = None, + sleep: Callable[[float], None] | None = None, + clock: Callable[[], float] | None = None, + on_change: Callable[[dict], None] | None = None, +) -> Iterator[dict]: + """Yield a sync receipt for every tick on which the workspace fingerprint + changed (STALE), plus the initial baseline tick. Deterministic and testable: + inject `sleep`/`clock`; bound with `max_ticks`. `repo_paths` is re-read each + tick so newly added/removed repos are picked up by the caller-supplied map. + + The first yielded receipt is the baseline (tick 0, verdict FRESH vs itself). + Thereafter only ticks that DETECTED a change are yielded, so a consumer that + regenerates an artifact runs only when something actually moved.""" + sleep = sleep or time.sleep + prev = _safe_fingerprint(repo_paths) + baseline = sync_report(prev, prev, tick=0, at=_now(clock)) + if on_change: + on_change(baseline) + yield baseline + tick = 0 + while max_ticks is None or tick < max_ticks: + sleep(interval) + tick += 1 + curr = _safe_fingerprint(repo_paths) + if curr is None: # errored rescan: retain prev, report, continue + yield {"schema": SYNC_SCHEMA, "tick": tick, "at": round(_now(clock), 3), + "verdict": "UNVERIFIABLE", "error": "fingerprint rescan failed", + "prev_root": prev.get("root")} + continue + if curr.get("root") != prev.get("root"): + report = sync_report(prev, curr, tick=tick, at=_now(clock)) + prev = curr + if on_change: + on_change(report) + yield report + # FRESH ticks are silent by design (no yield): only movement is reported + + +def _safe_fingerprint(repo_paths: dict[str, Path]): + try: + return workspace_fingerprint(repo_paths) + except Exception: # never let a transient FS error kill the loop + return None diff --git a/client-plugin/server/src/index_graph/gitmeta.py b/client-plugin/server/src/index_graph/gitmeta.py new file mode 100644 index 0000000..258d513 --- /dev/null +++ b/client-plugin/server/src/index_graph/gitmeta.py @@ -0,0 +1,172 @@ +"""Git subprocess access and always-on credential redaction.""" + +from __future__ import annotations + +from dataclasses import dataclass +import re +import os +import subprocess +from pathlib import Path +from typing import Any + + +DEFAULT_GIT_TIMEOUT_SECONDS = 5.0 +STATUS_GLOBAL_ARGS = ["--no-optional-locks"] +STATUS_ARGS = ["status", "--porcelain=v2", "--branch", "--untracked-files=all"] + + +@dataclass(frozen=True) +class GitCommandResult: + ok: bool + stdout: str + error: str | None = None + + +class GitMetadataError(RuntimeError): + """Raised when repo cleanliness cannot be established safely.""" + + +def git_timeout_seconds() -> float: + raw = os.environ.get("INDEX_GIT_TIMEOUT_SECONDS") + if raw is None or raw == "": + return DEFAULT_GIT_TIMEOUT_SECONDS + try: + value = float(raw) + except ValueError: + return DEFAULT_GIT_TIMEOUT_SECONDS + return max(0.1, value) + + +def _git_env(repo: Path) -> dict[str, str]: + env = dict(os.environ) + ceiling = str(Path(repo).resolve().parent) + existing = env.get("GIT_CEILING_DIRECTORIES") + if existing: + parts = [part for part in existing.split(os.pathsep) if part] + if ceiling not in parts: + parts.append(ceiling) + env["GIT_CEILING_DIRECTORIES"] = os.pathsep.join(parts) + else: + env["GIT_CEILING_DIRECTORIES"] = ceiling + return env + +# A web remote carries its whole userinfo as a secret. The token often +# sits in the user slot with no password beside it, so the slot goes as a +# unit. +_USERINFO = re.compile(r"(?i)(https?://)[^/@]+@") +# Every other scheme keeps its username, because ssh://git@host names a +# user and nothing more. A colon means a password came with it, and that +# is a secret whatever the scheme says. +_PASSWORD_USERINFO = re.compile( + r"(?i)([a-z][a-z0-9+.-]*://)[^/@\s]*:[^/@\s]*@") +# Parameter names arrive prefixed as often as bare, so access_token has to +# match as readily as token. The value class runs to the end of the query +# on purpose: swallowing a trailing parameter redacts more than was asked +# for, and the alternative is publishing a second secret whose name this +# pattern does not know. +_SECRET_QUERY = re.compile( + r"(?i)(? str: + """Strip the parts of a remote URL a reader of the map must not get. + + What survives is the scheme, the host and the path, which is what + makes a remote recognisable. Every origin passes through here before + it reaches a row, so a map cannot carry a credential even when the + clone URL held one. + """ + clean = _USERINFO.sub(r"\1@", origin) + clean = _PASSWORD_USERINFO.sub(r"\1@", clean) + return _SECRET_QUERY.sub(r"\1=", clean) + + +def run_git_checked( + repo: Path, + args: list[str], + *, + global_args: list[str] | None = None, +) -> GitCommandResult: + try: + result = subprocess.run( + ["git", *(global_args or []), "-C", str(repo), *args], + text=True, capture_output=True, timeout=git_timeout_seconds(), check=False, + env=_git_env(repo), + ) + except subprocess.TimeoutExpired: + return GitCommandResult(False, "", "timeout") + if result.returncode != 0: + return GitCommandResult(False, "", f"exit-{result.returncode}") + return GitCommandResult(True, result.stdout.strip()) + + +def run_git(repo: Path, args: list[str]) -> str: + result = run_git_checked(repo, args) + return result.stdout if result.ok else "" + + +def _branch_from_status(lines: list[str]) -> str: + if not lines or not lines[0].startswith("## "): + return "" + branch = lines[0][3:].strip() + if branch.startswith("No commits yet on "): + return branch.removeprefix("No commits yet on ").strip() or "" + if "..." in branch: + branch = branch.split("...", 1)[0] + if branch.endswith("]") and "[" in branch: + branch = branch.rsplit("[", 1)[0].strip() + return branch or "" + + +def _metadata_from_porcelain_v2(status: str) -> tuple[str, str, int, int]: + branch = "" + head = "" + dirty = 0 + untracked = 0 + for line in status.splitlines(): + if line.startswith("# branch.head "): + branch = line.removeprefix("# branch.head ").strip() + if branch == "(detached)": + branch = "HEAD" + elif line.startswith("# branch.oid "): + oid = line.removeprefix("# branch.oid ").strip() + if oid and oid != "(initial)": + head = oid[:7] + elif line.startswith("? "): + untracked += 1 + elif line and not line.startswith("# "): + dirty += 1 + return branch, head, dirty, untracked + + +def _status_signature(status: str) -> str: + import hashlib + + return hashlib.sha256(status.encode("utf-8")).hexdigest() + + +def repo_metadata(repo: Path) -> dict[str, Any]: + status_result = run_git_checked(repo, STATUS_ARGS, global_args=STATUS_GLOBAL_ARGS) + if not status_result.ok: + raise GitMetadataError(f"git status unavailable: {status_result.error or 'unknown'}") + branch, head, dirty, untracked = _metadata_from_porcelain_v2(status_result.stdout) + status_lines = status_result.stdout.splitlines() + branch = ( + branch + or _branch_from_status(status_lines) + or run_git(repo, ["branch", "--show-current"]) + or run_git(repo, ["rev-parse", "--abbrev-ref", "HEAD"]) + or "unknown" + ) + head = head or run_git(repo, ["rev-parse", "--short=7", "HEAD"]) or "unknown" + origin = sanitize_credentials(run_git(repo, ["config", "--get", "remote.origin.url"]) or "") + return { + "branch": branch, + "head": head, + "origin": origin, + "dirty_count": dirty, + "untracked_count": untracked, + "metadata_status": "ok", + "metadata_error": None, + "status_signature": _status_signature(status_result.stdout), + } diff --git a/client-plugin/server/src/index_graph/graph/__init__.py b/client-plugin/server/src/index_graph/graph/__init__.py new file mode 100644 index 0000000..e4baed2 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/__init__.py @@ -0,0 +1,16 @@ +"""Repo-level dependency inference engine.""" + +from __future__ import annotations + +__all__ = ["DependencyGraph", "RepoNode", "build_graph"] + + +def __getattr__(name: str): + if name in __all__: + from .build import DependencyGraph, RepoNode, build_graph + return { + "DependencyGraph": DependencyGraph, + "RepoNode": RepoNode, + "build_graph": build_graph, + }[name] + raise AttributeError(name) diff --git a/client-plugin/server/src/index_graph/graph/build.py b/client-plugin/server/src/index_graph/graph/build.py new file mode 100644 index 0000000..a40bc2a --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/build.py @@ -0,0 +1,341 @@ +"""Assemble repo trees + resolvers into a DependencyGraph.""" +from __future__ import annotations + +import configparser +import json +import os +import tomllib +import multiprocessing +from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed +from collections.abc import Callable, Iterator, Mapping +from dataclasses import dataclass, field, replace +from pathlib import Path +from time import perf_counter + +from . import cache as _cache +from .edges import Edge, build_index, resolve_edges +from .description import description as _description +from .walk import ( + PreloadedFileValue, + cached_file_scope, + read_source_text, + source_reuse_complete, + walk_files, +) +from .resolvers import ALL_RESOLVERS +from .resolvers.base import RawEdge +from .roles import derive_roles + +make_lookup = _cache.make_lookup + +@dataclass(frozen=True) +class RepoNode: + name: str + path: str + ecosystems: tuple[str, ...] + exposed_names: frozenset[str] + description: str + markers: frozenset[str] + +@dataclass(frozen=True) +class DependencyGraph: + repos: tuple[RepoNode, ...] + edges: tuple[Edge, ...] + roles: dict[str, tuple[str, ...]] + warnings: tuple[str, ...] + cache_summary: dict[str, object] | None = field(default=None, compare=False) + +@dataclass(frozen=True) +class GraphProgress: + """Observed work, never an estimate of source coverage or semantic truth.""" + + phase: str + completed_repos: int + total_repos: int + elapsed_ms: int + +@dataclass(frozen=True) +class _RepoBuild: + name: str + node: RepoNode + exposed_names: set[str] + raw_edges: list[RawEdge] + markers: set[str] + cache_outcome: str = "bypassed" + cache_reason: str = "not_recorded" + cache_written: bool = False + cache_stats: dict[str, int] = field(default_factory=dict, compare=False) + +def detect_markers(repo_root: Path, exposed: set[str]) -> set[str]: + mk: set[str] = set() + if exposed: + mk.add("published") + pp = repo_root / "pyproject.toml" + if pp.is_file(): + try: + data = tomllib.loads(read_source_text(pp, encoding="utf-8", errors="replace")) + if data.get("project", {}).get("scripts") or \ + data.get("project", {}).get("entry-points"): + mk.add("entry") + except (tomllib.TOMLDecodeError, OSError): + pass + cfg = repo_root / "setup.cfg" + if cfg.is_file(): + try: + cp = configparser.ConfigParser() + cp.read_string(read_source_text(cfg, encoding="utf-8", errors="strict")) + if cp.has_option("options.entry_points", "console_scripts"): + mk.add("entry") + except (configparser.Error, OSError): + pass + pj = repo_root / "package.json" + if pj.is_file(): + try: + if json.loads(read_source_text(pj, encoding="utf-8", errors="replace")).get("bin"): + mk.add("entry") + except (json.JSONDecodeError, OSError): + pass + if any(walk_files(repo_root, names=("__main__.py",), stop_at_nested_repos=True)): + mk.add("entry") + return mk + +def _default_jobs() -> int: + return min(32, (os.cpu_count() or 4) * 5) + +def _build_one_repo(item: tuple[str, Path], resolvers) -> _RepoBuild: + name, root = item + ecos: list[str] = [] + names: set[str] = set() + raws: list[RawEdge] = [] + for resolver in resolvers: + if resolver.matches(root): + ecos.append(resolver.name) + names |= resolver.exposed_names(root) + raws += resolver.raw_edges(root) + mk = detect_markers(root, names) + node = RepoNode(name, str(root), tuple(ecos), frozenset(names), _description(root), frozenset(mk)) + return _RepoBuild(name, node, names, raws, mk) + +def _repo_build_to_json(build: _RepoBuild) -> dict: + return { + "node": { + "name": build.node.name, + "path": build.node.path, + "ecosystems": list(build.node.ecosystems), + "exposed_names": sorted(build.node.exposed_names), + "description": build.node.description, + "markers": sorted(build.node.markers), + }, + "exposed_names": sorted(build.exposed_names), + "raw_edges": [ + { + "target_name": edge.target_name, + "signal": edge.signal, + "evidence_file": edge.evidence_file, + "evidence_line": edge.evidence_line, + "raw_spec": edge.raw_spec, + } + for edge in build.raw_edges + ], + "markers": sorted(build.markers), + } + +def _repo_build_from_json(data: dict) -> _RepoBuild: + node = data["node"] + return _RepoBuild( + name=str(node["name"]), + node=RepoNode( + name=str(node["name"]), + path=str(node["path"]), + ecosystems=tuple(str(item) for item in node.get("ecosystems", ())), + exposed_names=frozenset(str(item) for item in node.get("exposed_names", ())), + description=str(node.get("description", "")), + markers=frozenset(str(item) for item in node.get("markers", ())), + ), + exposed_names=set(str(item) for item in data.get("exposed_names", ())), + raw_edges=[ + RawEdge( + str(edge["target_name"]), + str(edge["signal"]), + str(edge["evidence_file"]), + edge.get("evidence_line"), + str(edge["raw_spec"]), + ) + for edge in data.get("raw_edges", ()) + ], + markers=set(str(item) for item in data.get("markers", ())), + ) + +def _load_or_build_one_repo( + item: tuple[str, Path], + resolvers, + *, + use_cache: bool, + file_list: PreloadedFileValue | None = None, +) -> _RepoBuild: + name, requested_root = item + source_item = (name, requested_root.resolve()) + cache_stats = _cache.empty_repo_cache_stats() + + def presented(built: _RepoBuild) -> _RepoBuild: + return replace(built, node=replace(built.node, path=str(requested_root))) + + cache_allowed = use_cache and _cache.resolvers_use_shared_source_reads(resolvers) + scope_files = {source_item[1]: file_list} if file_list is not None else None + with cached_file_scope(scope_files): + if not cache_allowed: + build_started = perf_counter() + built = _build_one_repo(source_item, resolvers) + _cache.add_cache_timing(cache_stats, "fresh_build", build_started) + return presented(replace( + built, + cache_outcome="bypassed", + cache_reason="disabled_or_unsupported", + cache_stats=cache_stats, + )) + lookup = make_lookup(name, source_item[1], resolvers, stats=cache_stats) + cached, cache_outcome, cache_reason = _cache.read_repo_build_with_outcome(lookup, stats=cache_stats) + if cached is not None: + rehydrate_started = perf_counter() + try: + built = _repo_build_from_json(cached) + _cache.add_cache_timing(cache_stats, "cache_rehydrate", rehydrate_started) + return presented(replace( + built, + cache_outcome="hit", + cache_reason=cache_reason, + cache_stats=cache_stats, + )) + except (KeyError, TypeError, ValueError): + _cache.add_cache_timing(cache_stats, "cache_rehydrate", rehydrate_started) + cache_outcome = "invalid" + cache_reason = "build_payload" + build_started = perf_counter() + built = _build_one_repo(source_item, resolvers) + _cache.add_cache_timing(cache_stats, "fresh_build", build_started) + cache_written = False + if source_reuse_complete(): + cache_written = _cache.write_repo_build(lookup, _repo_build_to_json(built), stats=cache_stats) + return presented(replace( + built, + cache_outcome=cache_outcome, + cache_reason=cache_reason, + cache_written=cache_written, + cache_stats=cache_stats, + )) + +def _collect_repos( + items: list[tuple[str, Path]], + resolvers, + jobs: int, + *, + use_cache: bool, + file_lists: Mapping[str, PreloadedFileValue] | None = None, + executor: str = "thread", +) -> Iterator[_RepoBuild]: + if jobs <= 1 or len(items) <= 1: + for item in items: + file_list = file_lists.get(item[0]) if file_lists is not None else None + yield _load_or_build_one_repo(item, resolvers, use_cache=use_cache, + file_list=file_list) + return + # Spawn avoids inheriting locks or application state from an MCP host. The + # Python API retains threads by default for local/custom resolver objects. + pool_context = ( + ProcessPoolExecutor(max_workers=jobs, mp_context=multiprocessing.get_context("spawn")) + if executor == "process" else ThreadPoolExecutor(max_workers=jobs) + ) + with pool_context as pool: + pending = [ + pool.submit( + _load_or_build_one_repo, + item, + resolvers, + use_cache=use_cache, + file_list=file_lists.get(item[0]) if file_lists is not None else None, + ) + for item in items + ] + try: + for future in as_completed(pending): + yield future.result() + except BaseException: + # A failed scan or cancelled consumer cannot use any later result. + # Stop queued work before the context waits for active workers. + for future in pending: + future.cancel() + raise + +def build_graph( + repo_paths: dict[str, Path], + resolvers=ALL_RESOLVERS, + *, + jobs: int | None = None, + use_cache: bool = True, + on_progress: Callable[[GraphProgress], None] | None = None, + executor: str = "thread", + file_lists: Mapping[str, PreloadedFileValue] | None = None, +) -> DependencyGraph: + """Build complete dependency evidence; optional progress runs in the caller. + + Process workers are opt-in for Python callers, require picklable resolvers + and a guarded main entrypoint, and default to at most four workers. CLI/MCP + entrypoints select them to avoid Python parser contention across repos. + """ + if executor not in {"thread", "process"}: + raise ValueError("executor must be 'thread' or 'process'") + nodes: list[RepoNode] = [] + exposed: dict[str, set[str]] = {} + repo_raw: dict[str, list[RawEdge]] = {} + markers: dict[str, set[str]] = {} + cache_summary = _cache.empty_cache_summary() + default_jobs = min(4, os.cpu_count() or 1) if executor == "process" else _default_jobs() + worker_count = default_jobs if jobs is None else max(1, jobs) + started = perf_counter() + + def report(phase: str) -> None: + if on_progress is not None: + on_progress(GraphProgress(phase, len(nodes), len(repo_paths), + int((perf_counter() - started) * 1000))) + + try: + report("building") + for built in _collect_repos( + sorted(repo_paths.items()), resolvers, worker_count, use_cache=use_cache, + file_lists=file_lists, + executor=executor, + ): + _cache.record_cache_summary( + cache_summary, + built.cache_outcome, + built.cache_reason, + built.cache_written, + built.cache_stats, + ) + exposed[built.name] = built.exposed_names + repo_raw[built.name] = built.raw_edges + markers[built.name] = built.markers + nodes.append(built.node) + report("building") + + # Completion order is for observability only. Preserve stable graph order. + nodes.sort(key=lambda node: node.name) + exposed = dict(sorted(exposed.items())) + repo_raw = dict(sorted(repo_raw.items())) + markers = dict(sorted(markers.items())) + report("resolving") + index = build_index(exposed) + edges, warnings = resolve_edges(repo_raw, index) + roles = derive_roles(set(repo_paths), edges, markers) + graph = DependencyGraph( + tuple(nodes), + tuple(edges), + roles, + tuple(warnings), + _cache.final_cache_summary(cache_summary), + ) + except BaseException: + report("failed") + raise + report("complete") + return graph diff --git a/client-plugin/server/src/index_graph/graph/cache.py b/client-plugin/server/src/index_graph/graph/cache.py new file mode 100644 index 0000000..7a2a0c0 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/cache.py @@ -0,0 +1,239 @@ +"""Per-repo graph resolver cache. + +The cache stores derived resolver facts, never raw source. A repo entry is valid +only for the same repo key, resolved path, resolver signature, graph-relevant +content fingerprint, and resolver source-read contract. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from dataclasses import dataclass +from pathlib import Path +from time import perf_counter +from types import CodeType +from typing import Any + +from .walk import read_source_bytes +from .resolvers import BUILTIN_SHARED_SOURCE_READER_TYPES +from .cache_metrics import ( + add_cache_stat, + add_cache_timing, + empty_cache_summary, + empty_repo_cache_stats, + final_cache_summary, + record_cache_summary, +) +from .. import __version__ +from ..freshness.fingerprint import repo_fingerprint + +SCHEMA = "index.graph-repo-cache/v1" +CACHE_KEY_VERSION = "repo-build/v4" +_DESCRIPTION_NAMES = ("README.md", "README.rst", "README.txt", "readme.md") + + +@dataclass(frozen=True) +class RepoCacheLookup: + key: str + path: Path + repo_name: str + repo_path: str + fingerprint: str + resolver_signature: tuple[str, ...] + + +def _cache_dir() -> Path: + raw = os.environ.get("INDEX_GRAPH_REPO_CACHE_DIR") + if raw: + return Path(raw) + raw = os.environ.get("INDEX_CACHE_DIR") + if raw: + return Path(raw) / "graph-repos" + base = os.environ.get("LOCALAPPDATA") + if base: + return Path(base) / "index_graph" / "cache" / "graph-repos" + return Path.home() / ".cache" / "index_graph" / "graph-repos" + + +def _sha256_text(value: str) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def _implementation_value(value): + """Encode immutable code values without marshal's object-reference state.""" + if isinstance(value, CodeType): + fields = ("co_argcount", "co_posonlyargcount", "co_kwonlyargcount", + "co_nlocals", "co_stacksize", "co_flags", "co_code", + "co_consts", "co_names", "co_varnames", "co_freevars", + "co_cellvars", "co_exceptiontable") + return ["code", [[name, _implementation_value(getattr(value, name, None))] + for name in fields]] + if isinstance(value, tuple): + return ["tuple", [_implementation_value(item) for item in value]] + if isinstance(value, frozenset): + items = [_implementation_value(item) for item in value] + return ["frozenset", sorted(items, key=lambda item: json.dumps(item))] + if isinstance(value, bytes): + return ["bytes", value.hex()] + return [type(value).__name__, repr(value)] + + +def _resolver_signature(resolvers) -> tuple[str, ...]: + parts = [] + for resolver in resolvers: + cls = type(resolver) + name = str(getattr(resolver, "name", cls.__name__)) + implementation = hashlib.sha256() + implementation.update(str(sys.implementation.cache_tag).encode("utf-8")) + for method_name in sorted(dir(cls)): + method = getattr(cls, method_name, None) + code = getattr(method, "__code__", None) + if code is not None: + implementation.update(method_name.encode("utf-8")) + implementation.update(json.dumps(_implementation_value(code), + separators=(",", ":")).encode("utf-8")) + # Custom resolvers with external configuration can invalidate their + # derived facts explicitly. Package version covers shipped helper changes. + implementation.update(str(getattr(resolver, "cache_version", "1")).encode("utf-8")) + parts.append(f"{name}:{cls.__module__}.{cls.__qualname__}:{implementation.hexdigest()}") + return tuple(sorted(parts)) + + +def resolver_uses_shared_source_reads(resolver) -> bool: + """Whether a resolver can safely participate in persistent repo caching.""" + if type(resolver) in BUILTIN_SHARED_SOURCE_READER_TYPES: + return True + return getattr(resolver, "uses_shared_source_reader", False) is True + + +def resolvers_use_shared_source_reads(resolvers) -> bool: + return all(resolver_uses_shared_source_reads(resolver) for resolver in resolvers) + + +def _file_digest(path: Path, stats: dict[str, int] | None = None) -> str: + started = perf_counter() + try: + data = read_source_bytes(path) + add_cache_stat(stats, "fingerprint_description_files", 1) + add_cache_stat(stats, "fingerprint_description_bytes", len(data)) + return hashlib.sha256(data).hexdigest() + except OSError: + add_cache_stat(stats, "fingerprint_description_unreadable", 1) + return "unreadable" + finally: + add_cache_timing(stats, "description_read_hash", started) + + +def repo_graph_fingerprint(repo_root: Path, resolvers, stats: dict[str, int] | None = None) -> str: + """Fingerprint resolver-relevant content plus description files. + + `repo_fingerprint` covers files read by resolvers. The graph node also carries + README/package descriptions, so the repo graph cache includes those root docs + to avoid stale inventory text. + """ + root = Path(repo_root) + parts = [repo_fingerprint(root, resolvers, stats=stats)] + for name in _DESCRIPTION_NAMES: + path = root / name + if path.is_file(): + parts.append(f"{name}:{_file_digest(path, stats=stats)}") + else: + parts.append(f"{name}:missing") + return _sha256_text("|".join(parts)) + + +def make_lookup(repo_name: str, repo_root: Path, resolvers, stats: dict[str, int] | None = None) -> RepoCacheLookup: + root = Path(repo_root).resolve() + started = perf_counter() + fingerprint = repo_graph_fingerprint(root, resolvers, stats=stats) + add_cache_timing(stats, "fingerprint", started) + started = perf_counter() + signature = _resolver_signature(resolvers) + add_cache_timing(stats, "resolver_signature", started) + payload = { + "version": CACHE_KEY_VERSION, + "index_version": __version__, + "repo_name": repo_name, + "repo_path": str(root), + "fingerprint": fingerprint, + "resolver_signature": signature, + } + key = _sha256_text(json.dumps(payload, sort_keys=True, separators=(",", ":"))) + return RepoCacheLookup( + key=key, + path=_cache_dir() / f"{key}.json", + repo_name=repo_name, + repo_path=str(root), + fingerprint=fingerprint, + resolver_signature=signature, + ) + + +def read_repo_build_with_outcome( + lookup: RepoCacheLookup, stats: dict[str, int] | None = None +) -> tuple[dict[str, Any] | None, str, str]: + started = perf_counter() + try: + raw = lookup.path.read_text(encoding="utf-8") + except FileNotFoundError: + add_cache_timing(stats, "cache_read", started) + return None, "miss", "not_found" + except OSError: + add_cache_timing(stats, "cache_read", started) + return None, "invalid", "unreadable" + add_cache_timing(stats, "cache_read", started) + started = perf_counter() + try: + data = json.loads(raw) + except (json.JSONDecodeError, ValueError): + add_cache_timing(stats, "cache_decode", started) + return None, "invalid", "invalid_json" + add_cache_timing(stats, "cache_decode", started) + if data.get("schema") != SCHEMA: + return None, "invalid", "schema" + if data.get("repo_name") != lookup.repo_name: + return None, "invalid", "repo_name" + if data.get("repo_path") != lookup.repo_path: + return None, "invalid", "repo_path" + if data.get("fingerprint") != lookup.fingerprint: + return None, "invalid", "fingerprint" + if tuple(data.get("resolver_signature") or ()) != lookup.resolver_signature: + return None, "invalid", "resolver_signature" + build = data.get("build") + if not isinstance(build, dict): + return None, "invalid", "build" + return build, "hit", "hit" + + +def read_repo_build(lookup: RepoCacheLookup) -> dict[str, Any] | None: + build, _outcome, _reason = read_repo_build_with_outcome(lookup) + return build + + +def write_repo_build( + lookup: RepoCacheLookup, build: dict[str, Any], stats: dict[str, int] | None = None +) -> bool: + started = perf_counter() + payload = { + "schema": SCHEMA, + "version": CACHE_KEY_VERSION, + "repo_name": lookup.repo_name, + "repo_path": lookup.repo_path, + "fingerprint": lookup.fingerprint, + "resolver_signature": list(lookup.resolver_signature), + "build": build, + } + try: + lookup.path.parent.mkdir(parents=True, exist_ok=True) + lookup.path.write_text( + json.dumps(payload, sort_keys=True, separators=(",", ":")), + encoding="utf-8", + ) + return True + except OSError: + return False + finally: + add_cache_timing(stats, "cache_write", started) diff --git a/client-plugin/server/src/index_graph/graph/cache_metrics.py b/client-plugin/server/src/index_graph/graph/cache_metrics.py new file mode 100644 index 0000000..ba747c7 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/cache_metrics.py @@ -0,0 +1,100 @@ +"""Aggregate graph-cache telemetry without source or path disclosure.""" +from __future__ import annotations + +from time import perf_counter + +CACHE_STAGE_KEYS = ( + "fingerprint", + "fingerprint_walk", + "fingerprint_read_hash", + "description_read_hash", + "resolver_signature", + "cache_read", + "cache_decode", + "cache_rehydrate", + "fresh_build", + "cache_write", +) +FINGERPRINT_KEYS = ( + "files", + "bytes", + "unreadable", + "description_files", + "description_bytes", + "description_unreadable", +) + + +def empty_repo_cache_stats() -> dict[str, int]: + stats = {f"{key}_ms": 0 for key in CACHE_STAGE_KEYS} + stats.update({f"fingerprint_{key}": 0 for key in FINGERPRINT_KEYS}) + return stats + + +def add_cache_stat(stats: dict[str, int] | None, key: str, value: int) -> None: + if stats is not None: + stats[key] = int(stats.get(key, 0)) + int(value) + + +def add_cache_timing(stats: dict[str, int] | None, key: str, started: float) -> None: + add_cache_stat(stats, f"{key}_ms", int((perf_counter() - started) * 1000)) + + +def empty_cache_summary() -> dict[str, object]: + return { + "hits": 0, + "misses": 0, + "invalid": 0, + "bypassed": 0, + "writes": 0, + "reasons": {}, + "stage_timings_ms": {key: 0 for key in CACHE_STAGE_KEYS}, + "fingerprint": {key: 0 for key in FINGERPRINT_KEYS}, + } + + +def record_cache_summary( + summary: dict[str, object], + outcome: str, + reason: str, + written: bool, + stats: dict[str, int] | None = None, +) -> None: + outcome_to_key = {"hit": "hits", "miss": "misses", "invalid": "invalid", "bypassed": "bypassed"} + key = outcome_to_key.get(outcome, "invalid") + summary[key] = int(summary.get(key, 0)) + 1 + if written: + summary["writes"] = int(summary.get("writes", 0)) + 1 + reasons = summary.setdefault("reasons", {}) + if isinstance(reasons, dict): + reasons[reason or outcome] = int(reasons.get(reason or outcome, 0)) + 1 + stage_timings = summary.setdefault("stage_timings_ms", {}) + if isinstance(stage_timings, dict): + for stage_key in CACHE_STAGE_KEYS: + stage_timings[stage_key] = int(stage_timings.get(stage_key, 0)) + int((stats or {}).get(f"{stage_key}_ms", 0)) + fingerprint = summary.setdefault("fingerprint", {}) + if isinstance(fingerprint, dict): + for stat_key in FINGERPRINT_KEYS: + fingerprint[stat_key] = int(fingerprint.get(stat_key, 0)) + int((stats or {}).get(f"fingerprint_{stat_key}", 0)) + + +def final_cache_summary(summary: dict[str, object]) -> dict[str, object]: + reasons = summary.get("reasons") + stage_timings = summary.get("stage_timings_ms") + fingerprint = summary.get("fingerprint") + return { + "hits": int(summary.get("hits", 0)), + "misses": int(summary.get("misses", 0)), + "invalid": int(summary.get("invalid", 0)), + "bypassed": int(summary.get("bypassed", 0)), + "writes": int(summary.get("writes", 0)), + "reasons": dict(sorted(reasons.items())) if isinstance(reasons, dict) else {}, + "stage_timings_ms": { + key: int(stage_timings.get(key, 0)) if isinstance(stage_timings, dict) else 0 + for key in CACHE_STAGE_KEYS + }, + "fingerprint": { + key: int(fingerprint.get(key, 0)) if isinstance(fingerprint, dict) else 0 + for key in FINGERPRINT_KEYS + }, + } diff --git a/client-plugin/server/src/index_graph/graph/cycles.py b/client-plugin/server/src/index_graph/graph/cycles.py new file mode 100644 index 0000000..288994d --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/cycles.py @@ -0,0 +1,73 @@ +"""Detect dependency cycles (strongly-connected components) over internal edges. + +Tarjan's SCC is linear; repo-level graphs are tiny (hundreds of nodes), so the +recursive form is safe and deterministic (inputs are sorted before traversal).""" +from __future__ import annotations + +from collections.abc import Sequence + +from .edges import Edge + + +def _adjacency(edges: Sequence[Edge]) -> dict[str, list[str]]: + adj: dict[str, list[str]] = {} + for e in edges: + if e.external or e.to_repo is None: + continue + adj.setdefault(e.from_repo, []) + adj.setdefault(e.to_repo, []) + if e.to_repo not in adj[e.from_repo]: + adj[e.from_repo].append(e.to_repo) + return adj + + +def find_cycles(edges: Sequence[Edge]) -> list[tuple[str, ...]]: + """Node sets of each dependency cycle, deterministically sorted. An SCC with + >1 node, or a self-loop, is a cycle; a pure DAG returns [].""" + adj = _adjacency(edges) + self_loops = {e.from_repo for e in edges + if not e.external and e.to_repo == e.from_repo} + counter = [0] + stack: list[str] = [] + on_stack: set[str] = set() + index: dict[str, int] = {} + low: dict[str, int] = {} + sccs: list[list[str]] = [] + + def connect(v: str) -> None: + index[v] = low[v] = counter[0] + counter[0] += 1 + stack.append(v) + on_stack.add(v) + for w in adj.get(v, []): + if w not in index: + connect(w) + low[v] = min(low[v], low[w]) + elif w in on_stack: + low[v] = min(low[v], index[w]) + if low[v] == index[v]: + comp: list[str] = [] + while True: + w = stack.pop() + on_stack.discard(w) + comp.append(w) + if w == v: + break + sccs.append(comp) + + for v in sorted(adj): + if v not in index: + connect(v) + + cycles = [tuple(sorted(c)) for c in sccs + if len(c) > 1 or (len(c) == 1 and c[0] in self_loops)] + return sorted(cycles) + + +def cycle_edge_keys(edges: Sequence[Edge], + cycles: Sequence[tuple[str, ...]]) -> set[tuple[str, str]]: + """(from, to) of internal edges whose endpoints share a cycle.""" + member: dict[str, tuple[str, ...]] = {n: c for c in cycles for n in c} + return {(e.from_repo, e.to_repo) for e in edges + if not e.external and e.to_repo is not None + and e.from_repo in member and member[e.from_repo] == member.get(e.to_repo)} diff --git a/client-plugin/server/src/index_graph/graph/description.py b/client-plugin/server/src/index_graph/graph/description.py new file mode 100644 index 0000000..27b1bdd --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/description.py @@ -0,0 +1,42 @@ +"""Repository description extraction for graph nodes.""" +from __future__ import annotations + +import json +import re +import tomllib +from pathlib import Path + +from .walk import read_source_text + +_PARA = re.compile(r"\n\s*\n") + + +def description(repo_root: Path) -> str: + for readme in ("README.md", "README.rst", "README.txt", "readme.md"): + p = repo_root / readme + if p.is_file(): + try: + text = read_source_text(p, encoding="utf-8", errors="replace").strip() + except OSError: + continue + for block in _PARA.split(text): + b = block.strip() + if b and not b.startswith("#") and not b.startswith("!["): + return " ".join(b.split())[:300] + pp = repo_root / "pyproject.toml" + if pp.is_file(): + try: + d = tomllib.loads(read_source_text(pp, encoding="utf-8", errors="replace")).get("project", {}) + if d.get("description"): + return str(d["description"]) + except (tomllib.TOMLDecodeError, OSError): + pass + pj = repo_root / "package.json" + if pj.is_file(): + try: + d = json.loads(read_source_text(pj, encoding="utf-8", errors="replace")) + if d.get("description"): + return str(d["description"]) + except (json.JSONDecodeError, OSError): + pass + return "(no description)" diff --git a/client-plugin/server/src/index_graph/graph/edges.py b/client-plugin/server/src/index_graph/graph/edges.py new file mode 100644 index 0000000..924b87e --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/edges.py @@ -0,0 +1,97 @@ +"""Resolve RawEdges into evidence-carrying repo->repo Edges.""" +from __future__ import annotations + +from dataclasses import dataclass + +from .resolvers.base import RawEdge, normalize_name + + +@dataclass(frozen=True) +class Signal: + kind: str # "manifest" | "import" + evidence_file: str + evidence_line: int | None + raw_spec: str + + +@dataclass(frozen=True) +class Edge: + from_repo: str + to_repo: str | None + target_name: str + external: bool + confidence: str # "high" | "moderate" | "low" + signals: tuple[Signal, ...] + + +def build_index(exposed: dict[str, set[str]]) -> dict[str, list[str]]: + index: dict[str, list[str]] = {} + for repo, names in exposed.items(): + for n in names: + key = normalize_name(n) + bucket = index.setdefault(key, []) + if repo not in bucket: + bucket.append(repo) + return index + + +def _resolve_target(index: dict[str, list[str]], norm_target: str) -> tuple[str, list[str]]: + """Return (canonical_key, candidate_repos). + + Exact match wins. Otherwise, for a slash-containing target (a path-like name, + e.g. a Go import path), fall back to the longest exposed name that is a + segment-aligned prefix of the target. Unmatched targets keep their own name + (they become external edges). + """ + if norm_target in index: + return norm_target, index[norm_target] + if "/" in norm_target: + best: str | None = None + for key in index: + if "/" in key and (norm_target == key or norm_target.startswith(key + "/")): + if best is None or len(key) > len(best): + best = key + if best is not None: + return best, index[best] + return norm_target, [] + + +def _grade(signals: list[Signal], ambiguous: bool, target: str, short_len: int) -> str: + if ambiguous or len(normalize_name(target)) <= short_len: + return "low" + kinds = {s.kind for s in signals} + return "high" if {"manifest", "import"} <= kinds else "moderate" + + +def resolve_edges(repo_raw: dict[str, list[RawEdge]], index: dict[str, list[str]], + short_len: int = 2) -> tuple[list[Edge], list[str]]: + warnings: list[str] = [] + # group RawEdges by (from_repo, resolved_target_or_external_name) + grouped: dict[tuple[str, str | None, str], list[Signal]] = {} + ambiguous_keys: set[tuple[str, str | None, str]] = set() + for frm, raws in repo_raw.items(): + for r in raws: + canon, candidates = _resolve_target(index, normalize_name(r.target_name)) + internal = [c for c in candidates if c != frm] + sig = Signal(r.signal, r.evidence_file, r.evidence_line, r.raw_spec) + if not candidates: + key = (frm, None, canon) # external + elif not internal: + continue # self-edge only -> drop + else: + to = sorted(internal)[0] + key = (frm, to, canon) + if len(internal) > 1: + ambiguous_keys.add(key) + warnings.append( + f"ambiguous: {frm} -> {r.target_name!r} matches {sorted(internal)}") + grouped.setdefault(key, []).append(sig) + + edges: list[Edge] = [] + for (frm, to, target), signals in grouped.items(): + external = to is None + conf = "moderate" if external else _grade( + signals, (frm, to, target) in ambiguous_keys, target, short_len) + edges.append(Edge(frm, to, target, external, conf, tuple(signals))) + edges.sort(key=lambda e: (e.from_repo, e.to_repo or "", e.target_name)) + return edges, warnings diff --git a/client-plugin/server/src/index_graph/graph/progress.py b/client-plugin/server/src/index_graph/graph/progress.py new file mode 100644 index 0000000..ae2526f --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/progress.py @@ -0,0 +1,27 @@ +"""Rate-limited graph diagnostics, separate from CLI/MCP result streams.""" +from __future__ import annotations + +import json +import sys +from dataclasses import asdict + +from .build import GraphProgress + + +def stderr_progress(): + last_phase = None + last_ms = -1000 + + def report(event: GraphProgress) -> None: + nonlocal last_phase, last_ms + if event.phase == last_phase and event.elapsed_ms - last_ms < 1000: + return + last_phase, last_ms = event.phase, event.elapsed_ms + try: + print(json.dumps({"schema": "index.graph-progress/v1", **asdict(event)}), + file=sys.stderr, flush=True) + except OSError: + # Losing a diagnostic sink must not change dependency evidence. + pass + + return report diff --git a/client-plugin/server/src/index_graph/graph/resolvers/__init__.py b/client-plugin/server/src/index_graph/graph/resolvers/__init__.py new file mode 100644 index 0000000..ce5066f --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/__init__.py @@ -0,0 +1,25 @@ +"""Per-ecosystem dependency resolvers.""" +from .cpp import CppResolver +from .csharp import CSharpResolver +from .go import GoResolver +from .java import JavaResolver +from .javascript import JavaScriptResolver +from .php import PhpResolver +from .python import PythonResolver +from .ruby import RubyResolver +from .rust import RustResolver + +ALL_RESOLVERS = (PythonResolver(), JavaScriptResolver(), RustResolver(), GoResolver(), + JavaResolver(), CSharpResolver(), RubyResolver(), PhpResolver(), CppResolver()) + +BUILTIN_SHARED_SOURCE_READER_TYPES = ( + PythonResolver, + JavaScriptResolver, + RustResolver, + GoResolver, + JavaResolver, + CSharpResolver, + RubyResolver, + PhpResolver, + CppResolver, +) diff --git a/client-plugin/server/src/index_graph/graph/resolvers/base.py b/client-plugin/server/src/index_graph/graph/resolvers/base.py new file mode 100644 index 0000000..ea70a75 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/base.py @@ -0,0 +1,31 @@ +"""Resolver seam: the generic interface every ecosystem implements.""" +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Protocol + + +@dataclass(frozen=True) +class RawEdge: + target_name: str # name imported/declared, e.g. "requests", "@scope/pkg" + signal: str # "manifest" | "import" + evidence_file: str # repo-relative path of the witnessing file + evidence_line: int | None # line number where cheaply known, else None + raw_spec: str # literal text witnessed (dep spec or import line) + + +def normalize_name(name: str) -> str: + """Lowercase and unify '_'/'-' so a dist name matches an import name.""" + return name.strip().lower().replace("_", "-") + + +class Resolver(Protocol): + # Optional runtime opt-in for custom resolvers. Set + # uses_shared_source_reader = True only when every graph-relevant source read + # goes through index_graph.graph.walk.read_source_bytes/read_source_text. + name: str + + def matches(self, repo_root: Path) -> bool: ... + def exposed_names(self, repo_root: Path) -> set[str]: ... + def raw_edges(self, repo_root: Path) -> list[RawEdge]: ... diff --git a/client-plugin/server/src/index_graph/graph/resolvers/cpp.py b/client-plugin/server/src/index_graph/graph/resolvers/cpp.py new file mode 100644 index 0000000..48f6f5c --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/cpp.py @@ -0,0 +1,110 @@ +"""C/C++ ecosystem resolver: CMake target links + #include directives. + +Best-effort only. C/C++ has no single canonical dependency manifest. +This resolver reads: + - CMakeLists.txt for project/library/executable names (exposed_names) + - target_link_libraries(...) for manifest edges + - add_subdirectory(...) for manifest edges + - #include "..." and #include <...> for import edges + +Keywords PUBLIC, PRIVATE, INTERFACE in target_link_libraries are skipped; +they are not dependency targets. +""" +from __future__ import annotations + +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +# CMake command patterns (case-sensitive as written in CMakeLists.txt) +_PROJECT = re.compile(r"^\s*project\s*\(\s*([^)\s]+)", re.IGNORECASE) +_ADD_LIB = re.compile(r"^\s*add_library\s*\(\s*([^)\s]+)", re.IGNORECASE) +_ADD_EXE = re.compile(r"^\s*add_executable\s*\(\s*([^)\s]+)", re.IGNORECASE) +_ADD_SUBDIR = re.compile(r"^\s*add_subdirectory\s*\(\s*([^)\s]+)", re.IGNORECASE) +_TARGET_LINK = re.compile(r"^\s*target_link_libraries\s*\(", re.IGNORECASE) + +# Keywords that are scope modifiers, not library names +_CMAKE_KEYWORDS = frozenset({"PUBLIC", "PRIVATE", "INTERFACE"}) + +# #include directives +_INCLUDE = re.compile(r'^\s*#\s*include\s*["<]([^">]+)[">]') + + +class CppResolver: + name = "cpp" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("CMakeLists.txt",) + fingerprint_suffixes = (".c", ".cc", ".cpp", ".cxx", ".h", ".hpp") + + def matches(self, repo_root: Path) -> bool: + return any(True for _ in walk_files( + repo_root, names=("CMakeLists.txt",), stop_at_nested_repos=True)) + + def exposed_names(self, repo_root: Path) -> set[str]: + names: set[str] = set() + for cmake in walk_files(repo_root, names=("CMakeLists.txt",), stop_at_nested_repos=True): + try: + lines = read_source_text(cmake, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + for line in lines: + for pat in (_PROJECT, _ADD_LIB, _ADD_EXE): + m = pat.match(line) + if m: + names.add(m.group(1)) + break + return names + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + + # Manifest edges from CMakeLists.txt + for cmake in walk_files(repo_root, names=("CMakeLists.txt",), stop_at_nested_repos=True): + try: + lines = read_source_text(cmake, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = cmake.relative_to(repo_root).as_posix() + i, n = 0, len(lines) + while i < n: + line = lines[i] + # add_subdirectory(dir): dir is treated as a target name + m = _ADD_SUBDIR.match(line) + if m: + edges.append(RawEdge(m.group(1), "manifest", rel, i + 1, line.strip())) + i += 1 + continue + # target_link_libraries(target [PUBLIC|PRIVATE|INTERFACE] lib1 lib2 ...), + # accumulated across lines until the closing paren (CMake spreads these). + if _TARGET_LINK.match(line): + start = i + buf = line + while ")" not in buf and i + 1 < n: + i += 1 + buf += " " + lines[i] + inner = buf.split("(", 1)[1].split(")", 1)[0] + toks = inner.split() + for token in toks[1:]: # toks[0] is the target itself + if token not in _CMAKE_KEYWORDS: + edges.append(RawEdge(token, "manifest", rel, start + 1, + lines[start].strip())) + i += 1 + continue + i += 1 + + # Import edges from C/C++ source files + src_suffixes = (".c", ".cc", ".cpp", ".cxx", ".h", ".hpp") + for src in walk_files(repo_root, suffixes=src_suffixes, stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _INCLUDE.match(line) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, line.strip())) + + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/csharp.py b/client-plugin/server/src/index_graph/graph/resolvers/csharp.py new file mode 100644 index 0000000..dd899df --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/csharp.py @@ -0,0 +1,77 @@ +"""C#/.NET ecosystem resolver: .csproj manifest + using-statement scan.""" +from __future__ import annotations + +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_ASSEMBLY_NAME = re.compile(r"\s*([^<]+)\s*") +_ROOT_NAMESPACE = re.compile(r"\s*([^<]+)\s*") +_PKG_REF = re.compile(r']*Include="([^"]+)"', re.IGNORECASE) +_PROJ_REF = re.compile(r']*Include="([^"]+)"', re.IGNORECASE) +_USING = re.compile(r"^\s*using\s+([\w.]+)\s*;") + + +class CSharpResolver: + name = "csharp" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_suffixes = (".csproj", ".cs") + + def matches(self, repo_root: Path) -> bool: + # walk_files is fail-closed (never raises on a permission-denied subdir) + # and pruned; rglob would propagate an OSError out and crash the graph build. + return any(walk_files(repo_root, suffixes=(".csproj",), stop_at_nested_repos=True)) + + def exposed_names(self, repo_root: Path) -> set[str]: + names: set[str] = set() + for csproj in walk_files(repo_root, suffixes=(".csproj",), stop_at_nested_repos=True): + names.add(csproj.stem) + try: + text = read_source_text(csproj, encoding="utf-8", errors="replace") + except OSError: + continue + for pattern in (_ASSEMBLY_NAME, _ROOT_NAMESPACE): + m = pattern.search(text) + if m: + names.add(m.group(1).strip()) + return names + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + + # manifest edges: parse each .csproj for PackageReference and ProjectReference + for csproj in walk_files(repo_root, suffixes=(".csproj",), stop_at_nested_repos=True): + try: + lines = read_source_text(csproj, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = csproj.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _PKG_REF.search(line) + if m: + pkg = m.group(1).strip() + edges.append(RawEdge(pkg, "manifest", rel, i, line.strip())) + continue + m = _PROJ_REF.search(line) + if m: + # target is the stem of the referenced .csproj path + ref_stem = Path(m.group(1).replace("\\", "/")).stem + edges.append(RawEdge(ref_stem, "manifest", rel, i, line.strip())) + + # import edges: scan .cs files for using statements + for src in walk_files(repo_root, suffixes=(".cs",), stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _USING.match(line) + if m: + # target is the top-level namespace segment + top = m.group(1).split(".")[0] + edges.append(RawEdge(top, "import", rel, i, line.strip())) + + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/go.py b/client-plugin/server/src/index_graph/graph/resolvers/go.py new file mode 100644 index 0000000..4ca77fb --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/go.py @@ -0,0 +1,87 @@ +"""Go ecosystem resolver: go.mod requires + an import-path scan.""" +from __future__ import annotations + +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_MODULE = re.compile(r"^\s*module\s+(\S+)") +_REQUIRE_SINGLE = re.compile(r"^\s*require\s+(\S+)\s+\S+") +_IMPORT_SINGLE = re.compile(r'^\s*import\s+(?:[A-Za-z0-9_.]+\s+)?"([^"]+)"') +_GROUPED_IMPORT = re.compile(r'^\s*(?:[A-Za-z0-9_.]+\s+)?"([^"]+)"') + + +class GoResolver: + name = "go" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("go.mod",) + fingerprint_suffixes = (".go",) + + def matches(self, repo_root: Path) -> bool: + return (repo_root / "go.mod").is_file() + + def exposed_names(self, repo_root: Path) -> set[str]: + gm = repo_root / "go.mod" + try: + for line in read_source_text(gm, encoding="utf-8", errors="replace").splitlines(): + m = _MODULE.match(line) + if m: + return {m.group(1)} + except OSError: + pass + return set() + + def _require_paths(self, text: str) -> list[str]: + out: list[str] = [] + in_block = False + for line in text.splitlines(): + s = line.strip() + if s.startswith("require (") or s == "require (": + in_block = True + continue + if in_block: + if s == ")": + in_block = False + elif s and not s.startswith("//"): + out.append(s.split()[0]) + continue + m = _REQUIRE_SINGLE.match(line) + if m: + out.append(m.group(1)) + return out + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + gm = repo_root / "go.mod" + if gm.is_file(): + try: + for path in self._require_paths(read_source_text(gm, encoding="utf-8", errors="replace")): + edges.append(RawEdge(path, "manifest", "go.mod", None, f"require {path}")) + except OSError: + pass + for src in walk_files(repo_root, suffixes=(".go",), stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + in_block = False + for i, line in enumerate(lines, 1): + s = line.strip() + if s.startswith("import ("): + in_block = True + continue + if in_block: + if s == ")": + in_block = False + continue + m = _GROUPED_IMPORT.match(s) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, s)) + continue + m = _IMPORT_SINGLE.match(line) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, s)) + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/java.py b/client-plugin/server/src/index_graph/graph/resolvers/java.py new file mode 100644 index 0000000..0299214 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/java.py @@ -0,0 +1,92 @@ +"""Java ecosystem resolver: Maven pom.xml + best-effort Gradle. Manifest-only.""" +from __future__ import annotations + +import re +from pathlib import Path +from xml.etree import ElementTree as ET + +from ..walk import read_source_bytes, read_source_text, walk_files +from .base import RawEdge + +_GRADLE_FILES = ("build.gradle", "build.gradle.kts") +_GRADLE_DEP = re.compile( + r"""(?:implementation|api|compileOnly|runtimeOnly|testImplementation)\s*[(\s]\s*""" + r"""['"]([^'":]+:[^'":]+):[^'"]+['"]""") + + +def _local(tag: str) -> str: + return tag.rsplit("}", 1)[-1] # strip the {namespace} + + +def _pom_coords(pom: Path) -> tuple[str, str] | None: + """The pom's own (groupId, artifactId), falling back to .""" + try: + root = ET.fromstring(read_source_bytes(pom)) + except (ET.ParseError, OSError): + return None + group = artifact = parent_group = None + for child in root: + t = _local(child.tag) + if t == "groupId": + group = (child.text or "").strip() + elif t == "artifactId": + artifact = (child.text or "").strip() + elif t == "parent": + for pc in child: + if _local(pc.tag) == "groupId": + parent_group = (pc.text or "").strip() + group = group or parent_group + return (group, artifact) if (group and artifact) else None + + +class JavaResolver: + name = "java" + # files whose content feeds the graph (read by the freshness fingerprint); manifest-only + fingerprint_names = ("pom.xml", "build.gradle", "build.gradle.kts") + + def matches(self, repo_root: Path) -> bool: + return ((repo_root / "pom.xml").is_file() + or any((repo_root / g).is_file() for g in _GRADLE_FILES)) + + def exposed_names(self, repo_root: Path) -> set[str]: + names: set[str] = set() + for pom in walk_files(repo_root, names=("pom.xml",), stop_at_nested_repos=True): + coords = _pom_coords(pom) + if coords: + names.add(f"{coords[0]}:{coords[1]}") + return names + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + for pom in walk_files(repo_root, names=("pom.xml",), stop_at_nested_repos=True): + try: + root = ET.fromstring(read_source_bytes(pom)) + except (ET.ParseError, OSError): + continue + rel = pom.relative_to(repo_root).as_posix() + for dep in root.iter(): + if _local(dep.tag) != "dependency": + continue + group = artifact = None + for c in dep: + t = _local(c.tag) + if t == "groupId": + group = (c.text or "").strip() + elif t == "artifactId": + artifact = (c.text or "").strip() + if group and artifact: + coord = f"{group}:{artifact}" + edges.append(RawEdge(coord, "manifest", rel, None, coord)) + for gf in _GRADLE_FILES: + gp = repo_root / gf + if not gp.is_file(): + continue + try: + lines = read_source_text(gp, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + for i, line in enumerate(lines, 1): + m = _GRADLE_DEP.search(line) + if m: + edges.append(RawEdge(m.group(1), "manifest", gf, i, line.strip())) + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/javascript.py b/client-plugin/server/src/index_graph/graph/resolvers/javascript.py new file mode 100644 index 0000000..5971072 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/javascript.py @@ -0,0 +1,67 @@ +"""JavaScript/TypeScript resolver: package.json + conservative import scan.""" +from __future__ import annotations + +import json +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_EXTS = (".js", ".jsx", ".ts", ".tsx", ".mjs", ".cjs") +_IMPORT = re.compile(r"""(?:import\s[^'"]*?from\s*|import\s*|require\(\s*|import\(\s*)['"]([^'"]+)['"]""") + + +def _bare_package(spec: str) -> str | None: + """Return the package name for a bare specifier, else None for relative/absolute paths.""" + if spec.startswith(".") or spec.startswith("/"): + return None + parts = spec.split("/") + if spec.startswith("@") and len(parts) >= 2: + return "/".join(parts[:2]) + return parts[0] + + +class JavaScriptResolver: + name = "javascript" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("package.json",) + fingerprint_suffixes = _EXTS + + def matches(self, repo_root: Path) -> bool: + return (repo_root / "package.json").is_file() + + def exposed_names(self, repo_root: Path) -> set[str]: + pj = repo_root / "package.json" + if not pj.is_file(): + return set() + try: + data = json.loads(read_source_text(pj, encoding="utf-8", errors="replace")) + except (json.JSONDecodeError, OSError): + return set() + name = data.get("name") + return {str(name)} if name else set() + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + pj = repo_root / "package.json" + if pj.is_file(): + try: + data = json.loads(read_source_text(pj, encoding="utf-8", errors="replace")) + for field in ("dependencies", "devDependencies", "peerDependencies"): + for name, spec in (data.get(field, {}) or {}).items(): + edges.append(RawEdge(str(name), "manifest", "package.json", None, f"{name}: {spec}")) + except (json.JSONDecodeError, OSError): + pass + for src in walk_files(repo_root, suffixes=_EXTS, stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + for m in _IMPORT.finditer(line): + pkg = _bare_package(m.group(1)) + if pkg: + edges.append(RawEdge(pkg, "import", rel, i, line.strip())) + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/php.py b/client-plugin/server/src/index_graph/graph/resolvers/php.py new file mode 100644 index 0000000..ec25801 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/php.py @@ -0,0 +1,62 @@ +"""PHP ecosystem resolver: composer.json require/require-dev + use-statement scan.""" +from __future__ import annotations + +import json +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +# `use Ns\Thing;`, and also `use function Ns\fn;` / `use const Ns\C;`: a symbol +# import still depends on namespace Ns, so the optional function/const modifier is +# consumed and the leading namespace segment is captured as the target. The trailing +# \s+ in the modifier keeps a namespace that merely starts with "function" (say +# `use functional\X;`) from being mistaken for the keyword. +_USE_STMT = re.compile( + r"^\s*use\s+(?:function\s+|const\s+)?\\?([A-Za-z_][A-Za-z0-9_]*)(?:\\[^;]*)?\s*;" +) + + +class PhpResolver: + name = "php" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("composer.json",) + fingerprint_suffixes = (".php",) + + def matches(self, repo_root: Path) -> bool: + return (repo_root / "composer.json").is_file() + + def exposed_names(self, repo_root: Path) -> set[str]: + cj = repo_root / "composer.json" + if not cj.is_file(): + return set() + try: + data = json.loads(read_source_text(cj, encoding="utf-8", errors="replace")) + except (json.JSONDecodeError, OSError): + return set() + name = data.get("name") + return {str(name)} if name else set() + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + cj = repo_root / "composer.json" + if cj.is_file(): + try: + data = json.loads(read_source_text(cj, encoding="utf-8", errors="replace")) + for field in ("require", "require-dev"): + for pkg, spec in (data.get(field, {}) or {}).items(): + edges.append(RawEdge(str(pkg), "manifest", "composer.json", None, f"{pkg}: {spec}")) + except (json.JSONDecodeError, OSError): + pass + for src in walk_files(repo_root, suffixes=(".php",), stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _USE_STMT.match(line) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, line.strip())) + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/python.py b/client-plugin/server/src/index_graph/graph/resolvers/python.py new file mode 100644 index 0000000..7b1a463 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/python.py @@ -0,0 +1,140 @@ +"""Python ecosystem resolver: manifests + AST import scan.""" +from __future__ import annotations + +import ast +import configparser +import re +import tomllib +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_PEP508_NAME = re.compile(r"^\s*([A-Za-z0-9][A-Za-z0-9._-]*)") +_MANIFESTS = ("pyproject.toml", "setup.cfg", "setup.py") + + +def _dep_name(spec: str) -> str | None: + m = _PEP508_NAME.match(spec) + return m.group(1) if m else None + + +class PythonResolver: + name = "python" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("pyproject.toml", "setup.cfg", "setup.py") + fingerprint_suffixes = (".py",) + fingerprint_globs = ("requirements*.txt",) + + def matches(self, repo_root: Path) -> bool: + if any((repo_root / m).is_file() for m in _MANIFESTS): + return True + return any(repo_root.glob("requirements*.txt")) + + def exposed_names(self, repo_root: Path) -> set[str]: + names: set[str] = set() + pp = repo_root / "pyproject.toml" + if pp.is_file(): + try: + data = tomllib.loads(read_source_text(pp, encoding="utf-8", errors="replace")) + proj = data.get("project", {}) + if isinstance(proj, dict) and proj.get("name"): + names.add(str(proj["name"])) + except (tomllib.TOMLDecodeError, OSError): + pass + cfg = repo_root / "setup.cfg" + if cfg.is_file(): + try: + cp = configparser.ConfigParser() + cp.read_string(read_source_text(cfg, encoding="utf-8", errors="strict")) + if cp.has_option("metadata", "name"): + names.add(cp.get("metadata", "name")) + except (configparser.Error, OSError): + pass + # top-level importable packages/modules (repo root and src/) + for base in (repo_root, repo_root / "src"): + if not base.is_dir(): + continue + for child in base.iterdir(): + if child.is_dir() and (child / "__init__.py").is_file(): + names.add(child.name) + elif child.suffix == ".py" and child.stem not in {"setup", "conftest"}: + names.add(child.stem) + return names + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + edges += self._manifest_edges(repo_root) + edges += self._import_edges(repo_root) + return edges + + def _manifest_edges(self, repo_root: Path) -> list[RawEdge]: + out: list[RawEdge] = [] + pp = repo_root / "pyproject.toml" + if pp.is_file(): + try: + text = read_source_text(pp, encoding="utf-8", errors="replace") + data = tomllib.loads(text) + proj = data.get("project", {}) + deps = list(proj.get("dependencies", []) or []) + for group in (proj.get("optional-dependencies", {}) or {}).values(): + deps += list(group or []) + for spec in deps: + name = _dep_name(str(spec)) + if name: + line = _pyproject_dep_line(text, str(spec)) + out.append(RawEdge(name, "manifest", "pyproject.toml", line, str(spec))) + except (tomllib.TOMLDecodeError, OSError): + pass + for req in sorted(repo_root.glob("requirements*.txt")): + try: + for i, line in enumerate(read_source_text(req, encoding="utf-8", errors="replace").splitlines(), 1): + s = line.strip() + if not s or s.startswith(("#", "-")): + continue + name = _dep_name(s) + if name: + out.append(RawEdge(name, "manifest", req.name, i, s)) + except OSError: + pass + return out + + def _import_edges(self, repo_root: Path) -> list[RawEdge]: + out: list[RawEdge] = [] + for py in walk_files(repo_root, suffixes=(".py",), stop_at_nested_repos=True): + try: + tree = ast.parse(read_source_text(py, encoding="utf-8", errors="replace")) + except (OSError, SyntaxError, ValueError): + continue + rel = py.relative_to(repo_root).as_posix() + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for a in node.names: + top = a.name.split(".")[0] + out.append(RawEdge(top, "import", rel, node.lineno, f"import {a.name}")) + elif isinstance(node, ast.ImportFrom): + if node.level == 0 and node.module: + top = node.module.split(".")[0] + out.append(RawEdge(top, "import", rel, node.lineno, f"from {node.module} import ...")) + return out + + +def _pyproject_dep_line(text: str, spec: str) -> int | None: + variants = (f'"{spec}"', f"'{spec}'") + in_optional = False + in_array = False + for i, line in enumerate(text.splitlines(), 1): + stripped = line.strip() + if stripped.startswith("[") and stripped.endswith("]"): + in_optional = stripped == "[project.optional-dependencies]" + in_array = False + starts_deps = stripped.startswith("dependencies") + starts_optional_group = in_optional and "=" in stripped and not stripped.startswith("#") + candidate = starts_deps or starts_optional_group or in_array + if candidate and any(v in line for v in variants): + return i + if starts_deps or starts_optional_group: + in_array = "[" in stripped and "]" not in stripped + elif in_array and "]" in stripped: + in_array = False + return None diff --git a/client-plugin/server/src/index_graph/graph/resolvers/ruby.py b/client-plugin/server/src/index_graph/graph/resolvers/ruby.py new file mode 100644 index 0000000..ac76cb7 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/ruby.py @@ -0,0 +1,70 @@ +"""Ruby ecosystem resolver: Gemfile gem declarations + require/require_relative scan.""" +from __future__ import annotations + +import re +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_GEM_LINE = re.compile(r"""^\s*gem\s+['"]([^'"]+)['"]""") +_REQUIRE = re.compile(r"""^\s*require\s+['"]([^'"]+)['"]""") +_REQUIRE_RELATIVE = re.compile(r"""^\s*require_relative\s+['"]([^'"]+)['"]""") +_SPEC_NAME = re.compile(r"""(?:spec|s)\.name\s*=\s*['"]([^'"]+)['"]""") + + +class RubyResolver: + name = "ruby" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("Gemfile",) + fingerprint_suffixes = (".rb", ".gemspec") + + def matches(self, repo_root: Path) -> bool: + if (repo_root / "Gemfile").is_file(): + return True + return any(repo_root.glob("*.gemspec")) + + def exposed_names(self, repo_root: Path) -> set[str]: + for gemspec in repo_root.glob("*.gemspec"): + try: + text = read_source_text(gemspec, encoding="utf-8", errors="replace") + except OSError: + continue + m = _SPEC_NAME.search(text) + if m: + return {m.group(1)} + return {repo_root.name} + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + + # manifest edges: Gemfile + gemfile = repo_root / "Gemfile" + if gemfile.is_file(): + try: + lines = read_source_text(gemfile, encoding="utf-8", errors="replace").splitlines() + except OSError: + lines = [] + for i, line in enumerate(lines, 1): + m = _GEM_LINE.match(line) + if m: + gem = m.group(1) + edges.append(RawEdge(gem, "manifest", "Gemfile", i, line.strip())) + + # import edges: .rb source files + for src in walk_files(repo_root, suffixes=(".rb",), stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _REQUIRE.match(line) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, line.strip())) + continue + m = _REQUIRE_RELATIVE.match(line) + if m: + edges.append(RawEdge(m.group(1), "import", rel, i, line.strip())) + + return edges diff --git a/client-plugin/server/src/index_graph/graph/resolvers/rust.py b/client-plugin/server/src/index_graph/graph/resolvers/rust.py new file mode 100644 index 0000000..cb32fa9 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/resolvers/rust.py @@ -0,0 +1,65 @@ +"""Rust ecosystem resolver: Cargo.toml manifests + a use/extern-crate scan.""" +from __future__ import annotations + +import re +import tomllib +from collections.abc import Iterator +from pathlib import Path + +from ..walk import read_source_text, walk_files +from .base import RawEdge + +_DEP_TABLES = ("dependencies", "dev-dependencies", "build-dependencies") +_USE = re.compile(r"^\s*use\s+([A-Za-z_][A-Za-z0-9_]*)") +_EXTERN = re.compile(r"^\s*extern\s+crate\s+([A-Za-z_][A-Za-z0-9_]*)") +_INTRA = {"crate", "self", "super"} # path roots that name the current crate, not a dep + + +class RustResolver: + name = "rust" + # files whose content feeds the graph (read by the freshness fingerprint) + fingerprint_names = ("Cargo.toml",) + fingerprint_suffixes = (".rs",) + + def matches(self, repo_root: Path) -> bool: + return (repo_root / "Cargo.toml").is_file() + + def _manifests(self, repo_root: Path) -> Iterator[Path]: + return walk_files(repo_root, names=("Cargo.toml",), stop_at_nested_repos=True) + + def exposed_names(self, repo_root: Path) -> set[str]: + names: set[str] = set() + for ct in self._manifests(repo_root): + try: + data = tomllib.loads(read_source_text(ct, encoding="utf-8", errors="replace")) + except (tomllib.TOMLDecodeError, OSError): + continue + pkg = data.get("package", {}) + if isinstance(pkg, dict) and pkg.get("name"): + names.add(str(pkg["name"])) + return names + + def raw_edges(self, repo_root: Path) -> list[RawEdge]: + edges: list[RawEdge] = [] + for ct in self._manifests(repo_root): + try: + data = tomllib.loads(read_source_text(ct, encoding="utf-8", errors="replace")) + except (tomllib.TOMLDecodeError, OSError): + continue + rel = ct.relative_to(repo_root).as_posix() + for table in _DEP_TABLES: + section = data.get(table, {}) + if isinstance(section, dict): + for name in section: + edges.append(RawEdge(str(name), "manifest", rel, None, f"{table}.{name}")) + for src in walk_files(repo_root, suffixes=(".rs",), stop_at_nested_repos=True): + try: + lines = read_source_text(src, encoding="utf-8", errors="replace").splitlines() + except OSError: + continue + rel = src.relative_to(repo_root).as_posix() + for i, line in enumerate(lines, 1): + m = _USE.match(line) or _EXTERN.match(line) + if m and m.group(1) not in _INTRA: + edges.append(RawEdge(m.group(1), "import", rel, i, line.strip())) + return edges diff --git a/client-plugin/server/src/index_graph/graph/roles.py b/client-plugin/server/src/index_graph/graph/roles.py new file mode 100644 index 0000000..1eaa622 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/roles.py @@ -0,0 +1,65 @@ +"""Topology-derived structural roles + salience faithfulness (ported).""" +from __future__ import annotations + +from .edges import Edge + + +def structural_salience(edges: list[Edge]) -> dict[str, dict]: + indeg: dict[str, int] = {} + outdeg: dict[str, int] = {} + for e in edges: + if e.external or e.to_repo is None: + continue + outdeg[e.from_repo] = outdeg.get(e.from_repo, 0) + 1 + indeg[e.to_repo] = indeg.get(e.to_repo, 0) + 1 + nodes = set(indeg) | set(outdeg) + max_in = max(indeg.values(), default=0) + out: dict[str, dict] = {} + for n in sorted(nodes): + i, o = indeg.get(n, 0), outdeg.get(n, 0) + out[n] = {"in_degree": i, "out_degree": o, "hub": i == max_in and i >= 2} + return out + + +def derive_roles(repo_names: set[str], edges: list[Edge], + markers: dict[str, set[str]]) -> dict[str, tuple[str, ...]]: + sal = structural_salience(edges) + max_out = max((s["out_degree"] for s in sal.values()), default=0) + roles: dict[str, list[str]] = {} + for name in sorted(repo_names): + s = sal.get(name, {"in_degree": 0, "out_degree": 0, "hub": False}) + mk = markers.get(name, set()) + rs: list[str] = [] + if "entry" in mk: + rs.append("entrypoint") + if "published" in mk and s["in_degree"] >= 1 and "entry" not in mk: + rs.append("library") + if s["hub"]: + rs.append("hub") + if s["out_degree"] == max_out and s["out_degree"] >= 3: + rs.append("orchestrator") + if s["in_degree"] == 0 and s["out_degree"] == 0 and name in markers: + rs.append("leaf") + if name not in markers: + rs.append("isolated") + roles[name] = tuple(rs) + return roles + + +def salience_audit(salience: dict[str, dict], marked: dict[str, list[str]]) -> list[dict]: + hubs = sorted(n for n, s in salience.items() if s["hub"]) + warns: list[dict] = [] + for name, mk in sorted(marked.items()): + if name not in hubs: + warns.append({"kind": "decorative-non-hub", "node": name, "markers": mk, + "in_degree": salience.get(name, {}).get("in_degree", 0), + "hubs": hubs, + "note": "marked node is not the structural hub; a render must not let " + "its marker outshine the hub(s)"}) + for h in hubs: + if h not in marked: + warns.append({"kind": "unmarked-hub", "node": h, + "in_degree": salience.get(h, {}).get("in_degree", 0), + "note": "structural convergence hub carries no marker; a faithful render " + "should make it central"}) + return warns diff --git a/client-plugin/server/src/index_graph/graph/walk.py b/client-plugin/server/src/index_graph/graph/walk.py new file mode 100644 index 0000000..7b22f08 --- /dev/null +++ b/client-plugin/server/src/index_graph/graph/walk.py @@ -0,0 +1,317 @@ +"""Pruned filesystem walks; complete graph scopes fail on source I/O errors.""" +from __future__ import annotations + +import fnmatch +import hashlib +import io +import os +import threading +from contextlib import contextmanager +from collections.abc import Iterator, Mapping, Sequence +from dataclasses import dataclass +from pathlib import Path + +EXCLUDE_DIRS = frozenset({ + ".git", ".hg", ".svn", ".venv", "venv", "env", + "venvs", "node_modules", "site-packages", "lib64", "__pycache__", + ".tox", ".mypy_cache", ".pytest_cache", ".ruff_cache", + "build", "dist", ".eggs", ".cache", ".playwright-mcp", + ".warden-safe-cache", ".next", ".turbo", + "target", "coverage", ".coverage", ".nyc_output", + ".parcel-cache", ".svelte-kit", ".angular", ".expo", + ".gradle", ".idea", ".vscode", ".yarn", ".pnpm-store", + ".terraform", "out", +}) + +_LOCAL = threading.local() +_SOURCE_CACHE_MAX_BYTES = 16 * 1024 * 1024 +_SOURCE_CACHE_MAX_FILES = 4096 +_SOURCE_DIGEST_JOURNAL_MAX_ENTRIES = 100_000 + + +class GraphSourceError(RuntimeError): + """A complete graph cannot be derived because a source could not be read.""" + + +@dataclass(frozen=True) +class DirectoryMembershipSnapshot: + path: Path + entries: tuple[tuple[str, str], ...] + + +@dataclass(frozen=True) +class PreloadedFileListing: + files: tuple[Path, ...] + directory_snapshots: tuple[DirectoryMembershipSnapshot, ...] = () + captured_root: Path | None = None + + +PreloadedFileValue = Sequence[Path | str] | PreloadedFileListing + + +def _source_error(operation: str, path: Path, error: OSError) -> None: + if getattr(_LOCAL, "file_cache", None) is not None: + _LOCAL.source_reuse_complete = False + # Do not copy raw OS diagnostics or absolute private paths into receipts. + raise GraphSourceError( + f"graph source {operation} failed: {path.name!r} ({type(error).__name__})" + ) from None + + +def _walk_error(error: OSError) -> None: + _source_error("traversal", Path(error.filename) if error.filename else Path("source"), error) + + +def _directory_entry_kind(entry: os.DirEntry) -> str: + try: + if entry.is_dir(follow_symlinks=False): + return "dir" + if entry.is_file(follow_symlinks=False): + return "file" + except OSError: + return "unreadable" + return "other" + + +def directory_membership_from_walk( + path: Path, + dirnames: Sequence[str], + filenames: Sequence[str], +) -> DirectoryMembershipSnapshot: + entries = [(name, "dir") for name in dirnames] + entries.extend((name, "file") for name in filenames) + return DirectoryMembershipSnapshot( + path=Path(path).resolve(), + entries=tuple(sorted(entries, key=lambda item: (item[0].lower(), item[0], item[1]))), + ) + + +def _scan_directory_membership(path: Path) -> tuple[tuple[str, str], ...]: + entries: list[tuple[str, str]] = [] + with os.scandir(path) as scanner: + for entry in scanner: + entries.append((entry.name, _directory_entry_kind(entry))) + return tuple(sorted(entries, key=lambda item: (item[0].lower(), item[0], item[1]))) + + +def _preloaded_is_current(listing: PreloadedFileListing) -> bool: + for snapshot in listing.directory_snapshots: + try: + current = _scan_directory_membership(snapshot.path) + except OSError: + return False + if current != snapshot.entries: + return False + return True + + +def _preloaded_file_listing(value: PreloadedFileValue) -> PreloadedFileListing: + if isinstance(value, PreloadedFileListing): + return value + return PreloadedFileListing(tuple(Path(path) for path in value)) + + +def _preloaded_file_cache( + file_lists: Mapping[Path | str, PreloadedFileValue] | None, +) -> dict[str, PreloadedFileListing]: + if not file_lists: + return {} + return { + str(Path(root).resolve()): _preloaded_file_listing(paths) + for root, paths in file_lists.items() + } + + +@contextmanager +def cached_file_scope(file_lists: Mapping[Path | str, PreloadedFileValue] | None = None): + """Share one pruned filesystem listing across resolver walks in one repo build.""" + previous = getattr(_LOCAL, "file_cache", None) + preloaded = _preloaded_file_cache(file_lists) + if previous is None: + _LOCAL.file_cache = dict(preloaded) + _LOCAL.source_cache = {} + _LOCAL.source_digest_journal = {} + _LOCAL.source_cache_bytes = 0 + _LOCAL.source_reuse_complete = True + elif preloaded: + _LOCAL.file_cache = {**previous, **preloaded} + try: + yield + finally: + if previous is None: + try: + delattr(_LOCAL, "file_cache") + delattr(_LOCAL, "source_cache") + delattr(_LOCAL, "source_digest_journal") + delattr(_LOCAL, "source_cache_bytes") + delattr(_LOCAL, "source_reuse_complete") + except AttributeError: + pass + else: + _LOCAL.file_cache = previous + + +def _record_unretained_source_read(key: str, data: bytes) -> None: + """Track consistency for source bytes that are too large to retain. + + The journal holds path digests, not file bodies. It is still bounded because + very large repositories can exceed byte and file retention at the same time. + If the journal bound is exceeded, persistent cache is disabled for the build + while source coverage continues from the bytes that were read. + """ + journal = getattr(_LOCAL, "source_digest_journal", None) + if journal is None: + _LOCAL.source_reuse_complete = False + return + digest = hashlib.sha256(data).hexdigest() + previous = journal.get(key) + if previous is None: + if len(journal) >= _SOURCE_DIGEST_JOURNAL_MAX_ENTRIES: + _LOCAL.source_reuse_complete = False + return + journal[key] = digest + elif previous != digest: + _LOCAL.source_reuse_complete = False + + +def read_source_bytes(path: Path) -> bytes: + """Reuse exact working bytes only within a build, with bounded memory. + + This never trusts Git metadata or a previous process's source bytes. Files + beyond the memory/file limit are read normally; coverage is unchanged. Their + first-read digest is journaled so repeated resolver reads can still prove + they saw the same bytes before the derived build is written to disk cache. + """ + cache = getattr(_LOCAL, "source_cache", None) + key = str(path.absolute()) + if cache is not None and key in cache: + return cache[key] + try: + data = path.read_bytes() + except OSError as error: + _source_error("read", path, error) + raise + if cache is not None: + can_retain = (len(cache) < _SOURCE_CACHE_MAX_FILES + and _LOCAL.source_cache_bytes + len(data) <= _SOURCE_CACHE_MAX_BYTES) + # Check previously journaled bytes even if this read now fits the byte + # cache. A file can shrink between fingerprinting and resolver parsing. + if key in _LOCAL.source_digest_journal or not can_retain: + _record_unretained_source_read(key, data) + if can_retain: + cache[key] = data + _LOCAL.source_cache_bytes += len(data) + return data + + +def source_reuse_complete() -> bool: + """Whether shared source reads remain consistent enough for cache persistence.""" + return getattr(_LOCAL, "source_reuse_complete", False) + + +def read_source_text(path: Path, *, encoding="utf-8", errors="replace") -> str: + # TextIOWrapper preserves Path.read_text's universal-newline behavior. + with io.TextIOWrapper(io.BytesIO(read_source_bytes(path)), + encoding=encoding, errors=errors) as stream: + return stream.read() + + +def _matches(filename: str, suffixes: tuple[str, ...] | None, + names: tuple[str, ...] | None, + globs: tuple[str, ...] | None = None) -> bool: + if names is not None and filename in names: + return True + if suffixes is not None and filename.endswith(suffixes): + return True + return globs is not None and any(fnmatch.fnmatchcase(filename, g) for g in globs) + + +def _within_captured_repo(path: Path, root: Path) -> bool: + physical = path.resolve() + if not physical.is_relative_to(root): + return False + while physical != root: + # A junction may enter below a nested repository's marker, bypassing + # the lexical walker's ordinary stop-at-.git check. + if os.path.lexists(physical / ".git"): + return False + physical = physical.parent + return True + + +def _scoped_all_files(root: Path, checkpoint=None) -> list[Path]: + """All local files under existing pruning and nested repository boundaries. + + Do not use Git ignore rules or index metadata here: ignored and index-hidden + working files can contain dependencies. Cache only inside one repo build. + """ + key = str(root.resolve()) + cache = getattr(_LOCAL, "file_cache", None) + contained_root = None + if cache is not None and key in cache: + cached = cache[key] + if isinstance(cached, PreloadedFileListing): + contained_root = cached.captured_root + if contained_root is not None and root.resolve() != contained_root: + raise GraphSourceError("repository root changed after router inventory") + if _preloaded_is_current(cached): + files = list(cached.files) + cache[key] = files + return files + del cache[key] + else: + return cached + out: list[Path] = [] + for dirpath, dirnames, filenames in os.walk(root, onerror=_walk_error): + current = Path(dirpath) + # A stale router preload must not widen its original physical scope + # when the fallback walker encounters a repository-internal junction. + if contained_root is not None and not _within_captured_repo(current, contained_root): + dirnames[:] = [] + continue + if checkpoint is not None and not checkpoint(current): + # An interrupted listing is never reusable as a complete scope. + return out + if current != root and (".git" in dirnames or ".git" in filenames): + dirnames[:] = [] + continue + dirnames[:] = sorted((d for d in dirnames if d not in EXCLUDE_DIRS), key=str.lower) + out.extend(current / filename for filename in sorted(filenames)) + if cache is not None: + cache[key] = out + return out + + +def walk_files(root: Path, suffixes: tuple[str, ...] | None = None, + names: tuple[str, ...] | None = None, + checkpoint=None, + globs: tuple[str, ...] | None = None, + stop_at_nested_repos: bool = False) -> Iterator[Path]: + """Yield files under `root`, pruning EXCLUDE_DIRS. + + Match by `suffixes` (e.g. (".py",)) or exact `names` (e.g. ("__main__.py",)). + Outside graph-build scopes, a missing/unreadable root yields nothing. Inside + a scoped graph build, traversal errors raise GraphSourceError so omitted + sources cannot become a successful complete graph or cached file listing. + + When `stop_at_nested_repos` is true, a child directory with a .git marker is + treated as a separate repository boundary. The root itself is still scanned. + """ + root = Path(root) + if stop_at_nested_repos: + for path in _scoped_all_files(root, checkpoint=checkpoint): + if checkpoint is not None and not checkpoint(path.parent): + break + if _matches(path.name, suffixes, names, globs): + yield path + return + + for dirpath, dirnames, filenames in os.walk(root, onerror=_walk_error): + current = Path(dirpath) + if checkpoint is not None and not checkpoint(current): + dirnames[:] = [] + break + dirnames[:] = sorted((d for d in dirnames if d not in EXCLUDE_DIRS), key=str.lower) + for fn in filenames: + if _matches(fn, suffixes, names, globs): + yield Path(dirpath) / fn diff --git a/client-plugin/server/src/index_graph/internals/__init__.py b/client-plugin/server/src/index_graph/internals/__init__.py new file mode 100644 index 0000000..599e1a0 --- /dev/null +++ b/client-plugin/server/src/index_graph/internals/__init__.py @@ -0,0 +1,10 @@ +"""Intra-repo module graph: see inside a repo, not only repo as atom.""" +from __future__ import annotations + +from .modules import ModuleNode, InternalEdge, Unresolved, discover_modules, extract_internal_edges +from .build import InternalGraph, Coverage, build_internals + +__all__ = [ + "ModuleNode", "InternalEdge", "Unresolved", "InternalGraph", "Coverage", + "discover_modules", "extract_internal_edges", "build_internals", +] diff --git a/client-plugin/server/src/index_graph/internals/build.py b/client-plugin/server/src/index_graph/internals/build.py new file mode 100644 index 0000000..fac29af --- /dev/null +++ b/client-plugin/server/src/index_graph/internals/build.py @@ -0,0 +1,76 @@ +"""Assemble an InternalGraph: modules, internal edges, cycles, fan-in/out, coverage.""" +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path + +from ..graph.edges import Edge +from ..graph.cycles import find_cycles +from .modules import ModuleNode, InternalEdge, Unresolved, discover_modules, _extract + + +@dataclass(frozen=True) +class Coverage: + """How structurally complete the module graph is, and what it could not see. + A verdict over this graph is sound only for what coverage marks resolved + (call-graph soundness: static analysis cannot see dynamic dispatch or + unparseable files, so it must say so rather than imply completeness).""" + modules: int + internal_edges: int + parse_errors: tuple[str, ...] + dynamic_imports: tuple[tuple[str, int | None], ...] + + @property + def complete(self) -> bool: + return not self.parse_errors and not self.dynamic_imports + + +@dataclass(frozen=True) +class InternalGraph: + repo: str + modules: tuple[ModuleNode, ...] + edges: tuple[InternalEdge, ...] + cycles: tuple[tuple[str, ...], ...] + fan_in: dict[str, int] + fan_out: dict[str, int] + coverage: Coverage + + +def _cycles(edges: tuple[InternalEdge, ...]) -> tuple[tuple[str, ...], ...]: + # Reuse the repo-level Tarjan SCC by constructing minimal internal Edges; + # find_cycles reads only from_repo/to_repo/external. + as_edges = [Edge(e.from_id, e.to_id, e.to_id, False, "high", ()) for e in edges] + return tuple(find_cycles(as_edges)) + + +def _coverage(modules: tuple[ModuleNode, ...], edges: tuple[InternalEdge, ...], + unresolved: list[Unresolved]) -> Coverage: + return Coverage( + modules=len(modules), + internal_edges=len(edges), + parse_errors=tuple(sorted({u.file for u in unresolved if u.reason == "parse_error"})), + dynamic_imports=tuple(sorted( + {(u.file, u.line) for u in unresolved if u.reason == "dynamic"}, + key=lambda t: (t[0], t[1] or 0))), + ) + + +def build_internals(repo_root: Path, repo_name: str | None = None) -> InternalGraph: + root = repo_root.resolve() + name = repo_name or root.name + modules = tuple(discover_modules(root)) + edge_list, unresolved = _extract(root, list(modules)) + edges = tuple(edge_list) + fan_out: dict[str, int] = {} + fan_in: dict[str, int] = {} + seen_out: set[tuple[str, str]] = set() + seen_in: set[tuple[str, str]] = set() + for e in edges: + if (e.from_id, e.to_id) not in seen_out: + seen_out.add((e.from_id, e.to_id)) + fan_out[e.from_id] = fan_out.get(e.from_id, 0) + 1 + if (e.to_id, e.from_id) not in seen_in: + seen_in.add((e.to_id, e.from_id)) + fan_in[e.to_id] = fan_in.get(e.to_id, 0) + 1 + return InternalGraph(name, modules, edges, _cycles(edges), fan_in, fan_out, + _coverage(modules, edges, unresolved)) diff --git a/client-plugin/server/src/index_graph/internals/modules.py b/client-plugin/server/src/index_graph/internals/modules.py new file mode 100644 index 0000000..ae25826 --- /dev/null +++ b/client-plugin/server/src/index_graph/internals/modules.py @@ -0,0 +1,337 @@ +"""Module discovery and intra-repo import extraction, per language. + +Python is AST-exact. JavaScript/TypeScript, Rust, and Go are best-effort and +file-level: relative or path-aligned imports resolve to sibling modules, bare +or external specifiers are ignored. Dynamic and aliased imports may be missed. +""" +from __future__ import annotations + +import ast +import re +from dataclasses import dataclass +from pathlib import Path + +from ..graph.walk import walk_files + + +@dataclass(frozen=True) +class ModuleNode: + id: str + path: str + language: str + + +@dataclass(frozen=True) +class InternalEdge: + from_id: str + to_id: str + evidence_file: str + evidence_line: int | None + raw: str + + +@dataclass(frozen=True) +class Unresolved: + """An import the static scan could not turn into a definite internal edge. + The soundness gap a verdict must be honest about (call-graph soundness: + static analysis cannot see dynamic dispatch or unparseable files). Dynamic + detection over-approximates toward reporting (a method coincidentally named + import_module may be flagged); over-reporting unverifiability is the safe + direction for a soundness gap, never under-reporting it.""" + file: str + line: int | None + reason: str # "parse_error" | "dynamic" + raw: str + + +_LANG_BY_SUFFIX = { + ".py": "python", + ".js": "javascript", ".jsx": "javascript", ".mjs": "javascript", + ".cjs": "javascript", ".ts": "typescript", ".tsx": "typescript", + ".rs": "rust", ".go": "go", +} +_SUFFIXES = tuple(_LANG_BY_SUFFIX) + + +def _strip_suffix(rel: str) -> str: + dot = rel.rfind(".") + slash = rel.rfind("/") + return rel[:dot] if dot > slash else rel + + +def discover_modules(repo_root: Path) -> list[ModuleNode]: + mods: list[ModuleNode] = [] + for f in walk_files(repo_root, suffixes=_SUFFIXES): + rel = f.relative_to(repo_root).as_posix() + lang = _LANG_BY_SUFFIX.get(f.suffix, "") + if not lang: + continue + mods.append(ModuleNode(id=_strip_suffix(rel), path=rel, language=lang)) + return sorted(mods, key=lambda m: m.id) + + +# --- Python (AST-exact) ---------------------------------------------------- + +def _id_for(base: str, ids: set[str]) -> str | None: + """Resolve a slash-path base to an internal module id (file or package).""" + if base in ids: + return base + pkg = base + "/__init__" + return pkg if pkg in ids else None + + +def _dotted_to_id(dotted: str, ids: set[str]) -> str | None: + return _id_for(dotted.replace(".", "/"), ids) + + +def _pkg_depth(repo_root: Path, importer_id: str) -> int: + """How many package levels a relative import may legally walk up from this + module: the run of __init__.py-bearing directories from the module's own + directory up to and including the repo root. A `from .x` needs depth >= 1, + `from ..x` needs depth >= 2, and so on. This makes resolution correct for + both a src-layout single package (the root is itself a package) and a + workspace of separate packages (the root is not).""" + parts = importer_id.split("/")[:-1] + depth = 0 + while True: + d = repo_root.joinpath(*parts) if parts else repo_root + if (d / "__init__.py").is_file(): + depth += 1 + if not parts: + break + parts = parts[:-1] + else: + break + return depth + + +def _resolve_relative(importer_id: str, level: int, module: str | None, + ids: set[str], depth: int) -> str | None: + if level > depth: + return None # walks above the top-level package: not an internal import + pkg_parts = importer_id.split("/")[:-1] + base = pkg_parts[:len(pkg_parts) - (level - 1)] + target = base + (module.split(".") if module else []) + if not target: + return None + return _id_for("/".join(target), ids) + + +def _python_edges(repo_root: Path, ids: set[str]) -> tuple[list[InternalEdge], list[Unresolved]]: + out: list[InternalEdge] = [] + unresolved: list[Unresolved] = [] + for py in walk_files(repo_root, suffixes=(".py",)): + rel = py.relative_to(repo_root).as_posix() + from_id = _strip_suffix(rel) + try: + tree = ast.parse(py.read_text(encoding="utf-8-sig")) + except OSError: + continue + except (SyntaxError, ValueError): + unresolved.append(Unresolved(rel, None, "parse_error", "")) + continue + depth = _pkg_depth(repo_root, from_id) + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for a in node.names: + tid = _dotted_to_id(a.name, ids) + if tid and tid != from_id: + out.append(InternalEdge(from_id, tid, rel, node.lineno, f"import {a.name}")) + elif isinstance(node, ast.ImportFrom): + if node.level and node.module is None: + if node.level <= depth: + pkg_parts = from_id.split("/")[:-1] + base = pkg_parts[:len(pkg_parts) - (node.level - 1)] + for a in node.names: + tid = _id_for("/".join([*base, a.name]), ids) + if tid and tid != from_id: + out.append(InternalEdge( + from_id, tid, rel, node.lineno, + f"from {'.' * node.level} import {a.name}")) + continue + if node.level and node.level > 0: + tid = _resolve_relative(from_id, node.level, node.module, ids, depth) + raw = f"from {'.' * node.level}{node.module or ''} import ..." + elif node.module: + tid = _dotted_to_id(node.module, ids) + raw = f"from {node.module} import ..." + else: + tid = None + raw = "" + if tid and tid != from_id: + out.append(InternalEdge(from_id, tid, rel, node.lineno, raw)) + elif isinstance(node, ast.Call): + fn = node.func + if (isinstance(fn, ast.Name) and fn.id == "__import__") or \ + (isinstance(fn, ast.Attribute) and fn.attr in ("import_module", "__import__")): + unresolved.append(Unresolved(rel, node.lineno, "dynamic", "dynamic import")) + return out, unresolved + + +# --- JavaScript / TypeScript (best-effort, relative specifiers) ------------ + +_JS_IMPORT = re.compile( + r"""(?:import|export)[^'"]*?from\s*['"]([^'"]+)['"]""" + r"""|require\(\s*['"]([^'"]+)['"]\s*\)""" + r"""|import\(\s*['"]([^'"]+)['"]\s*\)""") +_JS_DYNAMIC = re.compile(r"""(? str | None: + if not spec.startswith("."): + return None + base = (Path(importer_rel).parent / spec).as_posix() + parts: list[str] = [] + for seg in base.split("/"): + if seg in ("", "."): + continue + if seg == "..": + if not parts: + return None # the specifier escapes above the repo root + parts.pop() + continue + parts.append(seg) + cand = "/".join(parts) + if "." in Path(cand).name: + cand = _strip_suffix(cand) + if cand in ids: + return cand + idx = cand + "/index" + return idx if idx in ids else None + + +def _js_edges(repo_root: Path, ids: set[str]) -> tuple[list[InternalEdge], list[Unresolved]]: + out: list[InternalEdge] = [] + unresolved: list[Unresolved] = [] + for f in walk_files(repo_root, suffixes=_JS_SUFFIXES): + rel = f.relative_to(repo_root).as_posix() + from_id = _strip_suffix(rel) + try: + text = f.read_text(encoding="utf-8-sig") + except OSError: + continue + for i, line in enumerate(text.splitlines(), 1): + for m in _JS_IMPORT.finditer(line): + spec = m.group(1) or m.group(2) or m.group(3) + if not spec: + continue + tid = _js_resolve(rel, spec, ids) + if tid and tid != from_id: + out.append(InternalEdge(from_id, tid, rel, i, line.strip())) + if _JS_DYNAMIC.search(line): + unresolved.append(Unresolved(rel, i, "dynamic", line.strip())) + return out, unresolved + + +# --- Rust (best-effort, mod declarations) ---------------------------------- + +_RUST_MOD = re.compile(r"^\s*(?:pub\s+)?mod\s+([A-Za-z_][A-Za-z0-9_]*)\s*;") + + +def _rust_edges(repo_root: Path, ids: set[str]) -> tuple[list[InternalEdge], list[Unresolved]]: + out: list[InternalEdge] = [] + for f in walk_files(repo_root, suffixes=(".rs",)): + rel = f.relative_to(repo_root).as_posix() + from_id = _strip_suffix(rel) + parent = Path(rel).parent.as_posix() + parent = "" if parent == "." else parent + "/" + try: + text = f.read_text(encoding="utf-8-sig") + except OSError: + continue + for i, line in enumerate(text.splitlines(), 1): + m = _RUST_MOD.match(line) + if not m: + continue + name = m.group(1) + for cand in (f"{parent}{name}", f"{parent}{name}/mod", f"{from_id}/{name}"): + if cand in ids and cand != from_id: + out.append(InternalEdge(from_id, cand, rel, i, line.strip())) + break + return out, [] + + +# --- Go (best-effort, internal package imports) ---------------------------- + +_GO_MODULE = re.compile(r"^\s*module\s+(\S+)") +_GO_IMPORT_SINGLE = re.compile(r'^\s*import\s+"([^"]+)"') +_GO_IMPORT_BLOCK_LINE = re.compile(r'^\s*"([^"]+)"') + + +def _go_module_path(repo_root: Path) -> str | None: + gomod = repo_root / "go.mod" + if not gomod.is_file(): + return None + try: + for line in gomod.read_text(encoding="utf-8-sig").splitlines(): + m = _GO_MODULE.match(line) + if m: + return m.group(1) + except OSError: + return None + return None + + +def _go_edges(repo_root: Path, ids: set[str]) -> tuple[list[InternalEdge], list[Unresolved]]: + mod_path = _go_module_path(repo_root) + if not mod_path: + return [], [] + pkg_dirs = {Path(m).parent.as_posix() for m in ids} + out: list[InternalEdge] = [] + for f in walk_files(repo_root, suffixes=(".go",)): + rel = f.relative_to(repo_root).as_posix() + from_id = _strip_suffix(rel) + try: + lines = f.read_text(encoding="utf-8-sig").splitlines() + except OSError: + continue + in_block = False + for i, line in enumerate(lines, 1): + spec = None + if line.strip().startswith("import ("): + in_block = True + continue + if in_block: + if ")" in line: + in_block = False + m = _GO_IMPORT_BLOCK_LINE.match(line) + if m: + spec = m.group(1) + else: + m = _GO_IMPORT_SINGLE.match(line) + if m: + spec = m.group(1) + if spec and spec.startswith(mod_path + "/"): + sub = spec[len(mod_path) + 1:] + if sub and sub in pkg_dirs: + target = next((m2 for m2 in sorted(ids) + if Path(m2).parent.as_posix() == sub), None) + if target and target != from_id: + out.append(InternalEdge(from_id, target, rel, i, line.strip())) + return out, [] + + +# --- Dispatch -------------------------------------------------------------- + +def _extract(repo_root: Path, modules: list[ModuleNode]) -> tuple[list[InternalEdge], list[Unresolved]]: + by_lang: dict[str, set[str]] = {} + for m in modules: + by_lang.setdefault(m.language, set()).add(m.id) + js_ids = by_lang.get("javascript", set()) | by_lang.get("typescript", set()) + edges: list[InternalEdge] = [] + unresolved: list[Unresolved] = [] + for fn, lang_ids in ((_python_edges, by_lang.get("python", set())), + (_js_edges, js_ids), + (_rust_edges, by_lang.get("rust", set())), + (_go_edges, by_lang.get("go", set()))): + e, u = fn(repo_root, lang_ids) + edges += e + unresolved += u + edges.sort(key=lambda x: (x.from_id, x.to_id, x.evidence_file, x.evidence_line or 0)) + unresolved.sort(key=lambda x: (x.file, x.line or 0, x.reason)) + return edges, unresolved + + +def extract_internal_edges(repo_root: Path, modules: list[ModuleNode]) -> list[InternalEdge]: + return _extract(repo_root, modules)[0] diff --git a/client-plugin/server/src/index_graph/interop.py b/client-plugin/server/src/index_graph/interop.py new file mode 100644 index 0000000..bfa4f7c --- /dev/null +++ b/client-plugin/server/src/index_graph/interop.py @@ -0,0 +1,90 @@ +"""Interop: index's context envelopes as organ-bundle interchange entries. + +The organ bundle is the shared spine gather, crucible, forum, learn, and index +compose on. This module maps index's context envelopes (the budgeted, +receipt-backed context packs) into that entry shape, so an index envelope can +feed the agent loop, compose into forum's routing context, or back a learn +lesson through the shared spine. + +Entry shape matches the proof-surface organ-bundle contract +(entry_id, organ_id, receipt_kind, status, payload_sha256, summary, payload_ref). +gather/src/gather/interop.py is the reference implementation. +""" +from __future__ import annotations + +import hashlib +import re + +ORGAN = "index" +SPINE_KIND = "index-context-envelope" +STATUSES = frozenset({ + "pass", "fail", "unverified", "warn", "needs-human", "not-applicable", "unknown", +}) +_HEX = re.compile(r"^[0-9a-f]{64}$") +_FIELDS = ("entry_id", "organ_id", "receipt_kind", "status", "payload_sha256", + "summary", "payload_ref") + + +def _entry(entry_id: str, status: str, payload_sha256: str, summary: str, ref: str) -> dict: + return { + "entry_id": entry_id, + "organ_id": ORGAN, + "receipt_kind": SPINE_KIND, + "status": status, + "payload_sha256": payload_sha256, + "summary": summary[:160], + "payload_ref": ref, + } + + +def envelope_entry(envelope: dict, *, entry_id: str = "index-envelope-1", + ref: str = "index://context-envelope") -> dict: + """Map an index context-envelope output into an organ-bundle entry. + + The envelope is a JSON dict (schema project-telos.context-envelope/v1) + produced by `index context --json` or the `index.context.envelope` MCP tool. + """ + verification = envelope.get("verification_verdict", "UNVERIFIABLE") + selection = envelope.get("selection", {}) + mode = selection.get("mode", "?") if isinstance(selection, dict) else "?" + retained = selection.get("retained_names", []) if isinstance(selection, dict) else [] + repo_count = len(retained) + + status = "pass" if verification == "MATCH" else "warn" if verification else "unverified" + + # Use the envelope's own hash if present, else hash the canonical form + sha = envelope.get("envelope_sha256", "") + if not sha or not _HEX.match(sha): + import json + canonical = json.dumps(envelope, sort_keys=True, separators=(",", ":")) + sha = hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + summary = f"context-envelope: {repo_count} repos, mode={mode}, verdict={verification}" + return _entry(entry_id, status, sha, summary, ref) + + +def map_entry(repo: str, *, n_repos: int = 1, map_sha: str = "", + entry_id: str = "index-map-1", + ref: str = "index://map") -> dict: + """Map an index map result into an organ-bundle entry.""" + if not map_sha: + map_sha = hashlib.sha256(repo.encode("utf-8")).hexdigest() + summary = f"workspace map: {repo} ({n_repos} repos)" + return _entry(entry_id, "pass", map_sha, summary, ref) + + +def validate_entry(entry: dict) -> bool: + """Validate one organ-bundle entry shape. Returns True if well-formed.""" + if not isinstance(entry, dict): + return False + if set(entry.keys()) != set(_FIELDS): + return False + if entry["organ_id"] != ORGAN: + return False + if entry["receipt_kind"] != SPINE_KIND: + return False + if entry["status"] not in STATUSES: + return False + if not _HEX.match(entry.get("payload_sha256", "")): + return False + return True diff --git a/client-plugin/server/src/index_graph/knowledge/__init__.py b/client-plugin/server/src/index_graph/knowledge/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/client-plugin/server/src/index_graph/knowledge/atlas.py b/client-plugin/server/src/index_graph/knowledge/atlas.py new file mode 100644 index 0000000..68c0334 --- /dev/null +++ b/client-plugin/server/src/index_graph/knowledge/atlas.py @@ -0,0 +1,168 @@ +"""Assemble repos (index) + docs into the two-layer atlas pack.""" +from __future__ import annotations + +import re +from pathlib import Path + +from ..context.pack import to_json +from ..graph.build import DependencyGraph +from .docs import Doc, _norm # reuse the SAME normalizer that built link_targets + +_EDGE_SORT = lambda e: (e["from"], e["type"], e["to_kind"], e["to"]) +_TOKEN = re.compile(r"[0-9a-z]+") + + +def _target_index(repo_names, docs): + """normalized name -> (to_kind, id); repos win over docs on collision; first doc wins. + + Uses `_norm` (space/underscore -> dash, lowercased), the SAME normalization + docs.py applied to `[[link]]` targets, so multi-word links like [[Auth Design]] + resolve to a doc titled "Auth Design".""" + idx: dict[str, tuple[str, str]] = {} + for r in sorted(repo_names): + idx.setdefault(_norm(r), ("repo", r)) + for d in docs: + for cand in (d.title, Path(d.rel_path).stem): + idx.setdefault(_norm(cand), ("doc", d.rel_path)) + return idx + + +def _describes(doc: Doc, repo_dirs: dict[str, str]) -> str | None: + """The most-specific repo whose dir contains the doc's dir (by location), else None. + + A repo dir "" (repo at the workspace root) matches only a root-level doc + (dir_rel == ""), never docs in subdirs, since the prefix branch requires + a non-empty rdir.""" + best, best_len = None, -1 + for repo, rdir in repo_dirs.items(): + if doc.dir_rel == rdir or (rdir != "" and doc.dir_rel.startswith(rdir + "/")): + if len(rdir) > best_len: + best, best_len = repo, len(rdir) + return best + + +def _repo_dirs_by_location(repo_dirs: dict[str, str]) -> dict[str, str]: + """repo dir -> first repo name for current longest-prefix describes semantics.""" + by_dir: dict[str, str] = {} + for repo, rdir in repo_dirs.items(): + by_dir.setdefault(rdir, repo) + return by_dir + + +def _describes_indexed(doc: Doc, repo_by_dir: dict[str, str]) -> str | None: + if doc.dir_rel == "": + return repo_by_dir.get("") + current = doc.dir_rel + while True: + repo = repo_by_dir.get(current) + if repo is not None: + return repo + parent, sep, _name = current.rpartition("/") + if not sep: + return None + current = parent + + +def _mentions_name(body: str, name: str) -> bool: + # case-insensitive whole-token match; treat '-'/'_' and spaces as separators + pattern = r"(? frozenset[str]: + return frozenset(_TOKEN.findall(text.lower())) + + +def _doc_rows(docs: list[Doc]) -> list[dict]: + return [{"id": d.rel_path, "title": d.title, "dir": d.dir_rel} for d in docs] + + +def _describes_edges_indexed(docs: list[Doc], repo_dirs: dict[str, str]) -> list[dict]: + edges: list[dict] = [] + seen: set[tuple[str, str, str]] = set() + repo_by_dir = _repo_dirs_by_location(repo_dirs) + for d in docs: + repo = _describes_indexed(d, repo_by_dir) + if repo is None: + continue + key = (d.rel_path, "repo", repo) + if key not in seen: + seen.add(key) + edges.append({ + "type": "describes", + "from": d.rel_path, + "to": repo, + "to_kind": "repo", + }) + return sorted(edges, key=_EDGE_SORT) + + +def _describes_edges(docs: list[Doc], repo_dirs: dict[str, str]) -> list[dict]: + return _describes_edges_indexed(docs, repo_dirs) + + +def build_router_pack(graph: DependencyGraph, docs: list[Doc], + repo_dirs: dict[str, str]) -> dict: + pack = to_json(graph) + pack["docs"] = _doc_rows(docs) + pack["knowledge_edges"] = _describes_edges(docs, repo_dirs) + pack["knowledge_warnings"] = [] + pack["repo_dirs"] = dict(repo_dirs) + return pack + + +def build_atlas_pack(graph: DependencyGraph, docs: list[Doc], + repo_dirs: dict[str, str]) -> dict: + pack = to_json(graph) + repo_names = {n["name"] for n in pack["repos"]} + idx = _target_index(repo_names, docs) + + pack["docs"] = _doc_rows(docs) + edges: list[dict] = [] + warnings: list[str] = [] + seen: set[tuple[str, str, str]] = set() # (from, to_kind, to), strongest wins + + def add(etype: str, frm: str, to_kind: str, to: str) -> None: + key = (frm, to_kind, to) + if to is None or key in seen: + return + seen.add(key) + edges.append({"type": etype, "from": frm, "to": to, "to_kind": to_kind}) + + # describes (by location): strongest + for edge in _describes_edges(docs, repo_dirs): + add("describes", edge["from"], edge["to_kind"], edge["to"]) + # links-to (from [[wiki-links]]) + for d in docs: + for t in d.link_targets: + hit = idx.get(t) + if hit is None: + warnings.append(f"{d.rel_path}: unresolved [[{t}]]") + continue + to_kind, to = hit + if to_kind == "doc" and to == d.rel_path: + continue # self-link + add("links-to", d.rel_path, to_kind, to) + + # mentions (prose name-drops): weakest; deduped via `seen` against describes/links-to + name_of = {("repo", r): r for r in repo_names} + name_of.update({("doc", d.rel_path): d.title for d in docs}) + mention_targets = [ + (target, display, _mention_tokens(display)) + for target, display in sorted(name_of.items()) + ] + for d in docs: + body_tokens = _mention_tokens(d.body) + for (to_kind, to), display, display_tokens in mention_targets: + if (d.rel_path, to_kind, to) in seen: # already a stronger edge + continue + if to_kind == "doc" and to == d.rel_path: + continue + if display_tokens and not display_tokens.issubset(body_tokens): + continue + if _mentions_name(d.body, display): # display = repo name or doc title + add("mentions", d.rel_path, to_kind, to) + + pack["knowledge_edges"] = sorted(edges, key=_EDGE_SORT) + pack["knowledge_warnings"] = warnings + return pack diff --git a/client-plugin/server/src/index_graph/knowledge/docs.py b/client-plugin/server/src/index_graph/knowledge/docs.py new file mode 100644 index 0000000..2cc96e8 --- /dev/null +++ b/client-plugin/server/src/index_graph/knowledge/docs.py @@ -0,0 +1,92 @@ +"""Discover + parse workspace markdown into Doc nodes for the atlas.""" +from __future__ import annotations + +import re +from collections.abc import Iterable +from dataclasses import dataclass +from pathlib import Path +from time import perf_counter + +from ..graph.walk import walk_files + +_MD_SUFFIXES = (".md", ".markdown") +_H1 = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE) +# [[target]] or [[target|alias]]: capture target only +_WIKILINK = re.compile(r"\[\[\s*([^\]|]+?)\s*(?:\|[^\]]*)?\]\]") + + +def _norm(s: str) -> str: + return s.strip().lower().replace("_", "-").replace(" ", "-") + + +@dataclass(frozen=True) +class Doc: + rel_path: str # workspace-relative, forward-slashed (stable id) + title: str # first H1, else filename stem + body: str # raw markdown + link_targets: tuple[str, ...] # normalized [[wiki-link]] targets, sorted-unique + dir_rel: str # workspace-relative dir ("" at root) + + +def _parse_doc(rel_path: str, text: str) -> Doc: + m = _H1.search(text) + title = m.group(1).strip() if m else Path(rel_path).stem + targets = tuple(sorted({_norm(t) for t in _WIKILINK.findall(text)})) + parent = Path(rel_path).parent.as_posix() + return Doc(rel_path, title, text, targets, "" if parent == "." else parent) + + +def discover_docs(root: Path) -> list[Doc]: + """All markdown under `root` (pruned dirs excluded), as Docs sorted by rel_path.""" + out: list[Doc] = [] + for p in walk_files(root, suffixes=_MD_SUFFIXES): + try: + text = p.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + out.append(_parse_doc(p.relative_to(root).as_posix(), text)) + out.sort(key=lambda d: d.rel_path) + return out + + +def discover_router_docs( + root: Path, + *, + paths: Iterable[Path] | None = None, + stats: dict[str, int] | None = None, +) -> list[Doc]: + """Markdown locations for router `describes` edges, without reading bodies. + + The router only needs each doc path and containing directory to connect docs + to the most-specific repo. Atlas/workbench/wiki still call `discover_docs` + when titles, wikilinks, or prose mentions are needed. + """ + root = Path(root) + out: list[Doc] = [] + started = perf_counter() + if paths is None: + doc_paths = list(walk_files(root, suffixes=_MD_SUFFIXES)) + if stats is not None: + stats["traversal_ms"] = int((perf_counter() - started) * 1000) + stats["physical_walks"] = 1 + else: + doc_paths = list(paths) + if stats is not None: + stats["traversal_ms"] = 0 + stats["physical_walks"] = 0 + started = perf_counter() + for p in doc_paths: + rel_path = p.relative_to(root).as_posix() + parent = Path(rel_path).parent.as_posix() + out.append(Doc( + rel_path=rel_path, + title=Path(rel_path).stem, + body="", + link_targets=(), + dir_rel="" if parent == "." else parent, + )) + out.sort(key=lambda d: d.rel_path) + if stats is not None: + stats["row_construction_ms"] = int((perf_counter() - started) * 1000) + stats["docs"] = len(out) + return out diff --git a/client-plugin/server/src/index_graph/knowledge/markdown.py b/client-plugin/server/src/index_graph/knowledge/markdown.py new file mode 100644 index 0000000..fbe9065 --- /dev/null +++ b/client-plugin/server/src/index_graph/knowledge/markdown.py @@ -0,0 +1,146 @@ +"""Zero-dependency GFM-lite markdown -> escaping-safe HTML for atlas docs.""" +from __future__ import annotations + +import re +from html import escape as _esc # &<>"' -> entities (quote=True by default) + +from .docs import _norm # shared normalizer: space/underscore -> dash, lower + +_CODE = re.compile(r"`([^`]+)`") +_IMAGE = re.compile(r"!\[([^\]]*)\]\([^)]*\)") +_WIKILINK = re.compile(r"\[\[\s*([^\]|]+?)\s*(?:\|\s*([^\]]*?)\s*)?\]\]") +_LINK = re.compile(r"\[([^\]]+)\]\(\s*([^()]*(?:\([^)]*\))*[^()]*?)\s*\)") +_BOLD = re.compile(r"\*\*([^*]+)\*\*") +_ITALIC = re.compile(r"(? str: + target, alias = m.group(1), m.group(2) + label = alias if alias else target # already inside escaped text + return ('%s' + % (_esc(_norm(target), quote=True), label)) + + +def _link_sub(m: "re.Match") -> str: + label, url = m.group(1), m.group(2) + if not _SAFE_URL.match(url): + return label # drop unsafe scheme, keep the text + return '%s' % (url, label) + + +def render_inline(text: str) -> str: + text = text.replace("\x00", "") # strip NUL so a doc body can't forge a code-span sentinel + codes: list[str] = [] + + def _stash(m: "re.Match") -> str: + codes.append("" + _esc(m.group(1)) + "") + return "\x00%d\x00" % (len(codes) - 1) # null-byte sentinel: absent from markdown, survives escaping + + text = _CODE.sub(_stash, text) + text = _esc(text) # escape all remaining literal text + text = _IMAGE.sub(lambda m: '' + m.group(1) + "", text) + text = _WIKILINK.sub(_wiki_sub, text) + text = _LINK.sub(_link_sub, text) + text = _BOLD.sub(r"\1", text) + text = _ITALIC.sub(r"\1", text) + text = re.sub(r"\x00(\d+)\x00", lambda m: codes[int(m.group(1))], text) + return text + + +_HEADING = re.compile(r"(#{1,6})\s+(.*)$") +_ULI = re.compile(r"\s*[-*+]\s+(.*)$") +_OLI = re.compile(r"\s*\d+[.)]\s+(.*)$") +_TASK = re.compile(r"\s*[-*+]\s+\[([ xX])\]\s+(.*)$") +_BQ = re.compile(r">\s?(.*)$") +_TABLE_SEP = re.compile(r"^\s*\|?\s*:?-{1,}:?\s*(\|\s*:?-{1,}:?\s*)+\|?\s*$") + + +def _starts_block(line: str) -> bool: + return bool(_HEADING.match(line) or _ULI.match(line) or _OLI.match(line) + or line.startswith("```") or line.startswith(">") or "|" in line) + + +def _render_li(text: str) -> str: + task = _TASK.match(text) + if task: + checked = " checked" if task.group(1) in ("x", "X") else "" + return ('
  • %s
  • ' + % (checked, render_inline(task.group(2).strip()))) + body = (_ULI.match(text) or _OLI.match(text)).group(1) + return "
  • " + render_inline(body.strip()) + "
  • " + + +def _consume_list(lines: list[str], i: int) -> tuple[str, int]: + ordered = bool(_OLI.match(lines[i]) and not _ULI.match(lines[i])) + items: list[str] = [] + while i < len(lines) and (_ULI.match(lines[i]) or _OLI.match(lines[i])): + items.append(_render_li(lines[i])) + i += 1 + tag = "ol" if ordered else "ul" + return "<%s>\n%s\n" % (tag, "\n".join(items), tag), i + + +def _is_table(lines: list[str], i: int) -> bool: + return ("|" in lines[i] and i + 1 < len(lines) and bool(_TABLE_SEP.match(lines[i + 1]))) + + +def _row_cells(line: str) -> list[str]: + return [c.strip() for c in line.strip().strip("|").split("|")] + + +def _consume_table(lines: list[str], i: int) -> tuple[str, int]: + head = _row_cells(lines[i]); i += 2 # header row + separator row + body: list[str] = [] + while i < len(lines) and "|" in lines[i] and lines[i].strip(): + cells = _row_cells(lines[i]) + body.append("" + "".join("" + render_inline(c) + "" for c in cells) + "") + i += 1 + thead = "" + "".join("" + render_inline(c) + "" for c in head) + "" + return "\n%s\n%s\n
    " % (thead, "\n".join(body)), i + + +def render_markdown(text: str) -> str: + lines = text.replace("\r\n", "\n").replace("\r", "\n").split("\n") + out: list[str] = [] + i, n = 0, len(lines) + while i < n: + line = lines[i] + if line.startswith("```"): + i += 1 + buf: list[str] = [] + while i < n and not lines[i].startswith("```"): + buf.append(lines[i]); i += 1 + i += 1 # skip the closing fence (or run off end) + out.append("
    " + _esc("\n".join(buf)) + "
    ") + continue + h = _HEADING.match(line) + if h: + lvl = len(h.group(1)) + out.append("%s" % (lvl, render_inline(h.group(2).strip()), lvl)) + i += 1; continue + if line.startswith(">"): + buf = [] + while i < n and lines[i].startswith(">"): + buf.append(_BQ.match(lines[i]).group(1)); i += 1 + out.append("
    " + render_inline(" ".join(b for b in buf if b)) + "
    ") + continue + if _is_table(lines, i): + block, i = _consume_table(lines, i) + out.append(block); continue + if _ULI.match(line) or _OLI.match(line): + block, i = _consume_list(lines, i) + out.append(block); continue + if line.strip() == "": + i += 1; continue + # Always consume the line that opened the paragraph. It may LOOK like a + # block starter (a prose line containing "|" that is not a table), but + # every real block was dispatched above; skipping it here would loop + # forever on the same line. + buf = [line] + i += 1 + while i < n and lines[i].strip() != "" and not _starts_block(lines[i]): + buf.append(lines[i]); i += 1 + out.append("

    " + render_inline(" ".join(buf)) + "

    ") + return "\n".join(out) diff --git a/client-plugin/server/src/index_graph/lsp/__init__.py b/client-plugin/server/src/index_graph/lsp/__init__.py new file mode 100644 index 0000000..ba0fa3b --- /dev/null +++ b/client-plugin/server/src/index_graph/lsp/__init__.py @@ -0,0 +1,15 @@ +"""LSP server: IDE-facing go-to-definition and find-references over the wave-1 +symbol graph, hand-rolled Content-Length framing, zero runtime dependencies. + +``index lsp --root ROOT`` starts a stdio JSON-RPC 2.0 language server that +answers ``textDocument/definition`` and ``textDocument/references`` from the +existing symbol-level call/reference graph. Every answer is evidence-backed +(a resolved file:line) or honestly empty; an unresolved reference is never +guessed into a jump, and a workspace that changed on disk is detected, not +silently answered from a stale graph. +""" +from __future__ import annotations + +from .server import LSPServer, serve + +__all__ = ["LSPServer", "serve"] diff --git a/client-plugin/server/src/index_graph/lsp/protocol.py b/client-plugin/server/src/index_graph/lsp/protocol.py new file mode 100644 index 0000000..c4e9c02 --- /dev/null +++ b/client-plugin/server/src/index_graph/lsp/protocol.py @@ -0,0 +1,93 @@ +"""LSP stdio framing and JSON-RPC 2.0 helpers, hand-rolled, zero dependencies. + +Unlike the newline-delimited MCP face, the Language Server Protocol frames each +message with a ``Content-Length`` header followed by a blank line and then the +UTF-8 JSON body (RFC 3156-style). These helpers read and write that framing over +binary streams and build the JSON-RPC response/error envelopes, nothing more: +dispatch and state live in ``server.py``. +""" +from __future__ import annotations + +import json +from typing import BinaryIO + +_HEADER_SEP = b"\r\n" +_HEADER_END = b"\r\n\r\n" + + +def read_message(stream: BinaryIO) -> dict | None: + """Read one Content-Length-framed JSON message, or None at end of stream. + + Reads header lines until the blank separator, honors the ``Content-Length`` + header (ignoring any others such as ``Content-Type``), then reads exactly + that many body bytes and parses them as JSON. A truncated frame or a body + that is not a JSON object reads as None (end of usable input). + """ + length: int | None = None + while True: + line = _read_header_line(stream) + if line is None: + return None # EOF before a complete header block + if line == b"": + break # blank line: header block ends + name, _, value = line.partition(b":") + if name.strip().lower() == b"content-length": + try: + length = int(value.strip()) + except ValueError: + return None + if length is None: + return None + body = _read_exact(stream, length) + if body is None: + return None + try: + msg = json.loads(body.decode("utf-8")) + except (json.JSONDecodeError, UnicodeDecodeError): + return None + return msg if isinstance(msg, dict) else None + + +def _read_header_line(stream: BinaryIO) -> bytes | None: + """Read one CRLF-terminated header line (without the CRLF); None at EOF.""" + buf = bytearray() + while True: + ch = stream.read(1) + if not ch: + return None if not buf else bytes(buf) + buf += ch + if buf.endswith(_HEADER_SEP): + return bytes(buf[:-2]) + + +def _read_exact(stream: BinaryIO, n: int) -> bytes | None: + """Read exactly n bytes; None if the stream ends early.""" + chunks: list[bytes] = [] + remaining = n + while remaining > 0: + chunk = stream.read(remaining) + if not chunk: + return None + chunks.append(chunk) + remaining -= len(chunk) + return b"".join(chunks) + + +def write_message(stream: BinaryIO, message: dict) -> None: + """Frame a JSON-RPC message with a Content-Length header and flush it.""" + body = json.dumps(message).encode("utf-8") + stream.write(b"Content-Length: " + str(len(body)).encode("ascii") + _HEADER_END) + stream.write(body) + flush = getattr(stream, "flush", None) + if callable(flush): + flush() + + +def make_response(rid, result) -> dict: + """A JSON-RPC 2.0 success response.""" + return {"jsonrpc": "2.0", "id": rid, "result": result} + + +def make_error(rid, code: int, message: str) -> dict: + """A JSON-RPC 2.0 error response.""" + return {"jsonrpc": "2.0", "id": rid, "error": {"code": code, "message": message}} diff --git a/client-plugin/server/src/index_graph/lsp/server.py b/client-plugin/server/src/index_graph/lsp/server.py new file mode 100644 index 0000000..1bcd85c --- /dev/null +++ b/client-plugin/server/src/index_graph/lsp/server.py @@ -0,0 +1,275 @@ +"""The LSP server: dispatch, workspace state, and the two providers. + +State is deliberately thin. On ``initialize``/``initialized`` the server builds +the wave-1 symbol graph for its single ``--root`` workspace and records a content +fingerprint plus a cheap metadata signature of the Python tree. Before answering +``textDocument/definition`` or ``textDocument/references`` it reuses metadata +only inside a short monotonic cache window, and only when file mtimes are outside +the recent-write uncertainty window. Otherwise it re-checks the content +fingerprint; if that moved, it returns a typed STALE error instead of an answer +derived from a graph that no longer describes the files. A deliberately restored +old mtime can still hide until the bounded content recheck. Every positive +answer is an evidence-backed Location from the graph; an unresolved name is null +(definition) or [] (references), never a guess, and never a symbol from outside +this root. +""" +from __future__ import annotations + +from dataclasses import dataclass +import hashlib +import sys +import time +from pathlib import Path + +from ..graph.walk import walk_files +from ..symbols import build_symbol_graph +from ..symbols.model import SymbolGraph +from . import protocol +from .protocol import make_error, make_response +from .symbols_lsp import (find_symbol_at_position, path_to_uri, to_lsp_location, + to_lsp_range) + +# JSON-RPC error codes. +METHOD_NOT_FOUND = -32601 +STALE_CODE = -32603 # Internal Error: the workspace changed since initialize + +_SERVER_NAME = "index-lsp" +_MTIME_UNCERTAINTY_NS = 1_000_000_000 +_CONTENT_RECHECK_NS = 2_000_000_000 + + +@dataclass(frozen=True) +class _StatSignature: + digest: str + newest_mtime_ns: int + + +def _wall_time_ns() -> int: + return time.time_ns() + + +def _monotonic_ns() -> int: + return time.monotonic_ns() + + +def _fingerprint(root: Path) -> str: + """A content fingerprint of the Python tree under root: SHA-256 over the + sorted (relative-path, file-sha) pairs. A changed, added, or removed .py + file moves the fingerprint; a stable tree keeps it byte-identical.""" + root = Path(root).resolve() + entries: list[str] = [] + for py in walk_files(root, suffixes=(".py",)): + try: + rel = py.relative_to(root).as_posix() + except ValueError: + rel = py.as_posix() + try: + digest = hashlib.sha256(py.read_bytes()).hexdigest() + except OSError: + digest = "unreadable" + entries.append(f"{rel}:{digest}") + entries.sort() + return hashlib.sha256("\n".join(entries).encode("utf-8")).hexdigest() + + +def _cheap_signature(root: Path) -> _StatSignature: + """A cheap staleness pre-check: SHA-256 over the sorted + (relative-path, mtime_ns, size) triples of the Python tree. This is + stat-only (no file reads), so it is O(files) rather than the + O(files * bytes) of [`_fingerprint`]. It is a bounded cache key, not an + authority: a match can skip the full read only outside the mtime uncertainty + window and only until the content-recheck interval expires.""" + root = Path(root).resolve() + entries: list[str] = [] + newest_mtime_ns = 0 + for py in walk_files(root, suffixes=(".py",)): + try: + rel = py.relative_to(root).as_posix() + except ValueError: + rel = py.as_posix() + try: + st = py.stat() + newest_mtime_ns = max(newest_mtime_ns, st.st_mtime_ns) + entries.append(f"{rel}:{st.st_mtime_ns}:{st.st_size}") + except OSError: + entries.append(f"{rel}:unstattable") + entries.sort() + digest = hashlib.sha256("\n".join(entries).encode("utf-8")).hexdigest() + return _StatSignature(digest, newest_mtime_ns) + + +class LSPServer: + """A single-workspace language server over the wave-1 symbol graph.""" + + def __init__(self, root: Path, trace: str = "off") -> None: + self.root = Path(root).resolve() + self.trace = trace + self.symbol_graph: SymbolGraph | None = None + self.fingerprint: str | None = None + self.cheap_sig: _StatSignature | None = None + self.last_content_check_mono_ns: int | None = None + self.should_exit = False + self._shutdown = False + + # --- lifecycle ----------------------------------------------------------- + + def _build(self) -> None: + """Build (or rebuild) the symbol graph and pin the current fingerprint.""" + self.symbol_graph = build_symbol_graph(self.root) + self.fingerprint = _fingerprint(self.root) + self.cheap_sig = _cheap_signature(self.root) + self.last_content_check_mono_ns = _monotonic_ns() + + def is_stale(self) -> bool: + """True when the Python tree changed on disk since the last build. + + Fail-closed around recent writes while staying fast in the common case: + a cheap stat-only signature is checked first, and a match skips the + full-content fingerprint only outside the recent-mtime uncertainty + window and only until the short monotonic content-recheck interval + expires. Metadata changes, recent mtimes, future mtimes, and expired + cache windows all fall back to the content fingerprint. + """ + if self.fingerprint is None: + return False + cheap_now = _cheap_signature(self.root) + wall_now = _wall_time_ns() + mono_now = _monotonic_ns() + if cheap_now == self.cheap_sig and self._metadata_cache_valid(cheap_now, wall_now, mono_now): + return False + # The cheap signature moved; the full-content fingerprint is the + # authority and decides staleness (fail-closed on a real change). + if _fingerprint(self.root) != self.fingerprint: + return True + # Either metadata moved while content stayed identical, or the bounded + # metadata cache expired. Refresh both checks from this content read. + self.cheap_sig = cheap_now + self.last_content_check_mono_ns = mono_now + return False + + def _metadata_cache_valid(self, stat_sig: _StatSignature, wall_now: int, mono_now: int) -> bool: + if self.last_content_check_mono_ns is None: + return False + mtime_age = wall_now - stat_sig.newest_mtime_ns + if stat_sig.newest_mtime_ns and (mtime_age < 0 or mtime_age <= _MTIME_UNCERTAINTY_NS): + return False + content_age = mono_now - self.last_content_check_mono_ns + return 0 <= content_age <= _CONTENT_RECHECK_NS + + # --- dispatch ------------------------------------------------------------ + + def handle_request(self, req: dict) -> dict | None: + method = req.get("method") + rid = req.get("id") + + if method == "initialize": + root = self._root_from_params(req.get("params") or {}) + if root is not None: + self.root = root + return make_response(rid, self._capabilities()) + if method == "initialized": + self._build() + return None + if method == "shutdown": + self._shutdown = True + return make_response(rid, None) + if method == "exit": + self.should_exit = True + return None + if method == "textDocument/didOpen": + return None # tracked implicitly; staleness is fingerprint-based + if method == "textDocument/definition": + return self._guarded(rid, self._definition, req.get("params") or {}) + if method == "textDocument/references": + return self._guarded(rid, self._references, req.get("params") or {}) + if rid is None: + return None # an unknown notification: nothing to answer + return make_error(rid, METHOD_NOT_FOUND, f"method not found: {method}") + + def _root_from_params(self, params: dict) -> Path | None: + """Honor an explicit rootUri/rootPath from the client, else keep --root.""" + uri = params.get("rootUri") + if isinstance(uri, str) and uri: + from .symbols_lsp import uri_to_path + return uri_to_path(uri) + path = params.get("rootPath") + if isinstance(path, str) and path: + return Path(path).resolve() + return None + + def _capabilities(self) -> dict: + from .. import __version__ + return {"capabilities": {"definitionProvider": True, + "referencesProvider": True, + "textDocumentSync": 1}, + "serverInfo": {"name": _SERVER_NAME, "version": __version__}} + + def _guarded(self, rid, handler, params: dict) -> dict: + """Run a provider, but first fail closed if the workspace is stale.""" + if self.symbol_graph is None: + self._build() + if self.is_stale(): + return make_error(rid, STALE_CODE, + "workspace changed on disk since initialize; " + "re-initialize the LSP server (stale graph blocked)") + return make_response(rid, handler(params)) + + # --- providers ----------------------------------------------------------- + + def _symbol_under_cursor(self, params: dict): + uri = (params.get("textDocument") or {}).get("uri") or "" + position = params.get("position") or {} + return find_symbol_at_position(uri, position, self.symbol_graph, self.root) + + def _definition(self, params: dict): + """Return an LSP Location for the symbol under the cursor, or null.""" + sym = self._symbol_under_cursor(params) + if sym is None: + return None # unresolved / outside root / not an identifier: never a guess + return to_lsp_location(sym, self.root) + + def _references(self, params: dict) -> list: + """Return every resolved caller of the symbol under the cursor. + + Unresolved references are excluded: they are not evidence-backed edges. + """ + sym = self._symbol_under_cursor(params) + if sym is None: + return [] + graph = self.symbol_graph + assert graph is not None + locations: list[dict] = [] + seen: set[tuple[str, int]] = set() + for call in graph.calls: + if call.to_symbol != sym.id: + continue + key = (call.evidence_file, call.evidence_line) + if key in seen: + continue + seen.add(key) + uri = path_to_uri(self.root / call.evidence_file) + locations.append({"uri": uri, "range": to_lsp_range(call.evidence_line)}) + include_decl = ((params.get("context") or {}).get("includeDeclaration")) + if include_decl: + locations.insert(0, to_lsp_location(sym, self.root)) + return locations + + # --- main loop ----------------------------------------------------------- + + def serve(self, stdin=None, stdout=None) -> int: + """Read Content-Length-framed JSON-RPC from stdin, answer on stdout.""" + stdin = stdin if stdin is not None else sys.stdin.buffer + stdout = stdout if stdout is not None else sys.stdout.buffer + while not self.should_exit: + msg = protocol.read_message(stdin) + if msg is None: + break + resp = self.handle_request(msg) + if resp is not None: + protocol.write_message(stdout, resp) + return 0 + + +def serve(root: Path, trace: str = "off", stdin=None, stdout=None) -> int: + """Start an LSPServer on root and run its stdio loop.""" + return LSPServer(root=root, trace=trace).serve(stdin, stdout) diff --git a/client-plugin/server/src/index_graph/lsp/symbols_lsp.py b/client-plugin/server/src/index_graph/lsp/symbols_lsp.py new file mode 100644 index 0000000..0af3f18 --- /dev/null +++ b/client-plugin/server/src/index_graph/lsp/symbols_lsp.py @@ -0,0 +1,123 @@ +"""Bridge the wave-1 symbol model to LSP positions and Locations. + +Two conversions and one lookup, all deterministic and evidence-backed: + - ``path_to_uri`` / ``uri_to_path``: file:// URIs the IDE speaks. + - ``to_lsp_location``: a SymbolDefinition -> an LSP Location (0-indexed line, + column 0; the client highlights the symbol name). + - ``find_symbol_at_position``: given a document and a cursor, name the symbol + the cursor sits on. A cursor on a definition line resolves to that symbol; + a cursor on a call resolves to the identifier written under it, matched + against real definitions (never guessed). A file outside the server root, + or a cursor on whitespace/comment, resolves to None. +""" +from __future__ import annotations + +from pathlib import Path +from urllib.parse import unquote, urlparse + +from ..symbols.model import SymbolDefinition, SymbolGraph + +# Identifier characters for the "word under the cursor" scan. +_IDENT = set("abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789_") + + +def path_to_uri(path: Path) -> str: + """A file:// URI for a filesystem path (resolved, percent-encoded). + + Uses ``Path.as_uri`` so the result is a well-formed ``file://`` URI on both + POSIX (``file:///tmp/x``) and Windows (``file:///C:/x``); hand-building it + from ``pathname2url`` yielded a single-slash ``file:/tmp/x`` on POSIX. + """ + return Path(path).resolve().as_uri() + + +def uri_to_path(uri: str) -> Path: + """The filesystem path a file:// URI names, resolved.""" + parsed = urlparse(uri) + raw = unquote(parsed.path) + # On Windows, urlparse leaves a leading slash before the drive letter. + if len(raw) >= 3 and raw[0] == "/" and raw[2] == ":": + raw = raw[1:] + return Path(raw).resolve() + + +def to_lsp_range(line_1indexed: int) -> dict: + """A zero-width LSP range at the start of a 1-indexed source line.""" + line0 = max(line_1indexed - 1, 0) + return {"start": {"line": line0, "character": 0}, + "end": {"line": line0, "character": 0}} + + +def to_lsp_location(sym: SymbolDefinition, root: Path) -> dict: + """An LSP Location for a SymbolDefinition, rooted at the workspace root.""" + uri = path_to_uri(Path(root) / sym.file) + return {"uri": uri, "range": to_lsp_range(sym.line)} + + +def _word_at(line: str, character: int) -> str | None: + """The identifier the cursor sits within, or None on non-identifier text.""" + if character < 0 or character > len(line): + return None + # A cursor exactly at end-of-word (character == len(word)) still counts. + if character == len(line) or line[character] not in _IDENT: + if character == 0 or line[character - 1] not in _IDENT: + return None + start = character + while start > 0 and line[start - 1] in _IDENT: + start -= 1 + end = character + while end < len(line) and line[end] in _IDENT: + end += 1 + word = line[start:end] + return word or None + + +def find_symbol_at_position( + uri: str, position: dict, graph: SymbolGraph, root: Path, +) -> SymbolDefinition | None: + """Resolve the symbol under an LSP cursor against the wave-1 graph. + + Only files inside ``root`` are considered; a file outside the server's + workspace resolves to None (never a cross-repo guess). Resolution defers + entirely to the wave-1 graph's own verdict, so nothing is guessed: + + - A cursor on a definition line resolves to that definition. + - A cursor on a call site resolves via the ``SymbolCall`` recorded at that + exact (file, line): a resolved call (``to_symbol`` set) jumps to its + target definition; an unresolved call (``to_symbol`` None, e.g. a + same-named symbol elsewhere with no import/call edge) resolves to None, + never to a bare-name match against an unrelated definition. + """ + root = Path(root).resolve() + path = uri_to_path(uri) + try: + rel = path.relative_to(root).as_posix() + except ValueError: + return None # document is outside this server's workspace root + try: + text = path.read_text(encoding="utf-8-sig") + except OSError: + return None + lines = text.splitlines() + line_no = position.get("line", 0) + if line_no < 0 or line_no >= len(lines): + return None + word = _word_at(lines[line_no], position.get("character", 0)) + if not word: + return None + line_1 = line_no + 1 + # Cursor on a definition line in this file: return that definition. + on_line = [s for s in graph.symbols + if s.file == rel and s.name == word and s.line == line_1] + if on_line: + return on_line[0] + # Cursor on a call site: defer to the graph's resolution for THIS exact call + # site. A resolved call names its target; an unresolved call names None. + by_id = {s.id: s for s in graph.symbols} + for call in graph.calls: + if (call.evidence_file == rel and call.evidence_line == line_1 + and call.to_name == word): + if call.to_symbol is not None: + return by_id.get(call.to_symbol) + return None # graph classified this call unresolved: never guess + return None diff --git a/client-plugin/server/src/index_graph/mcp.py b/client-plugin/server/src/index_graph/mcp.py new file mode 100644 index 0000000..5842b60 --- /dev/null +++ b/client-plugin/server/src/index_graph/mcp.py @@ -0,0 +1,730 @@ +"""A zero-dependency, MCP-shaped stdio protocol face for index. + +Newline-delimited JSON-RPC 2.0 over stdin/stdout, no SDK and no model. An agent host +connects and calls deterministic tools (graph, focus, verify, router, internals) to +consume index's verified map natively. This is the clean seam a router or orchestrator +composes through: the protocol pillar, not embeddings. Every tool reuses an existing +index function, so the protocol face adds a surface, never a second source of truth. +""" +from __future__ import annotations + +import hashlib +import json +import os +import sys +import traceback +from pathlib import Path +from time import time + +from . import __version__ +from .graph.progress import stderr_progress + +_PROTOCOL_VERSION = "2024-11-05" +_CACHE_SCHEMA = "index.mcp-cache-entry/v1" +_CACHE: dict[str, dict] = {} +_CACHEABLE_TOOLS = { + "index.context", + "index.context.envelope", + "index_graph", + "index_router", +} + + +def _configure_stdio() -> None: + for stream in (sys.stdout, sys.stderr): + reconfigure = getattr(stream, "reconfigure", None) + if callable(reconfigure): + try: + reconfigure(encoding="utf-8", errors="replace") + except (OSError, ValueError): + pass + + +def _tool_error_payload(name: str, args: dict, exc: BaseException) -> str: + root = args.get("root", "") if isinstance(args, dict) else "" + payload = { + "schema": "index.mcp-tool-error/v1", + "tool": name, + "status": "UNVERIFIABLE", + "error_type": type(exc).__name__, + "message": str(exc), + "root": str(root), + "recoverable": not isinstance(exc, (KeyboardInterrupt,)), + "next_actions": [ + "Inspect the root configuration and filesystem permissions.", + "Run the matching index CLI command with --json to reproduce outside the MCP host.", + "If this came from a large workspace scan, reduce focus or add [scan].prune entries.", + ], + } + if os.environ.get("INDEX_MCP_DEBUG_ERRORS") == "1": + payload["traceback"] = traceback.format_exception(type(exc), exc, exc.__traceback__) + return json.dumps(payload, indent=2, sort_keys=True) + + +def _cache_ttl_seconds() -> float: + raw = os.environ.get("INDEX_MCP_CACHE_TTL_SECONDS", "900") + try: + return max(0.0, float(raw)) + except ValueError: + return 900.0 + + +def _cache_dir() -> Path: + raw = os.environ.get("INDEX_MCP_CACHE_DIR") + if raw: + return Path(raw) + base = os.environ.get("LOCALAPPDATA") + if base: + return Path(base) / "index_graph" / "mcp-cache" + return Path.home() / ".cache" / "index_graph" / "mcp-cache" + + +def _sha256_text(value: str) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def _workspace_signature(root: Path) -> str: + """Cheap MCP cache identity. + + Freshness is handled by typed fingerprints on the verification/envelope + surfaces. The MCP cache key must not recursively walk a large workspace before + it can discover whether a warm entry exists. + """ + from .cache import workspace_signature + + return workspace_signature(root) + + +def _interactive_budget_ms(args: dict) -> int: + raw = args.get("budget_ms") + if raw is None or raw == "": + from .scan import default_interactive_budget_ms + return default_interactive_budget_ms() + try: + value = int(raw) + except (TypeError, ValueError) as exc: + raise ValueError("budget_ms must be an integer millisecond budget") from exc + if value < 0: + raise ValueError("budget_ms must be non-negative") + return value + + +def _cache_args(args: dict, *, budget_ms: int | None = None) -> dict: + stable = dict(args) + if budget_ms is not None: + stable["budget_ms"] = budget_ms + return stable + + +def _cache_key(name: str, root: Path, args: dict) -> str: + stable_args = { + key: value + for key, value in args.items() + if key not in {"root"} and isinstance(value, (str, int, float, bool, type(None), list, dict)) + } + payload = { + "tool": name, + "root": str(root), + "args": stable_args, + "workspace_signature": _workspace_signature(root), + "tool_version": __version__, + } + return _sha256_text(json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str)) + + +def _cache_path(key: str) -> Path: + return _cache_dir() / f"{key}.json" + + +def _cache_read(key: str) -> str | None: + ttl = _cache_ttl_seconds() + if ttl <= 0: + return None + now = time() + entry = _CACHE.get(key) + if entry and now - float(entry.get("created_at", 0.0)) <= ttl: + return str(entry.get("text", "")) + path = _cache_path(key) + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError, json.JSONDecodeError): + return None + if data.get("schema") != _CACHE_SCHEMA: + return None + created_at = float(data.get("created_at", 0.0)) + if now - created_at > ttl: + return None + text = str(data.get("text", "")) + _CACHE[key] = {"created_at": created_at, "text": text} + return text + + +def _cache_write(key: str, text: str) -> str: + ttl = _cache_ttl_seconds() + if ttl <= 0: + return text + entry = {"schema": _CACHE_SCHEMA, "created_at": time(), "text": text} + _CACHE[key] = {"created_at": entry["created_at"], "text": text} + try: + path = _cache_path(key) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(entry, separators=(",", ":")), encoding="utf-8") + except OSError: + pass + return text + + +def _with_cache(name: str, root: Path, args: dict, build): + if "no_cache" in args and not isinstance(args["no_cache"], bool): + raise ValueError("no_cache must be a boolean") + if args.get("no_cache"): + return build() + if name not in _CACHEABLE_TOOLS: + return build() + key = _cache_key(name, root, args) + cached = _cache_read(key) + if cached is not None: + return cached + return _cache_write(key, build()) + + + + +def _bool_arg(args: dict, key: str, default: bool = False) -> bool: + value = args.get(key, default) + if isinstance(value, bool): + return value + raise ValueError(f"{key} must be a boolean") + +def _root_schema(extra: dict | None = None, required: list | None = None) -> dict: + props = {"root": {"type": "string", "description": "workspace root path"}} + if extra: + props.update(extra) + return {"type": "object", "properties": props, "required": required or ["root"]} + + +def _workspace_schema(extra: dict | None = None, required: list | None = None) -> dict: + props = { + "budget_ms": { + "type": "integer", + "description": "optional repository-discovery time budget in milliseconds; 0 means unbounded", + } + } + if extra: + props.update(extra) + return _root_schema(props, required=required) + + +def _hints(title: str, *, read_only: bool = True, idempotent: bool = True) -> dict: + return {"title": title, "readOnlyHint": read_only, "destructiveHint": False, + "idempotentHint": idempotent, "openWorldHint": False} + + +# MCP tool annotations. A hint describes the tool to the client and grants +# nothing. index.map writes only when given resume_state; router jobs keep +# private state files. +TOOL_ANNOTATIONS = { + "index.map": _hints("Map a repository", read_only=False), + "index.context": _hints("Repository dependency context"), + "index.context.envelope": _hints("Budgeted context envelope"), + "index.select": _hints("Select paths with receipts"), + "index.invalidate": _hints("Name what changes invalidate"), + "index.wiki": _hints("Build or verify a repository wiki"), + "index.symbol-graph": _hints("Symbol call graph"), + "index.symbol-definition": _hints("Go to a symbol definition"), + "index.symbol-references": _hints("Find symbol references"), + "index.symbol-implementations": _hints("Find symbol implementations"), + "index.status": _hints("Index status"), + "index.doctor": _hints("Index readiness check"), + "index_graph": _hints("Workspace dependency graph"), + "index_focus": _hints("Repository neighborhood"), + "index_verify": _hints("Check a structural claim"), + "index_router": _hints("Build a workspace map"), + "index.route": _hints("Context envelope for named paths"), + "index_internals": _hints("Module dependency graph"), + "index.router.job.start": _hints("Start a workspace map job", read_only=False, + idempotent=False), + "index.router.job.status": _hints("Workspace map job status"), + "index.router.job.result": _hints("Workspace map job result"), + "index.router.job.cancel": _hints("Cancel a workspace map job", read_only=False), + "index.router.job.resume": _hints("Resume a workspace map job", read_only=False, + idempotent=False), +} + + +def annotate(tool: dict) -> dict: + notes = dict(TOOL_ANNOTATIONS[tool["name"]]) + return {**tool, "title": notes["title"], "annotations": notes} + + +def _tool_defs() -> list[dict]: + return [annotate(tool) for tool in _raw_tool_defs()] + + +def _raw_tool_defs() -> list[dict]: + from .route import tool_definition as route_tool_definition + from .router_job_surface import tool_definitions + return [ + {"name": "index.map", + "description": "Repository inventory map as JSON, matching the `index map --json` CLI surface.", + "inputSchema": _root_schema({"resume_state": {"type": "string", "description": "optional JSONL state file for resumable complete map builds"}})}, + {"name": "index.context", + "description": "Repo-level dependency context pack as JSON, matching the `index context --json` CLI surface.", + "inputSchema": _workspace_schema()}, + {"name": "index.context.envelope", + "description": "Budgeted, receipt-backed context envelope for large-codebase agent workflows.", + "inputSchema": _workspace_schema({ + "budget": {"type": "integer"}, + "focus": {"type": "string"}, + "hops": {"type": "integer"}, + "bounded_output": {"type": "boolean"}, + })}, + {"name": "index.select", + "description": "Path selection with typed rejection receipts; candidates reconcile to selected + rejected, matching the `index select --json` CLI surface.", + "inputSchema": _root_schema({ + "suffixes": {"type": "array", "items": {"type": "string"}}, + "max_files": {"type": "integer"}, + })}, + {"name": "index.invalidate", + "description": "Diff the tree against a pinned fingerprint and name exactly what the changes invalidate (index.invalidation/1). Without 'pin', mints and returns a pin of the current tree, matching the `index invalidate` CLI surface.", + "inputSchema": _root_schema({ + "pin": {"type": "string", "description": "path to a pin JSON minted earlier"}, + })}, + {"name": "index.wiki", + "description": "Single-repo verified wiki pack (pages derived from the module graph, sealed manifest, commit-pinned), matching the `index wiki --format json` CLI surface; pass verify=PATH to re-check a sealed artifact (MATCH/DRIFT/UNVERIFIABLE).", + "inputSchema": _root_schema({ + "verify": {"type": "string", + "description": "path to a sealed wiki artifact to verify"}, + })}, + {"name": "index.symbol-graph", + "description": "Symbol-level call/reference graph for one repo (root IS the repo): function/class/method definitions plus resolved (exact within-module, best-effort cross-module) and honestly-unresolved calls, matching the `index internals-symbols --json` CLI surface.", + "inputSchema": _root_schema()}, + {"name": "index.symbol-definition", + "description": "GO-TO-DEFINITION: the file:line of every symbol whose id or bare name matches, derived (never guessed) from the AST.", + "inputSchema": _root_schema({"symbol": {"type": "string", + "description": "symbol id (module::name) or bare name"}}, + required=["root", "symbol"])}, + {"name": "index.symbol-references", + "description": "FIND-REFERENCES: every resolved caller of a symbol, each with file:line evidence; unresolved references are reported separately, never as a caller.", + "inputSchema": _root_schema({"symbol": {"type": "string", + "description": "symbol id (module::name) or bare name"}}, + required=["root", "symbol"])}, + {"name": "index.symbol-implementations", + "description": "FIND-IMPLEMENTATIONS: in-repo subclasses of a class or overrides of a method, each with file:line evidence and an exact/cross_module resolution label; an external or unbindable base yields no edge (never guessed).", + "inputSchema": _root_schema({"symbol": {"type": "string", + "description": "class or method symbol id (module::name / Class::method) or bare name"}}, + required=["root", "symbol"])}, + {"name": "index.status", + "description": "Project Telos operator-spine status action envelope, matching the `index status --json` CLI surface.", + "inputSchema": {"type": "object", "properties": {}}}, + {"name": "index.doctor", + "description": "Project Telos operator-spine readiness checks action envelope, matching the `index doctor --json` CLI surface.", + "inputSchema": {"type": "object", "properties": {}}}, + {"name": "index_graph", + "description": "Repo-level dependency graph (relations, roles, cycles) as JSON.", + "inputSchema": _workspace_schema({"no_cache": {"type": "boolean", + "description": "Bypass result and per-repository graph caches."}})}, + {"name": "index_focus", + "description": "A repo's dependency neighborhood plus a preservation manifest of what was dropped at the boundary.", + "inputSchema": _workspace_schema({"repo": {"type": "string"}, "hops": {"type": "integer"}}, + required=["root", "repo"])}, + {"name": "index_verify", + "description": "Ground a structural claim. Pass depends 'A -> B' or exists 'NAME'. Returns MATCH/REFUTED/UNVERIFIABLE with file:line evidence.", + "inputSchema": _workspace_schema({"depends": {"type": "string"}, "exists": {"type": "string"}})}, + {"name": "index_router", + "description": "Build a deterministic CLAUDE.md/AGENTS.md workspace map synchronously. For large workspaces, use index.router.job.start, then index.router.job.status and index.router.job.result to avoid an interactive request timeout.", + "inputSchema": _workspace_schema({"max_docs": {"type": "integer"}, + "no_cache": {"type": "boolean", + "description": "Bypass result and per-repository graph caches."}})}, + route_tool_definition(), + {"name": "index_internals", + "description": "Intra-repo module dependency graph for one repo, with cycles and coverage.", + "inputSchema": _workspace_schema({"repo": {"type": "string"}}, required=["root", "repo"])}, + *tool_definitions(), + ] + + +def _repo_paths(root: Path, *, budget_ms: int | None = None) -> dict: + from .config import load_config + from .scan import ( + ScanBudget, + ScanBudgetExceeded, + discover_repos, + enforce_interactive_repo_limit, + repo_key_map, + ) + + config = load_config(None, root) + budget = ScanBudget(budget_ms) + skipped: list[str] = [] + repos = discover_repos( + root, + config, + skipped=skipped, + checkpoint=budget.checkpoint if budget.budget_ms > 0 else None, + ) + if budget.exhausted: + raise ScanBudgetExceeded(root=root, budget=budget, repo_count=len(repos), skipped=skipped) + keyed = repo_key_map( + root, + repos, + include_root_repo=config.include_root_repo, + ) + enforce_interactive_repo_limit(len(keyed), budget_ms=budget.budget_ms) + return keyed + + +def _symbol_matches(sym, query: str) -> bool: + return sym.id == query or sym.name == query + + +def _symbol_tool(name: str, root: Path, args: dict) -> str: + from .symbols import (build_symbol_navigator, find_implementations, + symbol_graph_to_payload) + g, edges = build_symbol_navigator(root) + if name == "index.symbol-graph": + return json.dumps(symbol_graph_to_payload(g), indent=2, sort_keys=True) + query = (args.get("symbol") or "").strip() + if not query: + raise ValueError("missing required argument: symbol") + if name == "index.symbol-implementations": + impls = find_implementations(g, edges, query) + return json.dumps({"symbol": query, **impls}, indent=2, sort_keys=True) + if name == "index.symbol-definition": + defs = [{"id": s.id, "name": s.name, "kind": s.kind, "file": s.file, + "line": s.line, "parent": s.parent} + for s in g.symbols if _symbol_matches(s, query)] + return json.dumps({"symbol": query, "definitions": defs}, + indent=2, sort_keys=True) + # index.symbol-references: resolved callers + separately, unresolved refs + targets = {s.id for s in g.symbols if _symbol_matches(s, query)} + refs = [{"from_symbol": c.from_symbol, "file": c.evidence_file, + "line": c.evidence_line, "raw": c.raw, "confidence": c.confidence} + for c in g.calls if c.to_symbol in targets] + unresolved = [{"from_symbol": c.from_symbol, "to_name": c.to_name, + "file": c.evidence_file, "line": c.evidence_line} + for c in g.calls if c.to_symbol is None and c.to_name == query] + return json.dumps({"symbol": query, "references": refs, + "unresolved_references": unresolved}, indent=2, sort_keys=True) + + +def call_tool(name: str, args: dict, response_id=None) -> str: + if name.startswith("index.router.job."): + from .router_job_surface import call_router_job + return json.dumps(call_router_job(name.removeprefix("index.router.job."), args), + indent=2, sort_keys=True) + + if name == "index.status": + from .flagship import status_payload + return json.dumps(status_payload(), indent=2, sort_keys=True) + + if name == "index.doctor": + from .flagship import doctor_payload + return json.dumps(doctor_payload(), indent=2, sort_keys=True) + + if name == "index.select": + # a missing root yields a not-found receipt (CLI parity), not an error + from .context.select import run_select + if "root" not in args: + raise ValueError("missing required argument: root") + suffixes = tuple(args["suffixes"]) if args.get("suffixes") else None + payload = run_select(Path(args["root"]), suffixes, args.get("max_files")) + return json.dumps(payload, indent=2, sort_keys=True) + + if name == "index.route": + from .route import call_route + return json.dumps(call_route(args, response_id=response_id), indent=2, sort_keys=True) + + from .graph.build import build_graph + from .context.focus import FocusRejection, focus_rejection + from .context.pack import to_json, closure, preservation, focus_subgraph + + if "root" not in args: + raise ValueError("missing required argument: root") + root = Path(args["root"]).resolve() + if not root.is_dir(): + raise ValueError(f"root not found: {root}") + + if name == "index.wiki": + # single-repo altitude: the root IS the repo, no workspace scan + if args.get("verify"): + from .wiki import run_verify + return json.dumps(run_verify(Path(args["verify"]), root), + indent=2, sort_keys=True) + from .wiki import build_wiki_pack + return json.dumps(build_wiki_pack(root), indent=2, sort_keys=True) + + if name in ("index.symbol-graph", "index.symbol-definition", + "index.symbol-references", "index.symbol-implementations"): + return _symbol_tool(name, root, args) + + if name == "index.invalidate": + # without 'pin' this mints one; with 'pin' it emits the typed report. + # both are payloads, matching the CLI's --out / --pin modes. + from .freshness.invalidate import mint_pin + from .freshness.invalidate_cli import run_invalidate + if not args.get("pin"): + return json.dumps(mint_pin(root), indent=2, sort_keys=True) + pin_path = Path(args["pin"]) + try: + pin = json.loads(pin_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise ValueError(f"cannot read pin {pin_path}: {exc}") + return json.dumps(run_invalidate(root, pin), indent=2, sort_keys=True) + + if name == "index.map": + from .config import load_config + from .scan import build_map + resume_state = Path(args["resume_state"]) if args.get("resume_state") else None + return json.dumps( + build_map(root, load_config(None, root), __version__, resume_state=resume_state).to_json(), + indent=2, + sort_keys=True, + ) + + if name in ("index.context", "index_graph"): + budget_ms = _interactive_budget_ms(args) + + def _build_context() -> str: + paths = _repo_paths(root, budget_ms=budget_ms) + return json.dumps(to_json(build_graph(paths, executor="process", + use_cache=not args.get("no_cache", False), + on_progress=stderr_progress())), + indent=2, sort_keys=True) + + return _with_cache( + name, + root, + _cache_args(args, budget_ms=budget_ms), + _build_context, + ) + + if name == "index.context.envelope": + from .context.envelope import build_context_envelope + budget_ms = _interactive_budget_ms(args) + bounded_output = _bool_arg(args, "bounded_output", False) + + def _build_envelope(): + paths = _repo_paths(root, budget_ms=budget_ms) + try: + env = build_context_envelope( + build_graph(paths), + root=root, + token_budget=int(args.get("budget", 1200)), + focus=args.get("focus"), + hops=args.get("hops"), + bounded_output=bounded_output, + bounded_output_transport=( + "mcp_jsonrpc_tool_response" if bounded_output else "canonical_json" + ), + mcp_response_id=response_id, + ) + except FocusRejection as exc: + # an unresolvable focus is a typed receipt, not a protocol error + # (the index.select not-found precedent) + return json.dumps(exc.receipt, indent=2, sort_keys=True) + return json.dumps(env, indent=2, sort_keys=True) + + if bounded_output: + # The measurement includes the JSON-RPC response id, so cached text + # from a different id could understate or overstate the wrapper. + return _build_envelope() + return _with_cache( + name, + root, + _cache_args(args, budget_ms=budget_ms), + _build_envelope, + ) + + if name == "index_focus": + budget_ms = _interactive_budget_ms(args) + repo_paths = _repo_paths(root, budget_ms=budget_ms) + graph = build_graph(repo_paths) + repo = args.get("repo") or "" + names = {n.name for n in graph.repos} + if repo not in names: + return json.dumps(focus_rejection(repo, names), indent=2, sort_keys=True) + hops = args.get("hops") + keep = closure(list(graph.edges), repo, hops=hops) + pack = to_json(focus_subgraph(graph, keep)) + pack["preserved"] = preservation(list(graph.edges), keep, repo, hops) + return json.dumps(pack, indent=2, sort_keys=True) + + if name == "index_verify": + from .verify import build_verification + budget_ms = _interactive_budget_ms(args) + repo_paths = _repo_paths(root, budget_ms=budget_ms) + pack = to_json(build_graph(repo_paths)) + if args.get("depends"): + if "->" not in args["depends"]: + raise ValueError("depends must be 'A -> B'") + frm, _, to = args["depends"].partition("->") + claim = {"kind": "depends", "from": frm.strip(), "to": to.strip()} + elif args.get("exists"): + claim = {"kind": "exists", "name": args["exists"].strip()} + else: + raise ValueError("index_verify needs 'depends' or 'exists'") + rec = build_verification(pack, claim, tool_version=__version__, + recheck="index verify (via mcp)") + return json.dumps(rec, indent=2, sort_keys=True) + + if name == "index_router": + from .knowledge.atlas import build_router_pack + from .knowledge.docs import discover_router_docs + from .router import render_router + from .router_inventory import build_router_inventory + + budget_ms = _interactive_budget_ms(args) + max_docs = max(0, int(args.get("max_docs", 500))) + + def _build_router() -> str: + inventory = build_router_inventory(root, budget_ms=budget_ms) + return render_router(build_router_pack( + build_graph(inventory.repo_paths, executor="process", + use_cache=not args.get("no_cache", False), + on_progress=stderr_progress(), + file_lists=inventory.repo_file_lists), + discover_router_docs(root, paths=inventory.router_doc_paths), + inventory.repo_dirs, + ), max_docs=max_docs) + + return _with_cache( + name, + root, + _cache_args(args, budget_ms=budget_ms), + _build_router, + ) + + if name == "index_internals": + from .internals import build_internals + budget_ms = _interactive_budget_ms(args) + repo_paths = _repo_paths(root, budget_ms=budget_ms) + repo = args.get("repo") + if repo not in repo_paths: + raise ValueError(f"unknown repo: {repo}") + g = build_internals(repo_paths[repo], repo) + payload = { + "repo": g.repo, + "modules": [{"id": m.id, "path": m.path, "language": m.language} for m in g.modules], + "edges": [{"from": e.from_id, "to": e.to_id, "file": e.evidence_file, + "line": e.evidence_line, "raw": e.raw} for e in g.edges], + "cycles": [list(c) for c in g.cycles], + "fan_in": g.fan_in, "fan_out": g.fan_out, + "coverage": {"complete": g.coverage.complete, + "modules": g.coverage.modules, + "internal_edges": g.coverage.internal_edges, + "parse_errors": list(g.coverage.parse_errors), + "dynamic_imports": [{"file": f, "line": ln} + for f, ln in g.coverage.dynamic_imports]}, + } + return json.dumps(payload, indent=2, sort_keys=True) + + raise ValueError(f"unknown tool: {name}") + + +def handle_request(req: dict) -> dict | None: + """Handle one JSON-RPC request; return the response dict, or None for a notification.""" + method = req.get("method") + rid = req.get("id") + + if method == "initialize": + return {"jsonrpc": "2.0", "id": rid, "result": { + "protocolVersion": _PROTOCOL_VERSION, + "capabilities": {"tools": {}}, + "serverInfo": {"name": "index-graph", "version": __version__}}} + if rid is None: + return None # a notification (e.g. notifications/initialized): no response + if method == "ping": + return {"jsonrpc": "2.0", "id": rid, "result": {}} + if method == "tools/list": + return {"jsonrpc": "2.0", "id": rid, "result": {"tools": _tool_defs()}} + if method == "tools/call": + params = req.get("params") or {} + name = params.get("name") + args = params.get("arguments") or {} + if name not in {t["name"] for t in _tool_defs()}: + return {"jsonrpc": "2.0", "id": rid, + "error": {"code": -32602, "message": f"unknown tool: {name!r}"}} + try: + text = call_tool(name, args, response_id=rid) + return {"jsonrpc": "2.0", "id": rid, + "result": {"content": [{"type": "text", "text": text}], "isError": False}} + except BaseException as exc: + return {"jsonrpc": "2.0", "id": rid, + "result": {"content": [{"type": "text", "text": _tool_error_payload(name, args, exc)}], + "isError": True}} + return {"jsonrpc": "2.0", "id": rid, + "error": {"code": -32601, "message": f"method not found: {method}"}} + + +def _decode_line(line) -> str: + if isinstance(line, bytes): + return line.decode("utf-8", "replace") + return str(line) + + +def _read_framed_body(first_line, stdin) -> str | None: + header = _decode_line(first_line).strip() + try: + length = int(header.split(":", 1)[1].strip()) + except (IndexError, ValueError): + return None + while True: + line = stdin.readline() + if line in ("", b""): + return None + if _decode_line(line).strip() == "": + break + body = stdin.read(length) + if body in ("", b""): + return None + if isinstance(body, bytes): + return body.decode("utf-8", "replace") + return str(body) + + +def _write_response(stdout, resp: dict, framed: bool) -> None: + body = json.dumps(resp) + if not framed: + stdout.write(body + "\n") + stdout.flush() + return + payload = body.encode("utf-8") + frame = f"Content-Length: {len(payload)}\r\n\r\n".encode("ascii") + payload + buffer = getattr(stdout, "buffer", None) + if buffer is not None: + buffer.write(frame) + buffer.flush() + return + stdout.write(frame.decode("utf-8")) + stdout.flush() + + +def serve(stdin=None, stdout=None) -> int: + """Read MCP stdio frames or newline-delimited JSON-RPC from stdin.""" + _configure_stdio() + stdin = stdin if stdin is not None else sys.stdin + stdout = stdout if stdout is not None else sys.stdout + reader = getattr(stdin, "buffer", stdin) + while True: + line = reader.readline() + if line in ("", b""): + break + text = _decode_line(line).strip() + framed = text.lower().startswith("content-length:") + if framed: + text = _read_framed_body(line, reader) + if text is None: + continue + else: + text = text.strip() + if not line: + continue + try: + req = json.loads(text) + except json.JSONDecodeError: + continue # no id to address a parse error to; conformant hosts send valid frames + resp = handle_request(req) + if resp is not None: + _write_response(stdout, resp, framed) + return 0 diff --git a/client-plugin/server/src/index_graph/model.py b/client-plugin/server/src/index_graph/model.py new file mode 100644 index 0000000..6b591f9 --- /dev/null +++ b/client-plugin/server/src/index_graph/model.py @@ -0,0 +1,93 @@ +"""Pure data model for a workspace repository map. No I/O, git, or config.""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +SCHEMA_VERSION = 1 + + +@dataclass(frozen=True) +class RepoRow: + path: str + class_: str + branch: str + head: str + origin: str + dirty_count: int + untracked_count: int + markers: tuple[str, ...] + metadata_status: str = "ok" + metadata_error: str | None = None + + def to_json(self) -> dict[str, Any]: + data = { + "path": self.path, + "class": self.class_, + "branch": self.branch, + "head": self.head, + "origin": self.origin, + "dirty_count": self.dirty_count, + "untracked_count": self.untracked_count, + "markers": list(self.markers), + "metadata_status": self.metadata_status, + } + if self.metadata_error: + data["metadata_error"] = self.metadata_error + return data + + +@dataclass(frozen=True) +class Map: + schema_version: int + tool_version: str + generated_at: str + root_sha256_prefix: str + root: str | None + absolute_paths_included: bool + repo_count: int + dirty_count: int + class_counts: dict[str, int] + top_level: tuple[dict[str, Any], ...] + repositories: tuple[RepoRow, ...] + annotations: dict[str, Any] = field(default_factory=dict) + + @property + def metadata_ok_count(self) -> int: + return sum(1 for row in self.repositories if row.metadata_status == "ok") + + @property + def metadata_unknown_count(self) -> int: + return len(self.repositories) - self.metadata_ok_count + + @property + def metadata_status(self) -> str: + return "ok" if self.metadata_unknown_count == 0 else "partial" + + @property + def dirty_count_status(self) -> str: + return "complete" if self.metadata_unknown_count == 0 else "known_only" + + def to_json(self) -> dict[str, Any]: + data: dict[str, Any] = { + "schema_version": self.schema_version, + "tool_version": self.tool_version, + "generated_at": self.generated_at, + "root_sha256_prefix": self.root_sha256_prefix, + "absolute_paths_included": self.absolute_paths_included, + "repo_count": self.repo_count, + "dirty_count": self.dirty_count, + "dirty_count_status": self.dirty_count_status, + "metadata_status": self.metadata_status, + "metadata_ok_count": self.metadata_ok_count, + "metadata_unknown_count": self.metadata_unknown_count, + "class_counts": self.class_counts, + "top_level": list(self.top_level), + "repositories": [row.to_json() for row in self.repositories], + } + if self.root is not None: + data["root"] = self.root + if self.annotations: + data["annotations"] = self.annotations + return data diff --git a/client-plugin/server/src/index_graph/route.py b/client-plugin/server/src/index_graph/route.py new file mode 100644 index 0000000..e02201a --- /dev/null +++ b/client-plugin/server/src/index_graph/route.py @@ -0,0 +1,250 @@ +"""Explicit-path task routing backed by the context-envelope contract.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +from typing import Sequence + +from .context.envelope import build_context_envelope +from .graph.build import build_graph +from .scan import repo_key_map + +SCHEMA = "index.route/v1" + + +def _root_hash(root: Path) -> str: + return hashlib.sha256(str(root.resolve()).encode("utf-8")).hexdigest()[:16] + + +def _rel(path: Path, root: Path) -> str: + try: + return path.relative_to(root).as_posix() or "." + except ValueError: + return path.name + + +def _path_hash(path: Path) -> str: + return hashlib.sha256(str(path).encode("utf-8", errors="surrogatepass")).hexdigest() + + +def _reject(path: str, reason_code: str, rule_ref: str, **extra: object) -> dict: + return {"path": path, "reason_code": reason_code, "rule_ref": rule_ref, **extra} + + +def _outside_root_rejection(raw: Path, candidate: Path) -> dict: + digest = _path_hash(candidate) + return _reject( + f"outside-root:{digest[:16]}", + "outside-root", + "route.explicit_path.contained", + path_kind="outside-root", + path_sha256=digest, + was_absolute=raw.is_absolute(), + ) + + +def _selected(path: Path, root: Path, key: str) -> dict: + return {"key": key, "path": _rel(path, root)} + + +def _normalize_candidate(root: Path, raw_path: str) -> tuple[Path | None, dict | None]: + if not isinstance(raw_path, str) or not raw_path.strip(): + return None, _reject(str(raw_path), "empty-path", "route.explicit_path.nonempty") + raw = Path(raw_path) + candidate = raw.resolve() if raw.is_absolute() else (root / raw).resolve() + try: + candidate.relative_to(root) + except ValueError: + return None, _outside_root_rejection(raw, candidate) + portable_path = _rel(candidate, root) + if not candidate.exists(): + return None, _reject(portable_path, "not-found", "route.explicit_path.exists") + if not candidate.is_dir(): + return None, _reject(portable_path, "not-directory", "route.explicit_path.directory") + if not (candidate / ".git").exists(): + return None, _reject(portable_path, "not-repository", "route.explicit_path.repository") + return candidate, None + + +def _portable_envelope(envelope: dict) -> dict: + data = json.loads(json.dumps(envelope, sort_keys=True)) + data["root"] = "." + data["recheck"] = { + **data.get("recheck", {}), + "command": "index route --root ROOT --path PATH --json", + } + return data + + +def _empty_receipt(root: Path, *, paths: Sequence[str], rejected: list[dict]) -> dict: + return { + "schema": SCHEMA, + "status": "UNVERIFIABLE", + "failure_codes": ["no_selected_repositories"], + "root": {"sha256_prefix": _root_hash(root)}, + "mode": "explicit-path", + "selection": {"selected": [], "rejected": rejected}, + "reconciliation": { + "candidate_count": len(paths), + "selected_count": 0, + "rejected_count": len(rejected), + "omitted_count": 0, + }, + "envelope": None, + "privacy": {"absolute_paths_included": False}, + "recheck": {"command": "index route --root ROOT --path PATH --json"}, + } + + +def build_route( + root: Path | str, + *, + paths: Sequence[str], + token_budget: int = 1200, + hops: int | None = None, +) -> dict: + """Build a route envelope for explicit repository paths without workspace discovery.""" + root_path = Path(root).resolve() + raw_paths = [str(item) for item in paths] + if token_budget < 1: + raise ValueError("budget must be a positive integer") + if hops is not None and hops < 0: + raise ValueError("hops must be non-negative") + rejected: list[dict] = [] + selected_paths: list[Path] = [] + seen: set[Path] = set() + for raw_path in raw_paths: + candidate, rejection = _normalize_candidate(root_path, raw_path) + if rejection is not None: + rejected.append(rejection) + continue + assert candidate is not None + if candidate in seen: + rejected.append(_reject(_rel(candidate, root_path), "duplicate-path", "route.explicit_path.unique")) + continue + seen.add(candidate) + selected_paths.append(candidate) + if not selected_paths: + return _empty_receipt(root_path, paths=raw_paths, rejected=rejected) + + keyed = repo_key_map(root_path, sorted(selected_paths), include_root_repo=True) + graph = build_graph(keyed, executor="thread") + envelope = _portable_envelope(build_context_envelope( + graph, + root=root_path, + token_budget=token_budget, + hops=hops, + )) + envelope_verified = envelope.get("verification_verdict") == "MATCH" + if not envelope_verified: + status = "UNVERIFIABLE" + failure_codes = ["envelope_unverifiable"] + if rejected: + failure_codes.append("candidate_rejected") + elif rejected: + status = "PARTIAL" + failure_codes = ["candidate_rejected"] + else: + status = "MATCH" + failure_codes = [] + return { + "schema": SCHEMA, + "status": status, + "failure_codes": failure_codes, + "root": {"sha256_prefix": _root_hash(root_path)}, + "mode": "explicit-path", + "selection": { + "selected": [ + _selected(path, root_path, key) + for key, path in sorted(keyed.items(), key=lambda item: item[0]) + ], + "rejected": rejected, + }, + "reconciliation": { + "candidate_count": len(raw_paths), + "selected_count": len(keyed), + "rejected_count": len(rejected), + "omitted_count": 0, + }, + "envelope": envelope, + "privacy": {"absolute_paths_included": False}, + "recheck": {"command": "index route --root ROOT --path PATH --json"}, + } + + +def cmd_route(args) -> int: + try: + payload = build_route( + args.root, + paths=args.paths, + token_budget=args.budget, + hops=args.hops, + ) + except ValueError as exc: + raise SystemExit(str(exc)) from exc + if args.json: + print(json.dumps(payload, indent=2, sort_keys=True)) + else: + print( + f"route status={payload['status']} " + f"selected={payload['reconciliation']['selected_count']} " + f"rejected={payload['reconciliation']['rejected_count']}" + ) + return 0 if payload["status"] == "MATCH" else 2 + + +def call_route(args: dict, *, response_id=None) -> dict: + if not isinstance(args, dict): + raise ValueError("route arguments must be an object") + allowed = {"root", "paths", "budget", "hops"} + if set(args) - allowed: + raise ValueError("unexpected route arguments") + root = args.get("root") + paths = args.get("paths") + if not isinstance(root, str) or not root.strip(): + raise ValueError("root must be a non-empty string") + if ( + not isinstance(paths, list) + or not paths + or any(not isinstance(item, str) for item in paths) + ): + raise ValueError("paths must be a non-empty list of strings") + budget = args.get("budget", 1200) + if type(budget) is not int or budget < 1: + raise ValueError("budget must be a positive integer") + hops = args.get("hops") + if hops is not None and (type(hops) is not int or hops < 0): + raise ValueError("hops must be a non-negative integer") + return build_route( + Path(root), + paths=paths, + token_budget=budget, + hops=hops, + ) + + +def tool_definition() -> dict: + return { + "name": "index.route", + "description": ( + "Build a receipt-backed context envelope for explicit repository paths " + "without discovering unrelated workspace repositories." + ), + "inputSchema": { + "type": "object", + "additionalProperties": False, + "properties": { + "root": {"type": "string", "minLength": 1}, + "paths": { + "type": "array", + "minItems": 1, + "items": {"type": "string"}, + }, + "budget": {"type": "integer", "minimum": 1, "default": 1200}, + "hops": {"type": "integer", "minimum": 0}, + }, + "required": ["root", "paths"], + }, + } diff --git a/client-plugin/server/src/index_graph/router.py b/client-plugin/server/src/index_graph/router.py new file mode 100644 index 0000000..96babc2 --- /dev/null +++ b/client-plugin/server/src/index_graph/router.py @@ -0,0 +1,111 @@ +"""Emit a deterministic, evidence-carrying workspace router (CLAUDE.md / AGENTS.md). + +Derived from the dependency graph and the docs atlas; re-runnable, never hand-edited. +This replaces the index.md + read-first + brief that developers maintain by hand: a +model opening the workspace reads one map that says where things live, what is an entry +point, what is the core, and which docs describe what. Every line traces to a graph fact. +""" +from __future__ import annotations + + +def render_router(pack: dict, *, max_docs: int = 500, max_deps: int = 30) -> str: + relations = [r for r in pack.get("relations", []) if not r.get("external")] + roles = pack.get("roles", {}) + repos = sorted(r["name"] for r in pack.get("repos", [])) + deps_out: dict[str, dict[str, str]] = {} + deps_in: dict[str, set[str]] = {} + for r in relations: + frm, to = r.get("from"), r.get("to") + if to: + deps_out.setdefault(frm, {})[to] = _relation_label(r) + deps_in.setdefault(to, set()).add(frm) + + L = ["# Workspace map", "", + "Generated by `index` from evidence. Re-run `index router --root .` to refresh; " + "do not hand-edit.", ""] + + entry = [n for n in repos if "entrypoint" in roles.get(n, [])] + if entry: + L.append("## Entry points") + for n in entry: + out = _deps_label(deps_out.get(n, {}), max_items=max_deps) or "nothing internal" + L.append(f"- `{n}` starts here; depends on {out}") + L.append("") + + hubs = [n for n in repos if "hub" in roles.get(n, [])] + if hubs: + L.append("## Core (most depended-on, change with care)") + for n in hubs: + used = ", ".join(sorted(deps_in.get(n, []))) or "nothing" + L.append(f"- `{n}` is used by {used}") + L.append("") + + L.append("## Where things live") + for n in repos: + rs = ", ".join(roles.get(n, [])) or "unclassified" + out = _deps_label(deps_out.get(n, {}), max_items=max_deps) + dep = f"; depends on {out}" if out else "" + L.append(f"- `{n}` ({rs}){dep}") + L.append("") + + describes = sorted((e["from"], e["to"]) for e in pack.get("knowledge_edges", []) + if e.get("type") == "describes") + if describes: + L.append("## Docs") + capped = describes[:max(0, max_docs)] + for doc, repo in capped: + L.append(f"- `{doc}` describes `{repo}`") + omitted = len(describes) - len(capped) + if omitted > 0: + L.append( + f"- ... {omitted} doc edge(s) omitted from this router view; " + "run `index atlas --json` for the full evidence pack" + ) + L.append("") + + L.extend(_deep_dives(pack.get("repo_dirs", {}), repos)) + return "\n".join(L).rstrip() + "\n" + + +def _deep_dives(repo_dirs: dict, repos: list[str]) -> list[str]: + """Point an agent at the verified per-repo wiki, so the workspace map is + the entry to a deeper, re-checkable view rather than the whole story.""" + if not repo_dirs: + return [] + lines = ["## Per-repo deep dives", + "For a verified per-repo wiki (pages derived from the module graph, " + "sealed and commit-pinned), run:"] + for name in repos: + target = repo_dirs.get(name, name) + lines.append(f"- `{name}`: `index wiki --root {target}`") + lines.append("") + return lines + + +def _deps_label(deps: dict[str, str], *, max_items: int) -> str: + items = sorted(deps.items()) + labels = [label for _name, label in items[:max(0, max_items)]] + omitted = len(items) - len(labels) + if omitted > 0: + labels.append(f"... {omitted} more") + return ", ".join(labels) + + +def _relation_label(relation: dict) -> str: + to = str(relation.get("to") or "") + evidence = _first_signal(relation.get("signals")) + return f"{to} [{evidence}]" if evidence else to + + +def _first_signal(value: object) -> str: + if not isinstance(value, list): + return "" + for signal in value: + if not isinstance(signal, dict): + continue + file = signal.get("file") + if not isinstance(file, str) or not file: + continue + line = signal.get("line") + return f"{file}:{line}" if isinstance(line, int) and line > 0 else file + return "" diff --git a/client-plugin/server/src/index_graph/router_inventory.py b/client-plugin/server/src/index_graph/router_inventory.py new file mode 100644 index 0000000..2504fc3 --- /dev/null +++ b/client-plugin/server/src/index_graph/router_inventory.py @@ -0,0 +1,310 @@ +"""Operation-local router inventory for shared traversal work.""" +from __future__ import annotations + +import os +from collections.abc import Callable +from dataclasses import dataclass +from pathlib import Path +from time import perf_counter + +from .config import Config, load_config +from .graph.walk import ( + DirectoryMembershipSnapshot, + EXCLUDE_DIRS, + PreloadedFileListing, + directory_membership_from_walk, +) +from .knowledge.docs import _MD_SUFFIXES +from .scan import ( + ScanBudget, + ScanBudgetExceeded, + enforce_interactive_repo_limit, + repo_key_map, + _warn, +) + + +class RouterInventoryInterrupted(RuntimeError): + """Raised when inventory traversal is cooperatively interrupted.""" + + def __init__(self, *, root: Path, path: Path): + self.root = Path(root) + self.path = Path(path) + super().__init__(f"router inventory interrupted at {self.path.name!r}") + + +@dataclass(frozen=True) +class RouterInventory: + root: Path + repo_paths: dict[str, Path] + repo_dirs: dict[str, str] + repo_file_lists: dict[str, PreloadedFileListing] + router_doc_paths: tuple[Path, ...] + skipped: tuple[str, ...] + stats: dict[str, int] + + +def _relative(path: Path, root: Path) -> str: + try: + return path.relative_to(root).as_posix() or "." + except ValueError: + return path.name + + +def _rel_to_root(root: Path, path: Path) -> str: + rel = path.relative_to(root).as_posix() + return "" if rel == "." else rel + + +def _is_relative_to(path: Path, base: Path) -> bool: + try: + path.relative_to(base) + except ValueError: + return False + return True + + +def _walk_ancestors(path: Path, root: Path): + current = path + while True: + yield current + if current == root: + return + parent = current.parent + if parent == current: + return + current = parent + + +def _nearest_ancestor(path: Path, candidates: set[Path], root: Path) -> Path | None: + for current in _walk_ancestors(path, root): + if current in candidates: + return current + return None + + +def _inside_config_prune(path: Path, root: Path, prune: frozenset[str]) -> bool: + try: + parts = path.relative_to(root).parts + except ValueError: + return False + return any(part in prune for part in parts) + + +def _blocked_by_repo_boundary(repo: Path, root: Path, repo_markers: set[Path], config: Config) -> bool: + if config.descend_into_repos: + return False + for marker in repo_markers: + if marker in {repo, root}: + continue + if _is_relative_to(repo, marker): + return True + return False + + +def _discover_repo_paths( + root: Path, + repo_markers: set[Path], + config: Config, + *, + budget: ScanBudget, + skipped: list[str], +) -> dict[str, Path]: + repos = [ + repo + for repo in repo_markers + if not _inside_config_prune(repo, root, config.prune) + and not _blocked_by_repo_boundary(repo, root, repo_markers, config) + ] + if budget.exhausted: + raise ScanBudgetExceeded(root=root, budget=budget, repo_count=len(repos), skipped=skipped) + repos = sorted(set(repos), key=lambda path: path.relative_to(root).as_posix().lower()) + keyed = repo_key_map(root, repos, include_root_repo=config.include_root_repo) + enforce_interactive_repo_limit(len(keyed), budget_ms=budget.budget_ms) + return keyed + + +def _assign_repo_file_lists( + root: Path, + files: list[Path], + directory_snapshots: dict[Path, DirectoryMembershipSnapshot], + repo_paths: dict[str, Path], + repo_markers: set[Path], + unreadable_dirs: list[Path], +) -> tuple[dict[str, PreloadedFileListing], int]: + repo_by_root = {path: name for name, path in repo_paths.items()} + repo_roots = set(repo_by_root) + physical_roots = { + name: directory_snapshots[path].path for name, path in repo_paths.items() + } + physical_markers = {directory_snapshots[path].path for path in repo_markers} + buckets: dict[str, list[Path]] = {name: [] for name in repo_paths} + seal_buckets: dict[str, dict[Path, DirectoryMembershipSnapshot]] = { + name: {} for name in repo_paths + } + repo_error_names: set[str] = set() + + for error_dir in unreadable_dirs: + owner = _nearest_ancestor(error_dir, repo_roots, root) + if owner is None: + continue + boundary = _nearest_ancestor(error_dir, repo_markers, root) + if boundary is None or boundary == owner: + repo_error_names.add(repo_by_root[owner]) + + for path in files: + owner = _nearest_ancestor(path.parent, repo_roots, root) + if owner is None: + continue + boundary = _nearest_ancestor(path.parent, repo_markers, root) + if boundary is not None and boundary != owner: + continue + name = repo_by_root[owner] + if name not in repo_error_names: + physical_path = directory_snapshots[path.parent].path / path.name + physical_owner = _nearest_ancestor(physical_path.parent, physical_markers, physical_roots[name]) + if (physical_owner == physical_roots[name] + and _is_relative_to(physical_path, physical_roots[name])): + buckets[name].append(physical_path) + + for directory, snapshot in directory_snapshots.items(): + owner = _nearest_ancestor(directory, repo_roots, root) + if owner is None: + continue + boundary = _nearest_ancestor(directory, repo_markers, root) + if boundary is None or boundary == owner or directory == boundary: + name = repo_by_root[owner] + physical_boundary = _nearest_ancestor(snapshot.path, physical_markers, physical_roots[name]) + if (name not in repo_error_names + and _is_relative_to(snapshot.path, physical_roots[name]) + and physical_boundary in {physical_roots[name], snapshot.path}): + seal_buckets[name][directory] = snapshot + + for marker in repo_markers: + parent_owner = _nearest_ancestor(marker.parent, repo_roots, root) + if parent_owner is None or parent_owner == marker: + continue + parent_name = repo_by_root[parent_owner] + snapshot = directory_snapshots.get(marker) + if (snapshot is not None and parent_name not in repo_error_names + and _is_relative_to(snapshot.path, physical_roots[parent_name])): + seal_buckets[parent_name][marker] = snapshot + + return { + name: PreloadedFileListing( + files=tuple(paths), + captured_root=physical_roots[name], + directory_snapshots=tuple( + snapshot + for _path, snapshot in sorted( + seal_buckets[name].items(), + key=lambda item: _relative(item[0], root), + ) + ), + ) + for name, paths in buckets.items() + if name not in repo_error_names + }, len(repo_error_names) + + +def build_router_inventory( + root: Path | str, + *, + config: Config | None = None, + budget_ms: int | None = None, + checkpoint: Callable[[Path], bool] | None = None, +) -> RouterInventory: + """Walk once and derive router docs plus graph-scoped repo file listings. + + This is a traversal cache only. Source freshness continues to come from the + exact bytes read by graph fingerprints and resolvers after this inventory is + handed to `build_graph`. + """ + root = Path(root).resolve() + config = config or load_config(None, root) + budget = ScanBudget(budget_ms) + files: list[Path] = [] + doc_paths: list[Path] = [] + repo_markers: set[Path] = set() + skipped: list[str] = [] + unreadable_dirs: list[Path] = [] + directory_snapshots: dict[Path, DirectoryMembershipSnapshot] = {} + stats = { + "physical_walks": 1, + "physical_dirs": 0, + "physical_files": 0, + "physical_file_bytes": 0, + "physical_stat_unreadable": 0, + "traversal_ms": 0, + "repo_count": 0, + "router_doc_paths": 0, + "repo_file_paths": 0, + "repo_inventory_fallbacks": 0, + } + + def onerror(exc: OSError) -> None: + raw = getattr(exc, "filename", None) + path = Path(raw) if raw else root + if not path.is_absolute(): + path = root / path + unreadable_dirs.append(path) + skipped.append(str(raw or path)) + _warn(f"warning: skipped unreadable directory during router inventory: {exc}") + + started = perf_counter() + for dirpath, dirnames, filenames in os.walk(root, onerror=onerror): + # Keep the workspace spelling: resolving a junction can move this path + # outside root and break relative repository keys and document routes. + current = Path(dirpath) + if checkpoint is not None and not checkpoint(current): + skipped.append(f"router-inventory-checkpoint-rejected:{current}") + dirnames[:] = [] + raise RouterInventoryInterrupted(root=root, path=current) + if budget.budget_ms > 0 and not budget.checkpoint(current): + skipped.append(f"scan-budget-exhausted:{current}") + _warn(f"warning: router inventory budget exhausted at {current}") + dirnames[:] = [] + break + + stats["physical_dirs"] += 1 + directory_snapshots[current] = directory_membership_from_walk(current, dirnames, filenames) + if ".git" in dirnames or ".git" in filenames: + repo_markers.add(current) + + dirnames[:] = sorted((name for name in dirnames if name not in EXCLUDE_DIRS), key=str.lower) + for filename in sorted(filenames): + path = current / filename + files.append(path) + stats["physical_files"] += 1 + try: + stats["physical_file_bytes"] += path.stat().st_size + except OSError: + stats["physical_stat_unreadable"] += 1 + if filename.endswith(_MD_SUFFIXES): + doc_paths.append(path) + stats["traversal_ms"] = int((perf_counter() - started) * 1000) + + repo_paths = _discover_repo_paths(root, repo_markers, config, budget=budget, skipped=skipped) + repo_dirs = {name: _rel_to_root(root, path) for name, path in repo_paths.items()} + repo_file_lists, fallback_count = _assign_repo_file_lists( + root, + files, + directory_snapshots, + repo_paths, + repo_markers, + unreadable_dirs, + ) + stats["repo_count"] = len(repo_paths) + stats["router_doc_paths"] = len(doc_paths) + stats["repo_file_paths"] = sum(len(listing.files) for listing in repo_file_lists.values()) + stats["repo_inventory_fallbacks"] = fallback_count + + return RouterInventory( + root=root, + repo_paths=repo_paths, + repo_dirs=repo_dirs, + repo_file_lists=repo_file_lists, + router_doc_paths=tuple(sorted(doc_paths, key=lambda path: _relative(path, root))), + skipped=tuple(skipped), + stats=stats, + ) diff --git a/client-plugin/server/src/index_graph/router_job_surface.py b/client-plugin/server/src/index_graph/router_job_surface.py new file mode 100644 index 0000000..aaab227 --- /dev/null +++ b/client-plugin/server/src/index_graph/router_job_surface.py @@ -0,0 +1,78 @@ +"""Shared argument contract for the local router-job CLI and MCP adapters.""" +from __future__ import annotations + +import json +from pathlib import Path + +from . import router_jobs + +_OPERATIONS = { + "status": "read_router_job_status", + "result": "read_router_job_result", + "cancel": "cancel_router_job", + "resume": "resume_router_job", +} + + +def call_router_job(action: str, args: dict) -> dict: + if not isinstance(args, dict): + raise ValueError("router job arguments must be an object") + allowed = {"root", "max_docs", "budget_ms", "no_cache"} if action == "start" else {"job_id"} + if set(args) - allowed: + raise ValueError("unexpected router job arguments") + if action == "start": + root = args.get("root") + if not isinstance(root, str) or not root.strip(): + raise ValueError("root must be a non-empty string") + values = {key: args.get(key, default) for key, default in + (("max_docs", 500), ("budget_ms", 0))} + for key, value in values.items(): + if type(value) is not int or value < 0: + raise ValueError(f"{key} must be a non-negative integer") + no_cache = args.get("no_cache", False) + if type(no_cache) is not bool: + raise ValueError("no_cache must be a boolean") + return router_jobs.start_router_job(Path(root), **values, use_cache=not no_cache) + if action not in _OPERATIONS: + raise ValueError("unknown router job operation") + job_id = args.get("job_id") + if not isinstance(job_id, str) or not job_id: + raise ValueError("job_id must be a non-empty string") + return getattr(router_jobs, _OPERATIONS[action])(job_id) + + +def cmd_router_job(args) -> int: + payload = ({"root": str(args.root), "max_docs": args.max_docs, + "budget_ms": args.budget_ms, "no_cache": args.no_cache} + if args.action == "start" else {"job_id": args.job_id}) + try: + receipt = call_router_job(args.action, payload) + except (OSError, ValueError, router_jobs.RouterJobError) as exc: + print(json.dumps({"schema": "index.router-job-error/v1", "status": "failed", + "error_type": type(exc).__name__, "message": str(exc)})) + return 1 + print(json.dumps(receipt, indent=2, sort_keys=True)) + return 1 if receipt.get("status") == "failed" else 0 + + +def tool_definitions() -> list[dict]: + start_properties = { + "root": {"type": "string", "minLength": 1}, + "max_docs": {"type": "integer", "minimum": 0, "default": 500}, + "budget_ms": {"type": "integer", "minimum": 0, "default": 0, + "description": "Discovery budget; zero permits a complete background scan."}, + "no_cache": {"type": "boolean", "default": False}, + } + descriptions = { + "start": "Start a durable local router build and return a job receipt promptly. State and output remain private local files.", + "status": "Read current router job progress and recovery state.", + "result": "Retrieve router text only from a complete, validated local job result.", + "cancel": "Request cooperative cancellation; acknowledgment is reported separately.", + "resume": "Retry a stopped router job with its original request and fresh discovery.", + } + return [{"name": f"index.router.job.{action}", "description": description, + "inputSchema": {"type": "object", "additionalProperties": False, + "properties": start_properties if action == "start" else { + "job_id": {"type": "string", "minLength": 1}}, + "required": ["root"] if action == "start" else ["job_id"]}} + for action, description in descriptions.items()] diff --git a/client-plugin/server/src/index_graph/router_job_telemetry.py b/client-plugin/server/src/index_graph/router_job_telemetry.py new file mode 100644 index 0000000..6a95eaf --- /dev/null +++ b/client-plugin/server/src/index_graph/router_job_telemetry.py @@ -0,0 +1,50 @@ +"""Private router-job telemetry helpers.""" +from __future__ import annotations + +from time import perf_counter +from typing import Any, Callable + + +class RouterJobTelemetry: + def __init__(self, started: float, append_event: Callable[[dict[str, Any]], None]) -> None: + self._started = started + self._append_event = append_event + self.phase_timings_ms: dict[str, int] = {} + self.graph_cache: dict[str, object] | None = None + self.router_inventory: dict[str, object] | None = None + self.router_docs: dict[str, object] | None = None + self._events: list[dict[str, Any]] = [] + + def status_fields(self) -> dict[str, Any]: + fields: dict[str, Any] = {} + if self.phase_timings_ms: + fields["phase_timings_ms"] = dict(self.phase_timings_ms) + if self.graph_cache is not None: + fields["graph_cache"] = dict(self.graph_cache) + if self.router_inventory is not None: + fields["router_inventory"] = dict(self.router_inventory) + if self.router_docs is not None: + fields["router_docs"] = dict(self.router_docs) + return fields + + def record_timing(self, step: str, step_started: float, **counts: Any) -> None: + self.record_duration(step, int((perf_counter() - step_started) * 1000), **counts) + + def record_duration(self, step: str, duration_ms: int, **counts: Any) -> None: + self.phase_timings_ms[step] = duration_ms + event: dict[str, Any] = { + "phase": "timing", + "step": step, + "duration_ms": duration_ms, + "elapsed_ms": int((perf_counter() - self._started) * 1000), + } + for key, value in counts.items(): + if key in {"content", "repo_path"}: + continue + if isinstance(value, (bool, int, float, str)) or value is None: + event[key] = value + self._events.append(event) + + def flush(self) -> None: + while self._events: + self._append_event(self._events.pop(0)) diff --git a/client-plugin/server/src/index_graph/router_job_utils.py b/client-plugin/server/src/index_graph/router_job_utils.py new file mode 100644 index 0000000..f13b99b --- /dev/null +++ b/client-plugin/server/src/index_graph/router_job_utils.py @@ -0,0 +1,51 @@ +"""Small private helpers for durable router jobs.""" +from __future__ import annotations + +import hashlib +import json +from datetime import datetime +from pathlib import Path +from typing import Any + + +def now() -> str: + return datetime.now().astimezone().isoformat(timespec="seconds") + + +def root_hash(root: Path) -> str: + return hashlib.sha256(str(root.resolve()).encode("utf-8", "surrogateescape")).hexdigest()[:16] + + +def sha256_bytes(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def sha256_text(value: str) -> str: + return sha256_bytes(value.encode("utf-8", "surrogateescape")) + + +def canonical_json(data: dict[str, Any]) -> str: + return json.dumps(data, sort_keys=True, separators=(",", ":"), default=str) + + +def request_sha256(request: dict[str, Any]) -> str: + return sha256_text(canonical_json(dict(request))) + + +def config_sha256(root: Path) -> str: + parts: list[list[str]] = [] + for name in (".index.toml", ".repomap.toml"): + path = root / name + if not path.exists(): + parts.append([name, "missing"]) + continue + try: + parts.append([name, sha256_bytes(path.read_bytes())]) + except OSError: + parts.append([name, "unreadable"]) + return sha256_text(canonical_json({"root": str(root.resolve()), "config": parts})) + + +def safe_error_message(exc: BaseException) -> str: + text = str(exc) or type(exc).__name__ + return " ".join(text.split())[:500] diff --git a/client-plugin/server/src/index_graph/router_jobs.py b/client-plugin/server/src/index_graph/router_jobs.py new file mode 100644 index 0000000..53983fc --- /dev/null +++ b/client-plugin/server/src/index_graph/router_jobs.py @@ -0,0 +1,1086 @@ +"""Durable local jobs for complete workspace router builds. + +This module is intentionally independent from the CLI/MCP wiring. It owns the +private local job state and runs the existing router pipeline in a child process +so timeout-bound callers can start, poll, and fetch complete results without +receiving partial router content. +""" +from __future__ import annotations + +import json +import os +import subprocess +import stat +import sys +import threading +import time +import uuid +from contextlib import contextmanager +from dataclasses import asdict +from pathlib import Path +from time import perf_counter +from typing import Any, BinaryIO + +from . import __version__ +from .graph.build import GraphProgress, build_graph +from .router_job_telemetry import RouterJobTelemetry +from .router_job_utils import ( + config_sha256 as _config_sha256, + now as _now, + request_sha256 as _request_sha256, + root_hash as _root_hash, + safe_error_message as _safe_error_message, + sha256_text as _sha256_text, +) + +_STATUS_SCHEMA = "index.router-job-status/v1" +_RESULT_SCHEMA = "index.router-job-result/v1" +_REQUEST_SCHEMA = "index.router-job-request/v1" +_EVENT_SCHEMA = "index.router-job-event/v1" +_CANCEL_SCHEMA = "index.router-job-cancel/v1" +_CANCEL_FILE = "cancel.request" +_STATE_LOCK_FILE = "state.lock" +_STATUS_FILE = "status.json" +_REQUEST_FILE = "request.json" +_RESULT_FILE = "router.md" +_EVENTS_FILE = "events.jsonl" +_LEASE_FILE = "worker.lock" +_HEARTBEAT_FILE = "worker.heartbeat.json" +_LEASE_SCHEMA = "index.router-job-lease/v1" +_HEARTBEAT_SCHEMA = "index.router-job-heartbeat/v1" +_LOCK_BYTE_COUNT = 1 +_WINDOWS_REPARSE_ATTRIBUTE = 0x400 +_STATE_LOCKS = threading.local() +_STATE_THREAD_LOCKS: dict[str, threading.RLock] = {} +_STATE_THREAD_LOCKS_GUARD = threading.Lock() + +class RouterJobError(ValueError): + """Raised for invalid local job operations.""" + +class RouterJobCancelled(RuntimeError): + """Internal cooperative cancellation signal.""" + + +class RouterJobLostOwnership(RuntimeError): + """Internal signal for a stale worker attempt that must not write status.""" + + +class RouterJobAlreadyActive(RuntimeError): + """Internal signal for a duplicate worker when the active lease is fresh.""" + + +def _ensure_lock_byte(fh: BinaryIO) -> None: + fh.seek(0, os.SEEK_END) + if fh.tell() == 0: + fh.seek(0) + fh.write(b"0") + fh.flush() + fh.seek(0) + + +def _lock_handle(fh: BinaryIO, *, blocking: bool) -> None: + if os.name == "nt": + import msvcrt + + mode = msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK + msvcrt.locking(fh.fileno(), mode, _LOCK_BYTE_COUNT) + return + import fcntl + + flags = fcntl.LOCK_EX + if not blocking: + flags |= fcntl.LOCK_NB + fcntl.flock(fh.fileno(), flags) + + +def _unlock_handle(fh: BinaryIO) -> None: + fh.seek(0) + if os.name == "nt": + import msvcrt + + try: + msvcrt.locking(fh.fileno(), msvcrt.LK_UNLCK, _LOCK_BYTE_COUNT) + except OSError: + pass + return + import fcntl + + fcntl.flock(fh.fileno(), fcntl.LOCK_UN) + + +@contextmanager +def _locked_file(path: Path, *, blocking: bool = True): + path.parent.mkdir(parents=True, exist_ok=True) + fh = path.open("a+b") + locked = False + try: + _ensure_lock_byte(fh) + _lock_handle(fh, blocking=blocking) + locked = True + yield fh + finally: + if locked: + _unlock_handle(fh) + fh.close() + + +@contextmanager +def _state_lock(job_dir: Path): + lock_path = job_dir / _STATE_LOCK_FILE + key = str(lock_path.resolve()) + held = getattr(_STATE_LOCKS, "held", None) + if held is None: + held = set() + _STATE_LOCKS.held = held + if key in held: + yield + return + with _STATE_THREAD_LOCKS_GUARD: + thread_lock = _STATE_THREAD_LOCKS.setdefault(key, threading.RLock()) + with thread_lock: + with _locked_file(lock_path): + held.add(key) + try: + yield + finally: + held.remove(key) + + +def _heartbeat_stale_seconds() -> float: + raw = os.environ.get("INDEX_ROUTER_JOB_HEARTBEAT_STALE_SECONDS", "60") + try: + return max(1.0, float(raw)) + except ValueError: + return 60.0 + + +def _is_terminal(status: dict[str, Any]) -> bool: + return status.get("status") in {"complete", "failed", "cancelled"} + + +def default_job_root() -> Path: + raw = os.environ.get("INDEX_ROUTER_JOB_DIR") + if raw: + return Path(raw) + base = os.environ.get("LOCALAPPDATA") + if base: + return Path(base) / "index_graph" / "router-jobs" + return Path.home() / ".cache" / "index_graph" / "router-jobs" + + +def _job_root(job_root: Path | str | None = None) -> Path: + return Path(job_root) if job_root is not None else default_job_root() + + +def _validate_job_id(job_id: str) -> str: + try: + parsed = uuid.UUID(str(job_id)) + except ValueError as exc: + raise RouterJobError("invalid router job id") from exc + return str(parsed) + + +def _job_dir(job_id: str, job_root: Path | str | None = None) -> Path: + return _job_root(job_root) / _validate_job_id(job_id) + + +def _atomic_write_text(path: Path, text: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.{uuid.uuid4().hex}.tmp") + with tmp.open("w", encoding="utf-8", newline="\n") as fh: + fh.write(text) + tmp.replace(path) + + +def _atomic_write_json(path: Path, data: dict[str, Any]) -> None: + _atomic_write_text(path, json.dumps(data, indent=2, sort_keys=True) + "\n") + + +def _read_json(path: Path) -> dict[str, Any]: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except OSError as exc: + raise RouterJobError(f"cannot read router job state: {path.name}") from exc + except json.JSONDecodeError as exc: + raise RouterJobError(f"invalid router job state: {path.name}") from exc + if not isinstance(data, dict): + raise RouterJobError(f"invalid router job state: {path.name}") + return data + + +def _read_request(job_dir: Path) -> dict[str, Any]: + data = _read_json(job_dir / _REQUEST_FILE) + if data.get("schema") != _REQUEST_SCHEMA: + raise RouterJobError("invalid router job request schema") + if data.get("job_id") != job_dir.name: + raise RouterJobError("router job request id mismatch") + return data + + +def _read_status_file_unlocked(job_dir: Path) -> dict[str, Any]: + data = _read_json(job_dir / _STATUS_FILE) + if data.get("schema") != _STATUS_SCHEMA: + raise RouterJobError("invalid router job status schema") + if data.get("job_id") != job_dir.name: + raise RouterJobError("router job status id mismatch") + return data + + +def _read_status_file(job_dir: Path) -> dict[str, Any]: + with _state_lock(job_dir): + return _read_status_file_unlocked(job_dir) + + +def _read_json_if_present(path: Path) -> dict[str, Any] | None: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError, ValueError): + return None + return data if isinstance(data, dict) else None + + +def _write_status_unlocked(job_dir: Path, status: dict[str, Any]) -> dict[str, Any]: + status = dict(status) + status["schema"] = _STATUS_SCHEMA + status["job_id"] = job_dir.name + status["updated_at"] = _now() + _atomic_write_json(job_dir / _STATUS_FILE, status) + return status + + +def _write_status(job_dir: Path, status: dict[str, Any]) -> dict[str, Any]: + with _state_lock(job_dir): + return _write_status_unlocked(job_dir, status) + + +def _append_event(job_dir: Path, event: dict[str, Any]) -> None: + rec = {"schema": _EVENT_SCHEMA, "job_id": job_dir.name, "recorded_at": _now(), **event} + with (job_dir / _EVENTS_FILE).open("a", encoding="utf-8") as fh: + fh.write(json.dumps(rec, sort_keys=True, separators=(",", ":")) + "\n") + + +def _write_heartbeat(job_dir: Path, run_token: str, *, started_epoch: float) -> None: + payload = { + "schema": _HEARTBEAT_SCHEMA, + "job_id": job_dir.name, + "run_token": run_token, + "pid": os.getpid(), + "started_epoch": started_epoch, + "updated_epoch": time.time(), + } + _atomic_write_json(job_dir / _HEARTBEAT_FILE, payload) + + +def _start_heartbeat(job_dir: Path, run_token: str, *, started_epoch: float) -> tuple[threading.Event, threading.Thread]: + stop = threading.Event() + + def beat() -> None: + while not stop.wait(1.0): + try: + _write_heartbeat(job_dir, run_token, started_epoch=started_epoch) + except OSError: + pass + + _write_heartbeat(job_dir, run_token, started_epoch=started_epoch) + thread = threading.Thread(target=beat, name="index-router-job-heartbeat", daemon=True) + thread.start() + return stop, thread + + +def _worker_lock_is_held(job_dir: Path) -> bool: + lock_path = job_dir / _LEASE_FILE + if not lock_path.exists(): + return False + try: + with _locked_file(lock_path, blocking=False): + return False + except OSError: + return True + + +def _active_worker(status: dict[str, Any], job_dir: Path) -> bool: + run_token = status.get("run_token") + if not isinstance(run_token, str) or not run_token: + return False + return _worker_lock_is_held(job_dir) + + +def _acquire_lease(job_dir: Path, run_token: str) -> BinaryIO: + lease_path = job_dir / _LEASE_FILE + payload = { + "schema": _LEASE_SCHEMA, + "job_id": job_dir.name, + "run_token": run_token, + "pid": os.getpid(), + "started_epoch": time.time(), + } + lock: BinaryIO | None = None + try: + lock = (job_dir / _LEASE_FILE).open("a+b") + _ensure_lock_byte(lock) + _lock_handle(lock, blocking=False) + except OSError as exc: + if lock is not None: + lock.close() + raise RouterJobAlreadyActive("router job already has an active worker lock") from exc + lock.seek(0) + lock.truncate() + lock.write(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8") + b"\n") + lock.flush() + return lock + + +def _release_lease(job_dir: Path, run_token: str, lock: BinaryIO | None = None) -> None: + if lock is not None: + try: + _unlock_handle(lock) + finally: + lock.close() + + +def _cancel_requested(job_dir: Path, run_token: str | None = None) -> bool: + path = job_dir / _CANCEL_FILE + if not path.exists(): + return False + data = _read_json_if_present(path) + if data is None: + return True + marker_token = data.get("run_token") + return not marker_token or marker_token == run_token + + +def _recent_worker_start(status: dict[str, Any]) -> bool: + try: + started = float(status.get("worker_started_epoch")) + except (TypeError, ValueError): + return False + return time.time() - started <= _heartbeat_stale_seconds() + + +def _fail_status( + job_dir: Path, + status: dict[str, Any], + *, + error_type: str, + message: str, +) -> dict[str, Any]: + failed = dict(status) + failed["status"] = "failed" + failed["phase"] = "failed" + failed["error_type"] = error_type + failed["message"] = message + failed["result_available"] = False + failed["finished_at"] = failed.get("finished_at") or _now() + return _write_status_unlocked(job_dir, failed) + + +def _result_file_is_unsafe(path: Path) -> bool: + try: + info = os.lstat(path) + except FileNotFoundError: + return False + if stat.S_ISLNK(info.st_mode): + return True + if getattr(info, "st_file_attributes", 0) & _WINDOWS_REPARSE_ATTRIBUTE: + return True + return not stat.S_ISREG(info.st_mode) + + +def _read_regular_result_text(job_dir: Path) -> str: + result_path = job_dir / _RESULT_FILE + if _result_file_is_unsafe(result_path): + raise RouterJobError("router job result path is a link, reparse point, or non-regular file") + if not result_path.is_file(): + raise FileNotFoundError("router job result is missing") + resolved_job = job_dir.resolve() + resolved_result = result_path.resolve() + if resolved_result.parent != resolved_job: + raise RouterJobError("router job result path escapes its job directory") + flags = os.O_RDONLY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + fd = os.open(str(result_path), flags) + try: + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode): + raise RouterJobError("router job result is not a regular file") + with os.fdopen(fd, "r", encoding="utf-8", errors="surrogateescape", newline=None, closefd=False) as fh: + return fh.read() + finally: + os.close(fd) + + +def _base_status(job_id: str, request: dict[str, Any], *, attempt: int) -> dict[str, Any]: + return { + "schema": _STATUS_SCHEMA, + "job_id": job_id, + "tool_version": __version__, + "source_version": __version__, + "root": request["root"], + "root_sha256_prefix": request["root_sha256_prefix"], + "max_docs": request["max_docs"], + "budget_ms": request["budget_ms"], + "use_cache": request["use_cache"], + "executor": request["executor"], + "status": "running", + "phase": "queued", + "completed_repos": 0, + "total_repos": None, + "attempt": attempt, + "run_token": None, + "pid": None, + "worker_started_epoch": None, + "request_sha256": _request_sha256(request), + "config_sha256": None, + "result_sha256": None, + "result_bytes": None, + "created_at": request["created_at"], + "started_at": None, + "updated_at": _now(), + "finished_at": None, + "result_available": False, + } + + +def create_router_job( + root: Path | str, + *, + job_root: Path | str | None = None, + max_docs: int = 500, + budget_ms: int = 0, + use_cache: bool = True, + executor: str = "process", +) -> dict[str, Any]: + """Create local durable state for a complete router build without spawning it.""" + root_path = Path(root).resolve() + if max_docs < 0: + raise RouterJobError("max_docs must be non-negative") + if budget_ms < 0: + raise RouterJobError("budget_ms must be non-negative") + if executor not in {"thread", "process"}: + raise RouterJobError("executor must be 'thread' or 'process'") + job_id = str(uuid.uuid4()) + job_dir = _job_root(job_root) / job_id + job_dir.mkdir(parents=True, exist_ok=False) + created = _now() + request = { + "schema": _REQUEST_SCHEMA, + "job_id": job_id, + "tool_version": __version__, + "source_version": __version__, + "root": str(root_path), + "root_sha256_prefix": _root_hash(root_path), + "max_docs": int(max_docs), + "budget_ms": int(budget_ms), + "use_cache": bool(use_cache), + "executor": executor, + "created_at": created, + } + _atomic_write_json(job_dir / _REQUEST_FILE, request) + status = _base_status(job_id, request, attempt=1) + _write_status(job_dir, status) + receipt = dict(status) + receipt["job_dir"] = str(job_dir) + return receipt + + +def _spawn_worker(job_dir: Path, run_token: str) -> subprocess.Popen: + args = [sys.executable, "-m", "index_graph.router_jobs", "_worker", str(job_dir), run_token] + flags = getattr(subprocess, "CREATE_NO_WINDOW", 0) + stdout_path = job_dir / "worker.stdout.log" + stderr_path = job_dir / "worker.stderr.log" + stdout = stdout_path.open("ab") + stderr = stderr_path.open("ab") + try: + return subprocess.Popen( + args, + stdin=subprocess.DEVNULL, + stdout=stdout, + stderr=stderr, + cwd=str(Path.cwd()), + close_fds=True, + creationflags=flags, + ) + finally: + stdout.close() + stderr.close() + + +def start_router_job( + root: Path | str, + *, + job_root: Path | str | None = None, + max_docs: int = 500, + budget_ms: int = 0, + use_cache: bool = True, + executor: str = "process", +) -> dict[str, Any]: + """Start a router job and return a durable running receipt quickly.""" + receipt = create_router_job( + root, + job_root=job_root, + max_docs=max_docs, + budget_ms=budget_ms, + use_cache=use_cache, + executor=executor, + ) + job_dir = Path(receipt["job_dir"]) + run_token = uuid.uuid4().hex + with _state_lock(job_dir): + status = _read_status_file_unlocked(job_dir) + status["run_token"] = run_token + status["phase"] = "starting" + status["started_at"] = _now() + status["worker_started_epoch"] = None + _write_status_unlocked(job_dir, status) + try: + proc = _spawn_worker(job_dir, run_token) + except OSError as exc: + with _state_lock(job_dir): + status = _read_status_file_unlocked(job_dir) + status["status"] = "failed" + status["phase"] = "failed" + status["error_type"] = type(exc).__name__ + status["message"] = _safe_error_message(exc) + status["result_available"] = False + status["finished_at"] = _now() + written = _write_status_unlocked(job_dir, status) + written["job_dir"] = str(job_dir) + return written + with _state_lock(job_dir): + current = _read_status_file_unlocked(job_dir) + if current.get("run_token") != run_token or _is_terminal(current): + current["job_dir"] = str(job_dir) + return current + if current.get("phase") != "starting": + current["job_dir"] = str(job_dir) + return current + status = current + status["pid"] = proc.pid + status["phase"] = "started" + status["worker_started_epoch"] = time.time() + written = _write_status_unlocked(job_dir, status) + written["job_dir"] = str(job_dir) + return written + + +def _pid_running(pid: int | None) -> bool: + if not isinstance(pid, int) or pid <= 0: + return False + try: + os.kill(pid, 0) + except OSError: + return False + return True + + +def _recover_running_status(job_dir: Path, status: dict[str, Any]) -> dict[str, Any]: + if status.get("status") not in {"running", "cancellation_requested"}: + if status.get("status") == "complete" and not (job_dir / _RESULT_FILE).is_file(): + return _fail_status( + job_dir, + status, + error_type="MissingResult", + message="router job marked complete but result is missing", + ) + return status + if status.get("run_token") is None: + return status + if _active_worker(status, job_dir): + return status + if _recent_worker_start(status): + return status + if (job_dir / _RESULT_FILE).is_file(): + status = _fail_status( + job_dir, + status, + error_type="UnsealedResult", + message="router job worker exited with an unsealed result", + ) + elif status.get("status") == "cancellation_requested": + status = _fail_status( + job_dir, + status, + error_type="WorkerInterrupted", + message="router job worker stopped before acknowledging cancellation", + ) + else: + status = _fail_status( + job_dir, + status, + error_type="WorkerInterrupted", + message="router job worker lock is no longer active", + ) + return status + + +def read_router_job_status(job_id: str, *, job_root: Path | str | None = None) -> dict[str, Any]: + job_dir = _job_dir(job_id, job_root) + with _state_lock(job_dir): + status = _recover_running_status(job_dir, _read_status_file_unlocked(job_dir)) + receipt = dict(status) + receipt["job_dir"] = str(job_dir) + return receipt + + +def read_router_job_result(job_id: str, *, job_root: Path | str | None = None) -> dict[str, Any]: + status = read_router_job_status(job_id, job_root=job_root) + job_dir = Path(status["job_dir"]) + if status.get("status") != "complete": + return { + "schema": _RESULT_SCHEMA, + "job_id": status["job_id"], + "status": status.get("status"), + "phase": status.get("phase"), + "result_available": False, + "status_receipt": status, + } + with _state_lock(job_dir): + status = _read_status_file_unlocked(job_dir) + if status.get("status") != "complete": + receipt = dict(status) + receipt["job_dir"] = str(job_dir) + return { + "schema": _RESULT_SCHEMA, + "job_id": status["job_id"], + "status": status.get("status"), + "phase": status.get("phase"), + "result_available": False, + "status_receipt": receipt, + } + + def failed_result(error_type: str, message: str) -> dict[str, Any]: + failed = _fail_status(job_dir, status, error_type=error_type, message=message) + receipt = dict(failed) + receipt["job_dir"] = str(job_dir) + return { + "schema": _RESULT_SCHEMA, + "job_id": status["job_id"], + "status": "failed", + "phase": "failed", + "error_type": error_type, + "result_available": False, + "status_receipt": receipt, + } + + try: + request = _read_request(job_dir) + except RouterJobError as exc: + return failed_result("RequestSealInvalid", _safe_error_message(exc)) + expected_request_hash = status.get("request_sha256") + if not isinstance(expected_request_hash, str) or _request_sha256(request) != expected_request_hash: + return failed_result( + "RequestSealInvalid", + "router job request seal does not match complete status receipt", + ) + try: + content = _read_regular_result_text(job_dir) + except FileNotFoundError: + return failed_result("MissingResult", "router job result is missing") + except RouterJobError as exc: + return failed_result("UnsafeResult", _safe_error_message(exc)) + actual_hash = _sha256_text(content) + expected_hash = status.get("result_sha256") + if not isinstance(expected_hash, str) or actual_hash != expected_hash: + return failed_result( + "CorruptResult", + "router job result hash does not match complete status receipt", + ) + receipt = dict(status) + receipt["job_dir"] = str(job_dir) + return { + "schema": _RESULT_SCHEMA, + "job_id": status["job_id"], + "status": "complete", + "phase": "complete", + "result_available": True, + "result_sha256": actual_hash, + "request_sha256": status.get("request_sha256"), + "config_sha256": status.get("config_sha256"), + "content": content, + "status_receipt": receipt, + } + + +def cancel_router_job( + job_id: str, + *, + job_root: Path | str | None = None, + terminate: bool = False, +) -> dict[str, Any]: + job_dir = _job_dir(job_id, job_root) + with _state_lock(job_dir): + observed = _read_status_file_unlocked(job_dir) + marker = { + "schema": _CANCEL_SCHEMA, + "job_id": job_dir.name, + "run_token": observed.get("run_token"), + "attempt": observed.get("attempt"), + "requested_at": _now(), + } + _atomic_write_json(job_dir / _CANCEL_FILE, marker) + status = _read_status_file_unlocked(job_dir) + if status.get("status") == "complete": + status["message"] = "router job was already complete when cancellation was requested" + status["result_available"] = (job_dir / _RESULT_FILE).is_file() + elif _is_terminal(status): + status["message"] = "router job was already terminal when cancellation was requested" + status["result_available"] = False + elif status.get("run_token") != observed.get("run_token") or status.get("attempt") != observed.get("attempt"): + status["message"] = "router job changed attempts before cancellation could be applied" + status["result_available"] = False + else: + status["status"] = "cancellation_requested" + status["cancel_requested_at"] = marker["requested_at"] + status["result_available"] = False + status["message"] = "router job cancellation requested" + if terminate: + status["termination_supported"] = False + status["message"] = ( + "router job cancellation requested; hard termination was not attempted " + "because durable jobs require an owned worker handle, not PID-only identity" + ) + written = _write_status_unlocked(job_dir, status) + written["job_dir"] = str(job_dir) + return written + + +def resume_router_job(job_id: str, *, job_root: Path | str | None = None) -> dict[str, Any]: + job_dir = _job_dir(job_id, job_root) + with _state_lock(job_dir): + status = _recover_running_status(job_dir, _read_status_file_unlocked(job_dir)) + if status.get("status") == "complete": + status = dict(status) + status["job_dir"] = str(job_dir) + return status + if status.get("status") in {"running", "cancellation_requested"} and ( + _active_worker(status, job_dir) or _recent_worker_start(status) + ): + status = dict(status) + status["job_dir"] = str(job_dir) + return status + request = _read_request(job_dir) + try: + (job_dir / _CANCEL_FILE).unlink() + except FileNotFoundError: + pass + next_status = _base_status(job_dir.name, request, attempt=int(status.get("attempt") or 1) + 1) + next_status["phase"] = "restarting" + next_status["started_at"] = _now() + run_token = uuid.uuid4().hex + next_status["run_token"] = run_token + next_status["config_sha256"] = _config_sha256(Path(request["root"]).resolve()) + _write_status_unlocked(job_dir, next_status) + try: + proc = _spawn_worker(job_dir, run_token) + except OSError as exc: + next_status["status"] = "failed" + next_status["phase"] = "failed" + next_status["error_type"] = type(exc).__name__ + next_status["message"] = _safe_error_message(exc) + next_status["result_available"] = False + next_status["finished_at"] = _now() + written = _write_status_unlocked(job_dir, next_status) + written["job_dir"] = str(job_dir) + return written + current = _read_status_file_unlocked(job_dir) + if current.get("run_token") != run_token or _is_terminal(current): + current["job_dir"] = str(job_dir) + return current + if current.get("phase") != "restarting": + current["job_dir"] = str(job_dir) + return current + next_status = dict(current) + next_status["pid"] = proc.pid + next_status["phase"] = "started" + next_status["worker_started_epoch"] = time.time() + written = _write_status_unlocked(job_dir, next_status) + written["job_dir"] = str(job_dir) + return written + + +def run_router_job_worker(job_dir: Path | str, run_token: str | None = None) -> int: + """Run one router job in this process. Public for tests and subprocess entry.""" + job_dir = Path(job_dir).resolve() + request = _read_request(job_dir) + status = _read_status_file(job_dir) + run_token = run_token or status.get("run_token") or uuid.uuid4().hex + if not isinstance(run_token, str) or not run_token: + run_token = uuid.uuid4().hex + started = perf_counter() + worker_instance_id = uuid.uuid4().hex + + def emit(event: dict[str, Any], *, scope: str = "job", terminal: bool = False) -> None: + # Timing records may have been buffered. Stamp their emission, not their + # earlier collection time, on the same clock as all other attempt events. + _append_event(job_dir, { + **event, + "scope": scope, + "status": event["phase"] if terminal else "running", + "terminal": terminal, + "run_token": run_token, + "worker_instance_id": worker_instance_id, + "elapsed_clock": "worker_monotonic", + "elapsed_ms": int((perf_counter() - started) * 1000), + }) + + if status.get("run_token") and status.get("run_token") != run_token: + emit( + { + "phase": "failed", + "error_type": "RouterJobLostOwnership", + "message": "worker token does not match current job status", + }, + scope="attempt", terminal=True, + ) + return 1 + started_epoch = time.time() + stop_heartbeat: threading.Event | None = None + lease_lock: BinaryIO | None = None + telemetry = RouterJobTelemetry(started, lambda event: emit(event, scope="stage")) + + def update(**fields: Any) -> dict[str, Any]: + nonlocal status + with _state_lock(job_dir): + current = _read_status_file_unlocked(job_dir) + if current.get("run_token") and current.get("run_token") != run_token: + raise RouterJobLostOwnership("worker no longer owns this router job attempt") + if current.get("status") in {"complete", "failed", "cancelled"} and fields.get("status") not in { + "complete", + "failed", + "cancelled", + }: + raise RouterJobLostOwnership("worker cannot overwrite terminal router job status") + if current.get("status") == "cancellation_requested" and fields.get("status") == "running": + raise RouterJobCancelled("router job cancellation requested") + if fields.get("status") == "running" and _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + status = { + **current, + **telemetry.status_fields(), + **fields, + "run_token": run_token, + "elapsed_ms": int((perf_counter() - started) * 1000), + } + return _write_status_unlocked(job_dir, status) + + def progress(event: GraphProgress) -> None: + payload = asdict(event) + phase = f"graph_{event.phase}" if event.phase in {"complete", "failed"} else event.phase + payload.update( + phase=phase, + graph_phase=event.phase, + graph_elapsed_ms=event.elapsed_ms, + ) + emit(payload, scope="graph") + update( + status="running", + phase=phase, + completed_repos=event.completed_repos, + total_repos=event.total_repos, + ) + if _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + + try: + lease_lock = _acquire_lease(job_dir, run_token) + stop_heartbeat, _thread = _start_heartbeat(job_dir, run_token, started_epoch=started_epoch) + update( + status="running", + phase=status.get("phase") if status.get("phase") in {"starting", "restarting"} else "started", + pid=os.getpid(), + worker_started_epoch=started_epoch, + started_at=status.get("started_at") or _now(), + request_sha256=_request_sha256(request), + config_sha256=None, + result_sha256=None, + result_bytes=None, + result_available=False, + ) + if _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + root = Path(request["root"]).resolve() + if not root.is_dir(): + raise FileNotFoundError(f"root not found: {root}") + config_started = perf_counter() + config_sha256 = _config_sha256(root) + telemetry.record_timing("config", config_started) + update(status="running", phase="discovering", config_sha256=config_sha256) + discovery_started = perf_counter() + from .knowledge.atlas import build_router_pack + from .knowledge.docs import discover_router_docs + from .router_inventory import build_router_inventory + from .router import render_router + + def inventory_checkpoint(_path: Path) -> bool: + if _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + return True + + inventory = build_router_inventory( + root, + budget_ms=int(request.get("budget_ms", 0)), + checkpoint=inventory_checkpoint, + ) + paths = inventory.repo_paths + telemetry.router_inventory = { + "physical_walks": int(inventory.stats.get("physical_walks", 0)), + "physical_dirs": int(inventory.stats.get("physical_dirs", 0)), + "physical_files": int(inventory.stats.get("physical_files", 0)), + "physical_file_bytes": int(inventory.stats.get("physical_file_bytes", 0)), + "physical_stat_unreadable": int(inventory.stats.get("physical_stat_unreadable", 0)), + "traversal_ms": int(inventory.stats.get("traversal_ms", 0)), + "repo_count": len(paths), + "router_doc_paths": int(inventory.stats.get("router_doc_paths", 0)), + "repo_file_paths": int(inventory.stats.get("repo_file_paths", 0)), + "repo_inventory_fallbacks": int(inventory.stats.get("repo_inventory_fallbacks", 0)), + } + telemetry.record_timing( + "repo_discovery", + discovery_started, + total_repos=len(paths), + physical_walks=telemetry.router_inventory["physical_walks"], + physical_files=telemetry.router_inventory["physical_files"], + ) + if _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + update(status="running", phase="building", completed_repos=0, total_repos=len(paths)) + repo_dirs = inventory.repo_dirs + graph_started = perf_counter() + graph = build_graph( + paths, + executor=str(request.get("executor") or "process"), + use_cache=bool(request.get("use_cache", True)), + on_progress=progress, + file_lists=inventory.repo_file_lists, + ) + telemetry.graph_cache = getattr(graph, "cache_summary", None) + telemetry.record_timing("graph", graph_started, total_repos=len(paths)) + telemetry.flush() + if _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + update(status="running", phase="rendering", completed_repos=len(paths), total_repos=len(paths)) + docs_started = perf_counter() + docs_stats: dict[str, int] = {} + docs = discover_router_docs(root, paths=inventory.router_doc_paths, stats=docs_stats) + telemetry.router_docs = { + "docs": len(docs), + "traversal_ms": int(inventory.stats.get("traversal_ms", 0)), + "physical_walks": int(inventory.stats.get("physical_walks", 0)), + "row_construction_ms": int(docs_stats.get("row_construction_ms", 0)), + } + telemetry.record_timing("router_docs", docs_started, **telemetry.router_docs) + pack_started = perf_counter() + pack = build_router_pack(graph, docs, repo_dirs) + telemetry.record_timing( + "router_pack", + pack_started, + docs=len(docs), + knowledge_edges=len(pack.get("knowledge_edges", ())), + ) + render_started = perf_counter() + text = render_router(pack, max_docs=max(0, int(request.get("max_docs", 500)))) + result_bytes = len(text.encode("utf-8", "surrogateescape")) + telemetry.record_timing("render", render_started, result_bytes=result_bytes) + result_write_started = perf_counter() + result_sha256 = _sha256_text(text) + attempt_result = job_dir / f"router.{run_token}.md" + _atomic_write_text(attempt_result, text) + try: + with _state_lock(job_dir): + current = _read_status_file_unlocked(job_dir) + if current.get("run_token") != run_token: + raise RouterJobLostOwnership("worker no longer owns this router job attempt") + if current.get("status") in {"complete", "failed", "cancelled"}: + raise RouterJobLostOwnership("worker cannot overwrite terminal router job status") + if current.get("status") == "cancellation_requested" or _cancel_requested(job_dir, run_token): + raise RouterJobCancelled("router job cancellation requested") + if _result_file_is_unsafe(job_dir / _RESULT_FILE): + raise RouterJobError("router job result path is unsafe") + attempt_result.replace(job_dir / _RESULT_FILE) + telemetry.phase_timings_ms["result_write"] = int((perf_counter() - result_write_started) * 1000) + status = { + **current, + "status": "complete", + "phase": "complete", + "completed_repos": len(paths), + "total_repos": len(paths), + "result_available": True, + "result_path": str(job_dir / _RESULT_FILE), + "result_sha256": result_sha256, + "result_bytes": result_bytes, + "request_sha256": _request_sha256(request), + "config_sha256": config_sha256, + **telemetry.status_fields(), + "finished_at": _now(), + "elapsed_ms": int((perf_counter() - started) * 1000), + } + status = _write_status_unlocked(job_dir, status) + finally: + try: + attempt_result.unlink() + except FileNotFoundError: + pass + telemetry.record_duration("result_write", telemetry.phase_timings_ms.get("result_write", 0)) + telemetry.flush() + emit({"phase": "complete", "completed_repos": len(paths), "total_repos": len(paths)}, terminal=True) + return 0 + except RouterJobAlreadyActive as exc: + emit( + { + "phase": "failed", + "error_type": type(exc).__name__, + "message": _safe_error_message(exc), + "elapsed_ms": int((perf_counter() - started) * 1000), + }, + scope="attempt", terminal=True, + ) + return 1 + except RouterJobLostOwnership as exc: + emit( + { + "phase": "failed", + "error_type": type(exc).__name__, + "message": _safe_error_message(exc), + "elapsed_ms": int((perf_counter() - started) * 1000), + }, + scope="attempt", terminal=True, + ) + return 1 + except RouterJobCancelled as exc: + update( + status="cancelled", + phase="cancelled", + result_available=False, + error_type=type(exc).__name__, + message=_safe_error_message(exc), + finished_at=_now(), + ) + emit({"phase": "cancelled", "completed_repos": status.get("completed_repos", 0), "total_repos": status.get("total_repos")}, terminal=True) + return 2 + except BaseException as exc: + update( + status="failed", + phase="failed", + result_available=False, + error_type=type(exc).__name__, + message=_safe_error_message(exc), + finished_at=_now(), + ) + emit({"phase": "failed", "completed_repos": status.get("completed_repos", 0), "total_repos": status.get("total_repos")}, terminal=True) + return 1 + finally: + if stop_heartbeat is not None: + stop_heartbeat.set() + _release_lease(job_dir, run_token, lease_lock) + + +def main(argv: list[str] | None = None) -> int: + argv = list(sys.argv[1:] if argv is None else argv) + if len(argv) in {2, 3} and argv[0] == "_worker": + return run_router_job_worker(Path(argv[1]), argv[2] if len(argv) == 3 else None) + raise SystemExit("usage: python -m index_graph.router_jobs _worker JOB_DIR [RUN_TOKEN]") + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/client-plugin/server/src/index_graph/scan.py b/client-plugin/server/src/index_graph/scan.py new file mode 100644 index 0000000..1c078a2 --- /dev/null +++ b/client-plugin/server/src/index_graph/scan.py @@ -0,0 +1,579 @@ +"""Discovery, parallel git fan-out, and map assembly.""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from collections import Counter +from concurrent.futures import ThreadPoolExecutor, as_completed +from dataclasses import dataclass +from datetime import datetime +from pathlib import Path +from time import perf_counter +from typing import Any, Callable + +from .classify import classify +from .config import Config +from .gitmeta import STATUS_ARGS, STATUS_GLOBAL_ARGS, GitMetadataError, repo_metadata, run_git_checked +from .model import SCHEMA_VERSION, Map, RepoRow + + +DEFAULT_INTERACTIVE_BUDGET_MS = 12000 +DEFAULT_INTERACTIVE_REPO_LIMIT = 300 +_MAP_RESUME_ROW_SCHEMA = "index.map-resume-row/v2" + + +class ScanBudget: + """Cooperative wall-clock budget for workspace traversal.""" + + def __init__(self, budget_ms: int | None): + self.budget_ms = int(budget_ms or 0) + self.started = perf_counter() + self.deadline = self.started + (self.budget_ms / 1000.0) if self.budget_ms > 0 else None + self.exhausted = False + self.last_path: Path | None = None + + def checkpoint(self, path: Path) -> bool: + self.last_path = Path(path) + if self.deadline is None: + return True + if perf_counter() <= self.deadline: + return True + self.exhausted = True + return False + + @property + def elapsed_ms(self) -> int: + return int((perf_counter() - self.started) * 1000) + + +class ScanBudgetExceeded(RuntimeError): + """Raised by caller wrappers when a partial discovery must not be authoritative.""" + + def __init__(self, *, root: Path, budget: ScanBudget, repo_count: int, skipped: list | None = None): + self.root = Path(root) + self.budget_ms = budget.budget_ms + self.elapsed_ms = budget.elapsed_ms + self.repo_count = repo_count + self.skipped = list(skipped or []) + self.last_path = str(budget.last_path) if budget.last_path is not None else None + super().__init__( + f"repository discovery budget exhausted after {self.elapsed_ms} ms " + f"(budget_ms={self.budget_ms}, partial_repos={repo_count})" + ) + + +class ScanWorkloadExceeded(RuntimeError): + """Raised when complete discovery is too large for an interactive graph build.""" + + def __init__(self, *, repo_count: int, repo_limit: int, budget_ms: int): + self.repo_count = repo_count + self.repo_limit = repo_limit + self.budget_ms = budget_ms + super().__init__( + f"workspace has {repo_count} repositories, exceeding the interactive graph limit " + f"of {repo_limit} while budget_ms={budget_ms}" + ) + + +def default_interactive_budget_ms() -> int: + raw = os.environ.get("INDEX_INTERACTIVE_BUDGET_MS") + if raw is None or raw == "": + return DEFAULT_INTERACTIVE_BUDGET_MS + try: + value = int(raw) + except ValueError: + return DEFAULT_INTERACTIVE_BUDGET_MS + return max(0, value) + + +def default_interactive_repo_limit() -> int: + raw = os.environ.get("INDEX_INTERACTIVE_REPO_LIMIT") + if raw is None or raw == "": + return DEFAULT_INTERACTIVE_REPO_LIMIT + try: + value = int(raw) + except ValueError: + return DEFAULT_INTERACTIVE_REPO_LIMIT + return max(0, value) + + +def enforce_interactive_repo_limit(repo_count: int, *, budget_ms: int | None) -> None: + if not budget_ms: + return + limit = default_interactive_repo_limit() + if limit and repo_count > limit: + raise ScanWorkloadExceeded(repo_count=repo_count, repo_limit=limit, budget_ms=int(budget_ms)) + + +def _root_hash(root: Path) -> str: + return hashlib.sha256(str(root.resolve()).encode("utf-8")).hexdigest()[:16] + + +def _sha256_text(value: str) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def _sha256_bytes(value: bytes) -> str: + return hashlib.sha256(value).hexdigest() + + +def _warn(message: str) -> None: + try: + print(message, file=sys.stderr) + except UnicodeError: + safe = message.encode("utf-8", "backslashreplace").decode("ascii", "replace") + print(safe, file=sys.stderr) + + +def discover_repos(root: Path, config: Config, *, + skipped: list | None = None, + checkpoint: Callable[[Path], bool] | None = None) -> list[Path]: + root = Path(root) + prune = config.prune + repos: set[Path] = set() + def _onerror(exc: OSError) -> None: + # a directory os.walk could not read narrows the scan: record it so a + # partial scan is a receiptable fact, not a stderr-only warning that a + # certificate can MATCH straight past + if skipped is not None: + skipped.append(getattr(exc, "filename", None) or str(exc)) + _warn(f"warning: skipped unreadable directory during repo discovery: {exc}") + + for dirpath, dirnames, filenames in os.walk(root, onerror=_onerror): + current = Path(dirpath) + if checkpoint is not None and not checkpoint(current): + if skipped is not None: + skipped.append(f"scan-budget-exhausted:{current}") + _warn(f"warning: repository discovery budget exhausted at {current}") + dirnames[:] = [] + break + is_repo = ".git" in dirnames or ".git" in filenames + if is_repo: + repos.add(current) + if current != root and not config.descend_into_repos: + dirnames[:] = [] + continue + dirnames[:] = sorted((name for name in dirnames if name not in prune), key=str.lower) + return sorted(repos, key=lambda p: p.relative_to(root).as_posix().lower()) + + +def repo_key_map( + root: Path, + repos: list[Path], + *, + include_root_repo: bool = False, +) -> dict[str, Path]: + """Map discovered repos to stable keys without dropping duplicate basenames. + + Unique basenames keep the short legacy key. When multiple repos share a + basename, each duplicate gets its relative workspace path as the key. + """ + root = Path(root) + filtered = [ + repo for repo in repos + if include_root_repo or repo != root or len(repos) == 1 + ] + counts = Counter(repo.name for repo in filtered) + keyed: dict[str, Path] = {} + for repo in filtered: + if counts[repo.name] == 1: + key = repo.name + else: + key = repo.relative_to(root).as_posix() + keyed[key] = repo + return keyed + + +def _relative(path: Path, root: Path) -> str: + try: + return path.relative_to(root).as_posix() or "." + except ValueError: + return path.name + + +@dataclass(frozen=True) +class _ResumeEntry: + row: RepoRow + identity: dict[str, Any] + + +def _repo_row_from_meta(repo: Path, root: Path, config: Config, meta: dict[str, Any]) -> RepoRow: + rel = _relative(repo, root) + class_ = classify(rel, True, meta["origin"], config) + origin = "" if class_ in config.omit_origin_classes else meta["origin"] + path = rel if config.portable else str(repo) + markers = tuple(name for name in config.markers if (repo / name).exists()) + return RepoRow( + path=path, class_=class_, branch=meta["branch"], head=meta["head"], + origin=origin, dirty_count=meta["dirty_count"], + untracked_count=meta["untracked_count"], markers=markers, + metadata_status=str(meta.get("metadata_status") or "ok"), + metadata_error=meta.get("metadata_error"), + ) + + +def _repo_row(repo: Path, root: Path, config: Config) -> RepoRow: + return _repo_row_from_meta(repo, root, config, repo_metadata(repo)) + + +def _repo_row_from_json(data: dict[str, Any]) -> RepoRow | None: + try: + return RepoRow( + path=str(data["path"]), + class_=str(data["class"]), + branch=str(data["branch"]), + head=str(data["head"]), + origin=str(data.get("origin", "")), + dirty_count=int(data.get("dirty_count", 0)), + untracked_count=int(data.get("untracked_count", 0)), + markers=tuple(str(item) for item in data.get("markers", [])), + metadata_status=str(data.get("metadata_status") or "ok"), + metadata_error=( + str(data["metadata_error"]) + if data.get("metadata_error") is not None + else None + ), + ) + except (KeyError, TypeError, ValueError): + return None + + +def repo_status_signature(repo: Path) -> str | None: + result = run_git_checked(repo, STATUS_ARGS, global_args=STATUS_GLOBAL_ARGS) + if not result.ok: + return None + return _sha256_text(result.stdout) + + +def _config_resume_identity(config: Config) -> str: + payload = { + "rules": [(rule.pattern, rule.class_) for rule in config.rules], + "prune": sorted(config.prune), + "markers": list(config.markers), + "descend_into_repos": config.descend_into_repos, + "include_root_repo": config.include_root_repo, + "omit_origin_classes": sorted(config.omit_origin_classes), + "portable": config.portable, + } + return _sha256_text(json.dumps(payload, sort_keys=True, separators=(",", ":"), default=str)) + + +def _file_identity(path: Path) -> dict[str, Any]: + try: + stat = path.stat() + except FileNotFoundError: + return {"state": "missing"} + except OSError as exc: + return {"state": "unreadable", "error": type(exc).__name__} + if path.is_dir(): + return {"state": "directory"} + try: + digest = _sha256_bytes(path.read_bytes()) + except FileNotFoundError: + return {"state": "missing"} + except OSError as exc: + return { + "state": "unreadable", + "error": type(exc).__name__, + "mtime_ns": stat.st_mtime_ns, + "size": stat.st_size, + } + return { + "state": "file", + "size": stat.st_size, + "sha256": digest, + } + + +def _git_dir(repo: Path) -> Path | None: + marker = repo / ".git" + try: + if marker.is_dir(): + return marker.resolve() + text = marker.read_text(encoding="utf-8", errors="replace").strip() + except OSError: + return None + prefix = "gitdir:" + if not text.lower().startswith(prefix): + return None + raw = text[len(prefix):].strip() + if not raw: + return None + path = Path(raw) + if not path.is_absolute(): + path = marker.parent / path + try: + return path.resolve() + except OSError: + return None + + +def _common_git_dir(git_dir: Path) -> Path: + common = git_dir / "commondir" + try: + raw = common.read_text(encoding="utf-8", errors="replace").strip() + except OSError: + return git_dir + if not raw: + return git_dir + path = Path(raw) + if not path.is_absolute(): + path = git_dir / path + try: + return path.resolve() + except OSError: + return git_dir + + +def _git_control_identity(repo: Path) -> str: + marker = repo / ".git" + parts: list[tuple[str, dict[str, Any]]] = [(".git", _file_identity(marker))] + git_dir = _git_dir(repo) + if git_dir is None: + return _sha256_text(json.dumps(parts, sort_keys=True, separators=(",", ":"), default=str)) + common_dir = _common_git_dir(git_dir) + for base_name, base in (("gitdir", git_dir), ("commondir", common_dir)): + for name in ("HEAD", "config", "index", "packed-refs", "commondir"): + parts.append((f"{base_name}/{name}", _file_identity(base / name))) + try: + head_text = (git_dir / "HEAD").read_text(encoding="utf-8", errors="replace").strip() + except OSError: + head_text = "" + if head_text.startswith("ref:"): + ref = head_text.split(":", 1)[1].strip() + if ref: + for base_name, base in (("gitdir", git_dir), ("commondir", common_dir)): + parts.append((f"{base_name}/{ref}", _file_identity(base / ref))) + return _sha256_text(json.dumps(parts, sort_keys=True, separators=(",", ":"), default=str)) + + +def _repo_resume_identity( + repo: Path, + root: Path, + config: Config, + *, + status_signature: str | None = None, +) -> dict[str, Any] | None: + signature = status_signature or repo_status_signature(repo) + if signature is None: + return None + marker_state = [] + for name in config.markers: + path = repo / name + try: + exists = path.exists() + except OSError: + exists = False + marker_state.append([name, exists]) + return { + "repo_relative": _relative(repo, root), + "config": _config_resume_identity(config), + "markers": marker_state, + "git_status": signature, + "git_control": _git_control_identity(repo), + } + + +def _load_resume_rows(resume_state: Path | None, root: Path) -> dict[str, _ResumeEntry]: + if resume_state is None: + return {} + rows: dict[str, _ResumeEntry] = {} + root_id = _root_hash(root) + try: + lines = resume_state.read_text(encoding="utf-8").splitlines() + except OSError: + return rows + for line in lines: + if not line.strip(): + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + if rec.get("schema") != _MAP_RESUME_ROW_SCHEMA: + continue + if rec.get("root_sha256_prefix") != root_id: + continue + rel = rec.get("repo_relative") + row = _repo_row_from_json(rec.get("row") or {}) + identity = rec.get("identity") + if isinstance(rel, str) and row is not None and isinstance(identity, dict): + rows[rel] = _ResumeEntry(row, identity) + return rows + + +def _append_resume_row( + resume_state: Path | None, + root: Path, + repo: Path, + row: RepoRow, + identity: dict[str, Any] | None, +) -> None: + if resume_state is None or identity is None: + return + rec = { + "schema": _MAP_RESUME_ROW_SCHEMA, + "root_sha256_prefix": _root_hash(root), + "repo_relative": _relative(repo, root), + "identity": identity, + "row": row.to_json(), + } + try: + resume_state.parent.mkdir(parents=True, exist_ok=True) + with resume_state.open("a", encoding="utf-8") as fh: + fh.write(json.dumps(rec, sort_keys=True, separators=(",", ":")) + "\n") + except OSError as exc: + _warn(f"warning: could not write map resume state {resume_state}: {exc}") + + +def _unknown_repo_row(repo: Path, root: Path, config: Config, exc: Exception) -> RepoRow: + rel = _relative(repo, root) + _warn(f"warning: failed to scan {rel}: {exc}") + class_ = classify(rel, True, "", config) if isinstance(exc, GitMetadataError) else "unknown" + return RepoRow( + path=(rel if config.portable else str(repo)), class_=class_, + branch="unknown", head="unknown", origin="", dirty_count=0, + untracked_count=0, markers=(), metadata_status="unknown", + metadata_error=type(exc).__name__, + ) + + +def _safe_repo_row(repo: Path, root: Path, config: Config) -> RepoRow: + # Spec §9: one repo's failure must degrade to a row, never crash the scan. + try: + return _repo_row(repo, root, config) + except Exception as exc: + return _unknown_repo_row(repo, root, config, exc) + + +def _safe_resume_entry(repo: Path, root: Path, config: Config) -> tuple[RepoRow, dict[str, Any] | None]: + try: + meta = repo_metadata(repo) + row = _repo_row_from_meta(repo, root, config, meta) + identity = _repo_resume_identity( + repo, + root, + config, + status_signature=meta.get("status_signature"), + ) + if row.metadata_status != "ok": + identity = None + return row, identity + except Exception as exc: + return _unknown_repo_row(repo, root, config, exc), None + + +def _cached_resume_identity(repo: Path, root: Path, config: Config) -> dict[str, Any] | None: + try: + return _repo_resume_identity(repo, root, config) + except Exception: + return None + + +def _top_level(root: Path, config: Config) -> list[dict[str, Any]]: + entries: list[dict[str, Any]] = [] + for path in sorted(root.iterdir(), key=lambda item: item.name.lower()): + if path.name == ".git": + continue + try: + stat = path.stat() + is_dir = path.is_dir() + except OSError as exc: + _warn(f"warning: skipped top-level entry {path.name}: {exc}") + continue + entries.append({ + "name": path.name, + "kind": "directory" if is_dir else "file", + "class": classify(path.name, False, "", config), + "bytes": None if is_dir else stat.st_size, + "modified": datetime.fromtimestamp(stat.st_mtime).astimezone().isoformat(timespec="seconds"), + }) + return entries + + +def _map_rows(root: Path, config: Config, repos: list[Path], resume_state: Path | None) -> list[RepoRow]: + cached = _load_resume_rows(resume_state, root) + rows_by_rel: dict[str, RepoRow] = {} + pending: list[Path] = [] + if resume_state is None: + with ThreadPoolExecutor(max_workers=config.jobs) as pool: + futures = {pool.submit(_safe_repo_row, repo, root, config): repo for repo in repos} + for future in as_completed(futures): + repo = futures[future] + rows_by_rel[_relative(repo, root)] = future.result() + return [rows_by_rel[_relative(repo, root)] for repo in repos] + + cached_candidates: list[Path] = [] + for repo in repos: + if _relative(repo, root) in cached: + cached_candidates.append(repo) + else: + pending.append(repo) + if cached_candidates: + with ThreadPoolExecutor(max_workers=config.jobs) as pool: + futures = {pool.submit(_cached_resume_identity, repo, root, config): repo for repo in cached_candidates} + for future in as_completed(futures): + repo = futures[future] + rel = _relative(repo, root) + entry = cached[rel] + current_identity = future.result() + if current_identity is not None and entry.identity == current_identity: + rows_by_rel[rel] = entry.row + else: + pending.append(repo) + if pending: + with ThreadPoolExecutor(max_workers=config.jobs) as pool: + futures = {pool.submit(_safe_resume_entry, repo, root, config): repo for repo in pending} + for future in as_completed(futures): + repo = futures[future] + row, identity = future.result() + rel = _relative(repo, root) + rows_by_rel[rel] = row + _append_resume_row(resume_state, root, repo, row, identity) + return [rows_by_rel[_relative(repo, root)] for repo in repos] + + +def build_map( + root: Path, + config: Config, + tool_version: str, + *, + resume_state: Path | None = None, +) -> Map: + root = root.resolve() + resume_state = resume_state.resolve() if resume_state is not None else None + repo_paths = discover_repos(root, config) + rows = _map_rows(root, config, repo_paths, resume_state) + class_counts: dict[str, int] = {} + for row in rows: + class_counts[row.class_] = class_counts.get(row.class_, 0) + 1 + return Map( + schema_version=SCHEMA_VERSION, + tool_version=tool_version, + generated_at=datetime.now().astimezone().isoformat(timespec="seconds"), + root_sha256_prefix=_root_hash(root), + root=None if config.portable else str(root), + absolute_paths_included=not config.portable, + repo_count=len(rows), + dirty_count=sum(row.dirty_count for row in rows if row.metadata_status == "ok"), + class_counts=class_counts, + top_level=tuple(_top_level(root, config)), + repositories=tuple(rows), + annotations=dict(config.annotations), + ) + + +def write_map( + root: Path, + config: Config, + tool_version: str, + output: Path, + *, + resume_state: Path | None = None, +) -> Map: + data = build_map(root, config, tool_version, resume_state=resume_state) + output.write_text(json.dumps(data.to_json(), indent=2) + "\n", encoding="utf-8") + return data diff --git a/client-plugin/server/src/index_graph/symbols/__init__.py b/client-plugin/server/src/index_graph/symbols/__init__.py new file mode 100644 index 0000000..cfab7bd --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/__init__.py @@ -0,0 +1,24 @@ +"""Symbol-level intelligence: a re-checkable, deterministic call/reference graph. + +Extends the module-import graph down to functions, classes, and methods. Within +a module, calls resolve exactly via AST (high confidence). Across modules, an +imported name that binds to a real definition resolves best-effort (moderate); +anything the static scan cannot bind is surfaced honestly as unresolved, never +guessed. The graph seals with the same MATCH/DRIFT/UNVERIFIABLE certificate the +module graph uses. +""" +from __future__ import annotations + +from .model import (SymbolCall, SymbolCoverage, SymbolDefinition, SymbolGraph) +from .inheritance import InheritanceEdge, extract_inheritance_edges +from .navigate import (find_definitions, find_implementations, find_references) +from .build import (build_symbol_graph, build_symbol_navigator, + symbol_graph_to_claims, symbol_graph_to_payload) + +__all__ = [ + "SymbolCall", "SymbolCoverage", "SymbolDefinition", "SymbolGraph", + "InheritanceEdge", "extract_inheritance_edges", + "find_definitions", "find_implementations", "find_references", + "build_symbol_graph", "build_symbol_navigator", + "symbol_graph_to_claims", "symbol_graph_to_payload", +] diff --git a/client-plugin/server/src/index_graph/symbols/build.py b/client-plugin/server/src/index_graph/symbols/build.py new file mode 100644 index 0000000..e9bd5e3 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/build.py @@ -0,0 +1,111 @@ +"""Assemble the SymbolGraph: definitions + resolved/unresolved calls + coverage. + +Deterministic by construction: definitions and calls are sorted, fan-in/out are +derived from the sorted resolved calls, and the whole graph serializes to a +canonical payload whose SHA is byte-identical across runs. +""" +from __future__ import annotations + +from pathlib import Path + +from ..internals.modules import discover_modules +from .calls import extract_symbol_calls +from .definitions import extract_symbol_definitions +from .imports import module_import_bindings +from .inheritance import InheritanceEdge, extract_inheritance_edges +from .model import SymbolCall, SymbolCoverage, SymbolDefinition, SymbolGraph + + +def _python_ids(repo_root: Path) -> set[str]: + return {m.id for m in discover_modules(repo_root) if m.language == "python"} + + +def _fan(calls: tuple[SymbolCall, ...]) -> tuple[dict[str, int], dict[str, int]]: + fan_out: dict[str, int] = {} + fan_in: dict[str, int] = {} + seen_out: set[tuple[str, str]] = set() + seen_in: set[tuple[str, str]] = set() + for c in calls: + if c.to_symbol is None: + continue + if (c.from_symbol, c.to_symbol) not in seen_out: + seen_out.add((c.from_symbol, c.to_symbol)) + fan_out[c.from_symbol] = fan_out.get(c.from_symbol, 0) + 1 + if (c.to_symbol, c.from_symbol) not in seen_in: + seen_in.add((c.to_symbol, c.from_symbol)) + fan_in[c.to_symbol] = fan_in.get(c.to_symbol, 0) + 1 + return fan_in, fan_out + + +def build_symbol_graph(repo_root: Path, repo_name: str | None = None) -> SymbolGraph: + graph, _ = build_symbol_navigator(repo_root, repo_name) + return graph + + +def build_symbol_navigator( + repo_root: Path, repo_name: str | None = None, +) -> tuple[SymbolGraph, list[InheritanceEdge]]: + """Build the symbol graph and its resolved inheritance edges in one pass. + + The extraction (definitions, imports, calls) is shared, so navigation + (definition/references/implementations) rides on the exact same evidence the + graph seals. Inheritance edges power find-implementations and are never + guessed: an external or unbindable base yields no edge. + """ + root = repo_root.resolve() + name = repo_name or root.name + ids = _python_ids(root) + definitions, parse_errors = extract_symbol_definitions(root, ids) + import_tables = module_import_bindings(root, ids) + calls_list, dynamic = extract_symbol_calls(root, definitions, ids, import_tables) + edges = extract_inheritance_edges(root, definitions, ids, import_tables) + calls = tuple(calls_list) + resolved = sum(1 for c in calls if c.to_symbol is not None) + coverage = SymbolCoverage( + symbols=len(definitions), + resolved_calls=resolved, + unresolved_calls=len(calls) - resolved, + parse_errors=tuple(parse_errors), + dynamic_calls=tuple(dynamic), + ) + fan_in, fan_out = _fan(calls) + graph = SymbolGraph(repo=name, symbols=tuple(definitions), calls=calls, + coverage=coverage, fan_in=fan_in, fan_out=fan_out) + return graph, edges + + +def _definition_payload(d: SymbolDefinition) -> dict: + return {"id": d.id, "name": d.name, "kind": d.kind, "module_id": d.module_id, + "file": d.file, "line": d.line, "parent": d.parent, "is_public": d.is_public} + + +def _call_payload(c: SymbolCall) -> dict: + return {"from_symbol": c.from_symbol, "to_symbol": c.to_symbol, "to_name": c.to_name, + "kind": c.kind, "file": c.evidence_file, "line": c.evidence_line, + "raw": c.raw, "resolution": c.resolution, "confidence": c.confidence} + + +def symbol_graph_to_payload(g: SymbolGraph) -> dict: + """A canonical, JSON-serializable projection for hashing, CLI, and MCP.""" + return { + "repo": g.repo, + "symbols": [_definition_payload(d) for d in g.symbols], + "calls": [_call_payload(c) for c in g.calls], + "fan_in": dict(sorted(g.fan_in.items())), + "fan_out": dict(sorted(g.fan_out.items())), + "coverage": { + "complete": g.coverage.complete, + "symbols": g.coverage.symbols, + "resolved_calls": g.coverage.resolved_calls, + "unresolved_calls": g.coverage.unresolved_calls, + "parse_errors": list(g.coverage.parse_errors), + "dynamic_calls": [{"file": f, "line": ln} for f, ln in g.coverage.dynamic_calls], + }, + } + + +def symbol_graph_to_claims(g: SymbolGraph) -> set[tuple[str, str]]: + """The set of resolved (caller, callee) edges the wiki may claim and the + verifier re-derives. Only edges with a real resolved target are claimable; + an unresolved reference is never a claimable edge.""" + return {(c.from_symbol, c.to_symbol) for c in g.calls if c.to_symbol is not None} diff --git a/client-plugin/server/src/index_graph/symbols/calls.py b/client-plugin/server/src/index_graph/symbols/calls.py new file mode 100644 index 0000000..ccbb554 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/calls.py @@ -0,0 +1,249 @@ +"""Extract call sites and resolve them: exact within-module, best-effort cross-module. + +Resolution is layered and honest: + - ``foo()`` where ``foo`` is defined at module level in this file -> exact/high. + - ``self.m()`` inside a method of class C where C defines ``m`` -> exact/high. + - ``foo()`` where ``foo`` is a ``from import foo`` binding that + names a real definition in the target module -> cross_module/moderate. + - anything else (undefined name, attribute on a non-self object, a name that + imports but names no definition) -> cross_module_unresolved/low, never guessed. + - ``getattr(...)`` / a call whose func is neither a Name nor an Attribute + (a variable holding a function) -> flagged as a dynamic-dispatch coverage gap. +""" +from __future__ import annotations + +import ast +from collections.abc import Callable +from pathlib import Path + +from ..graph.walk import walk_files +from ..internals.modules import _strip_suffix +from .model import SymbolCall, SymbolDefinition + +_FUNC = (ast.FunctionDef, ast.AsyncFunctionDef) +_DEF = (*_FUNC, ast.ClassDef) + + +def _line_text(lines: list[str], lineno: int) -> str: + return lines[lineno - 1].strip() if 1 <= lineno <= len(lines) else "" + + +def _is_getattr(func: ast.AST) -> bool: + return isinstance(func, ast.Name) and func.id == "getattr" + + +def _scope_locals(func: ast.AST) -> frozenset: + """Names bound in a function's own scope: parameters plus every name stored + (assignment, for-target, with-alias, walrus, except-alias, nested def/class + name) within its body, NOT descending into nested function/lambda scopes + (those open their own scope). A call on any of these is a local value, so it + must not resolve to a module-level def of the same name.""" + names: set[str] = set() + args = getattr(func, "args", None) + if isinstance(args, ast.arguments): + for a in (*getattr(args, "posonlyargs", []), *args.args, *args.kwonlyargs): + names.add(a.arg) + if args.vararg: + names.add(args.vararg.arg) + if args.kwarg: + names.add(args.kwarg.arg) + + def walk(node: ast.AST) -> None: + for child in ast.iter_child_nodes(node): + if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda)): + # a nested def/lambda opens its own scope, but its NAME is a + # local binding in this scope + if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)): + names.add(child.name) + continue + if isinstance(child, ast.ClassDef): + names.add(child.name) + continue + if isinstance(child, ast.Name) and isinstance(child.ctx, ast.Store): + names.add(child.id) + walk(child) + + for stmt in getattr(func, "body", []): + walk(stmt) + return frozenset(names) + + +class _Walker: + """Walk one module's AST, emitting a SymbolCall per resolvable/unresolvable + call site and recording dynamic-dispatch sites as coverage gaps.""" + + def __init__(self, module_id: str, rel: str, lines: list[str], + local_defs: dict[str, str], class_methods: dict[str, dict[str, str]], + imports: dict[str, tuple[str, str]], + cross_target: Callable[[tuple[str, str]], str | None]): + self.module_id = module_id + self.rel = rel + self.lines = lines + self.local_defs = local_defs # bare name -> module-level symbol id + self.class_methods = class_methods # class id -> {method name -> method id} + self.imports = imports # bare name -> (target_module_id, name) + self.cross_target = cross_target # (module_id, name) -> resolved symbol id | None + self.calls: list[SymbolCall] = [] + self.dynamic: list[tuple[str, int]] = [] + + def visit_body(self, body, enclosing: str | None, parent_id: str | None, + class_id: str | None, scope: frozenset = frozenset()) -> None: + """`enclosing` is the caller symbol calls are attributed to; `parent_id` + is the lexical id used to build child symbol ids (matches definitions.py); + `class_id` is the nearest enclosing class, for `self.m()` resolution; + `scope` is the set of names bound in enclosing function scopes, so a call + on a locally-bound name that shadows a module def is not mis-resolved.""" + for node in body: + self._visit(node, enclosing, parent_id, class_id, scope) + + def _visit(self, node: ast.AST, enclosing: str | None, parent_id: str | None, + class_id: str | None, scope: frozenset = frozenset()) -> None: + if isinstance(node, ast.ClassDef): + base = parent_id or self.module_id + cid = f"{base}::{node.name}" + # class-level statements have no caller symbol; class_id becomes cid + self.visit_body(node.body, enclosing=None, parent_id=cid, class_id=cid, + scope=scope) + return + if isinstance(node, _FUNC): + base = parent_id or self.module_id + sym_id = f"{base}::{node.name}" + # a name bound in THIS function (parameter, assignment, for-target, + # with-alias, ...) shadows a module-level def of the same name, so a + # bare call on it is a local value, not the module symbol. Union with + # enclosing scopes so a closure over an outer local is honored too. + inner = scope | _scope_locals(node) + # class_id persists into the function body so `self.m()` resolves + # against the enclosing class (also for nested closures in a method). + self.visit_body(node.body, enclosing=sym_id, parent_id=sym_id, + class_id=class_id, scope=inner) + return + # a plain statement: collect its calls, but hand any nested def/class + # back to _visit so their bodies are attributed to the right symbol. + self._scan(node, enclosing, parent_id, class_id, scope) + + def _scan(self, node: ast.AST, enclosing: str | None, parent_id: str | None, + class_id: str | None, scope: frozenset = frozenset()) -> None: + """Recurse a statement's expression tree, emitting calls attributed to + `enclosing`, but re-dispatch nested defs/classes to `_visit` (they open a + new caller scope) so their inner calls are not mis-attributed.""" + for child in ast.iter_child_nodes(node): + if isinstance(child, _DEF): + self._visit(child, enclosing, parent_id, class_id, scope) + continue + if isinstance(child, ast.Call): + self._call(child, enclosing, class_id, scope) + self._scan(child, enclosing, parent_id, class_id, scope) + + def _call(self, node: ast.Call, enclosing: str | None, class_id: str | None, + scope: frozenset = frozenset()) -> None: + if enclosing is None: + return # module/class-level calls have no caller symbol to attribute + func = node.func + if _is_getattr(func): + self.dynamic.append((self.rel, node.lineno)) + return + if isinstance(func, ast.Name): + self._resolve_name(func.id, node.lineno, enclosing, scope) + elif isinstance(func, ast.Attribute): + self._resolve_attr(func, node.lineno, enclosing, class_id) + else: + # a call on a subscript/call result: a variable function, dynamic + self.dynamic.append((self.rel, node.lineno)) + + def _resolve_name(self, name: str, lineno: int, enclosing: str, + scope: frozenset = frozenset()) -> None: + raw = _line_text(self.lines, lineno) + if name in scope: + # a locally-bound name shadows any module def or import: the callee + # is a runtime value, statically unknown, never an exact edge + self._emit(enclosing, None, name, lineno, raw, + "local_binding_unresolved", "low") + return + local = self.local_defs.get(name) + if local is not None: + self._emit(enclosing, local, name, lineno, raw, "exact", "high") + return + binding = self.imports.get(name) + if binding is not None: + resolved = self.cross_target(binding) + if resolved is not None: + self._emit(enclosing, resolved, name, lineno, raw, "cross_module", "moderate") + return + self._emit(enclosing, None, name, lineno, raw, "cross_module_unresolved", "low") + + def _resolve_attr(self, func: ast.Attribute, lineno: int, enclosing: str, + class_id: str | None) -> None: + raw = _line_text(self.lines, lineno) + obj = func.value + # self.method() inside a method of a known class -> exact sibling lookup + if isinstance(obj, ast.Name) and obj.id == "self" and class_id is not None: + methods = self.class_methods.get(class_id, {}) + target = methods.get(func.attr) + if target is not None: + self._emit(enclosing, target, func.attr, lineno, raw, "exact", "high") + return + # any other attribute call: object type is not statically known + self._emit(enclosing, None, func.attr, lineno, raw, "cross_module_unresolved", "low") + + def _emit(self, frm: str, to: str | None, to_name: str, lineno: int, + raw: str, resolution: str, confidence: str) -> None: + self.calls.append(SymbolCall( + from_symbol=frm, to_symbol=to, to_name=to_name, kind="call", + evidence_file=self.rel, evidence_line=lineno, raw=raw, + resolution=resolution, confidence=confidence)) + + +def _module_maps(definitions: list[SymbolDefinition], module_id: str + ) -> tuple[dict[str, str], dict[str, dict[str, str]]]: + """(module-level name -> id, class id -> {method name -> id}) for one module.""" + local_defs: dict[str, str] = {} + class_methods: dict[str, dict[str, str]] = {} + # Hoisted out of the loop: computing this set per definition made the whole + # build quadratic in total symbol count on large repos. + class_ids = {d.id for d in definitions + if d.module_id == module_id and d.kind == "class"} + for d in definitions: + if d.module_id != module_id: + continue + if d.parent is None: + local_defs[d.name] = d.id + elif d.parent in class_ids: + class_methods.setdefault(d.parent, {})[d.name] = d.id + return local_defs, class_methods + + +def extract_symbol_calls( + repo_root: Path, definitions: list[SymbolDefinition], ids: set[str], + import_tables: dict[str, dict[str, tuple[str, str]]], +) -> tuple[list[SymbolCall], list[tuple[str, int]]]: + """Return (calls sorted deterministically, dynamic-dispatch sites).""" + by_def_id = {d.id for d in definitions} + + def cross_target(binding: tuple[str, str]) -> str | None: + target_mod, name = binding + cand = f"{target_mod}::{name}" + return cand if cand in by_def_id else None + + calls: list[SymbolCall] = [] + dynamic: list[tuple[str, int]] = [] + for py in walk_files(repo_root, suffixes=(".py",)): + rel = py.relative_to(repo_root).as_posix() + module_id = _strip_suffix(rel) + if module_id not in ids: + continue + try: + text = py.read_text(encoding="utf-8-sig") + tree = ast.parse(text) + except (OSError, SyntaxError, ValueError): + continue + local_defs, class_methods = _module_maps(definitions, module_id) + walker = _Walker(module_id, rel, text.splitlines(), local_defs, + class_methods, import_tables.get(module_id, {}), cross_target) + walker.visit_body(tree.body, enclosing=None, parent_id=None, class_id=None) + calls += walker.calls + dynamic += walker.dynamic + calls.sort(key=lambda c: (c.from_symbol, c.to_symbol or "", c.to_name, + c.evidence_file, c.evidence_line)) + dynamic = sorted(set(dynamic)) + return calls, dynamic diff --git a/client-plugin/server/src/index_graph/symbols/definitions.py b/client-plugin/server/src/index_graph/symbols/definitions.py new file mode 100644 index 0000000..0ab4ee9 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/definitions.py @@ -0,0 +1,95 @@ +"""Extract function/class/method definitions from Python source via AST. + +Definitions are the ground truth a call site resolves against. Only Python is +AST-exact here; other languages carry no symbol-level extraction (their +module-level graph is unchanged). Every definition carries file:line so the +wiki and the verifier can point at the exact evidence. +""" +from __future__ import annotations + +import ast +from pathlib import Path + +from ..graph.walk import walk_files +from ..internals.modules import _strip_suffix +from .model import SymbolDefinition + +_FUNC = (ast.FunctionDef, ast.AsyncFunctionDef) + + +def _kind(node: ast.AST, in_class: bool) -> str: + if isinstance(node, ast.ClassDef): + return "class" + is_async = isinstance(node, ast.AsyncFunctionDef) + if in_class: + return "async_method" if is_async else "method" + return "async_function" if is_async else "function" + + +def _collect(node: ast.AST, module_id: str, rel: str, parent_id: str | None, + in_class: bool, out: list[SymbolDefinition]) -> None: + """Recurse the AST body, emitting a SymbolDefinition per def/class. + + Nested functions and classes are flattened under their lexical parent id so + a `Class::method` and a `func::inner` are both addressable and unique. + """ + body = getattr(node, "body", []) + for child in body: + if isinstance(child, (*_FUNC, ast.ClassDef)): + name = child.name + sym_id = f"{parent_id}::{name}" if parent_id else f"{module_id}::{name}" + out.append(SymbolDefinition( + id=sym_id, name=name, kind=_kind(child, in_class), + module_id=module_id, file=rel, line=child.lineno, + parent=parent_id, is_public=not name.startswith("_"))) + _collect(child, module_id, rel, sym_id, + in_class=isinstance(child, ast.ClassDef), out=out) + + +def extract_symbol_definitions( + repo_root: Path, ids: set[str] | None = None, +) -> tuple[list[SymbolDefinition], list[str]]: + """Return (definitions sorted by id, parse-error file list). + + `ids` restricts extraction to known Python module ids when supplied; when + None, every .py file is scanned. + """ + definitions: list[SymbolDefinition] = [] + parse_errors: list[str] = [] + for py in walk_files(repo_root, suffixes=(".py",)): + rel = py.relative_to(repo_root).as_posix() + module_id = _strip_suffix(rel) + if ids is not None and module_id not in ids: + continue + try: + tree = ast.parse(py.read_text(encoding="utf-8-sig")) + except OSError: + # an unreadable file (locked / permission-denied / removed + # mid-scan) is a coverage gap, not a file to drop silently: the + # map must not claim it saw everything while swallowing a file + parse_errors.append(rel) + continue + except (SyntaxError, ValueError): + parse_errors.append(rel) + continue + _collect(tree, module_id, rel, None, in_class=False, out=definitions) + definitions = _dedupe_last(definitions) + definitions.sort(key=lambda d: d.id) + parse_errors.sort() + return definitions, parse_errors + + +def _dedupe_last(definitions: list[SymbolDefinition]) -> list[SymbolDefinition]: + """Collapse same-id definitions to one, keeping the last (source order). + + A legal Python redefinition (two ``def foo`` at module scope, or a + decorator that rebinds a name) would otherwise emit two SymbolDefinitions + with an identical id, which silently collide in the wiki page map and make + resolution ambiguous. Python's runtime keeps the last binding, so the static + graph keeps the last definition site too. ``_collect`` appends in source + order, so the later dict write wins. + """ + by_id: dict[str, SymbolDefinition] = {} + for d in definitions: + by_id[d.id] = d + return list(by_id.values()) diff --git a/client-plugin/server/src/index_graph/symbols/imports.py b/client-plugin/server/src/index_graph/symbols/imports.py new file mode 100644 index 0000000..bc4cdb3 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/imports.py @@ -0,0 +1,72 @@ +"""Per-module import binding table: which local name maps to which module symbol. + +Only `from import name` bindings are recorded, because only +those can be resolved to a definition inside this repo. Plain `import pkg.mod` +and external imports are not bindable to a symbol and are left out (a call +through them stays honestly unresolved). +""" +from __future__ import annotations + +import ast +from pathlib import Path + +from ..graph.walk import walk_files +from ..internals.modules import (_dotted_to_id, _id_for, _pkg_depth, + _resolve_relative, _strip_suffix) + + +def _relative_target(from_id: str, level: int, module: str | None, + ids: set[str], depth: int) -> str | None: + if module is None: + return None + return _resolve_relative(from_id, level, module, ids, depth) + + +def _bindings_for_module(from_id: str, tree: ast.AST, ids: set[str], + depth: int) -> dict[str, tuple[str, str]]: + """Map a locally-bound name -> (target_module_id, imported_name).""" + bindings: dict[str, tuple[str, str]] = {} + for node in ast.walk(tree): + if not isinstance(node, ast.ImportFrom): + continue + if node.level and node.level > 0: + # relative: from .pkg import name / from . import name + if node.module is None: + pkg_parts = from_id.split("/")[:-1] + if node.level > depth: + continue + base = pkg_parts[:len(pkg_parts) - (node.level - 1)] + for a in node.names: + tid = _id_for("/".join([*base, a.name]), ids) + if tid: + bindings[a.asname or a.name] = (tid, a.name) + continue + target_mod = _relative_target(from_id, node.level, node.module, ids, depth) + elif node.module: + target_mod = _dotted_to_id(node.module, ids) + else: + target_mod = None + if not target_mod: + continue + for a in node.names: + if a.name == "*": + continue + bindings[a.asname or a.name] = (target_mod, a.name) + return bindings + + +def module_import_bindings(repo_root: Path, ids: set[str]) -> dict[str, dict[str, tuple[str, str]]]: + """For each Python module id, its {local_name: (target_module_id, name)} table.""" + tables: dict[str, dict[str, tuple[str, str]]] = {} + for py in walk_files(repo_root, suffixes=(".py",)): + rel = py.relative_to(repo_root).as_posix() + from_id = _strip_suffix(rel) + if from_id not in ids: + continue + try: + tree = ast.parse(py.read_text(encoding="utf-8-sig")) + except (OSError, SyntaxError, ValueError): + continue + depth = _pkg_depth(repo_root, from_id) + tables[from_id] = _bindings_for_module(from_id, tree, ids, depth) + return tables diff --git a/client-plugin/server/src/index_graph/symbols/inheritance.py b/client-plugin/server/src/index_graph/symbols/inheritance.py new file mode 100644 index 0000000..efadea5 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/inheritance.py @@ -0,0 +1,183 @@ +"""Class inheritance and method-override edges, resolved AST-exact in-repo. + +Powers find-implementations at symbol granularity. Two edge kinds, both carrying +file:line evidence and an honest resolution label: + + - a ``subclass`` edge: class C in this repo lists base B, where B binds (via a + ``from import B`` or a same-module ``class B``) to a real class + definition in this repo. A base that names an external class, or a name the + static scan cannot bind, is never guessed into an edge. + - an ``override`` edge: subclass C defines a method whose name also exists on + a resolved ancestor class. The edge points from the override site to the + ancestor's method, so find-implementations of ``Base::method`` returns every + in-repo override with its exact definition line. + +Only Python is AST-exact here; other languages carry no inheritance extraction, +matching the definitions/calls layers. +""" +from __future__ import annotations + +import ast +from dataclasses import dataclass +from pathlib import Path + +from ..graph.walk import walk_files +from ..internals.modules import _strip_suffix +from .model import SymbolDefinition + + +@dataclass(frozen=True) +class InheritanceEdge: + """A resolved class -> base edge, or a subclass-method -> base-method override.""" + kind: str # "subclass" | "override" + child: str # subclass id (subclass edge) or override method id (override edge) + parent: str # resolved base-class id, or ancestor method id + name: str # bare base-class name / method name as written + file: str # repo-relative path of the child site + line: int # 1-indexed line of the child (subclass def / override method def) + resolution: str # "exact" (same module) | "cross_module" (import-bound) + + +def _base_name(base: ast.expr) -> str | None: + """The bare name of a base-class expression, or None for a non-name base. + + ``class C(Base)`` -> "Base"; ``class C(pkg.Base)`` -> "Base" (attribute tail); + a subscript/call base (e.g. ``Generic[T]``) is not a bindable class name. + """ + if isinstance(base, ast.Name): + return base.id + if isinstance(base, ast.Attribute): + return base.attr + return None + + +def _class_bases(tree: ast.AST) -> dict[int, list[str]]: + """Map each class-def line to the list of bare base names it declares.""" + bases: dict[int, list[str]] = {} + for node in ast.walk(tree): + if isinstance(node, ast.ClassDef): + names = [n for b in node.bases if (n := _base_name(b)) is not None] + if names: + bases[node.lineno] = names + return bases + + +def _resolve_base( + base_name: str, + module_id: str, + class_by_name_in_module: dict[str, dict[str, str]], + imports: dict[str, tuple[str, str]], +) -> tuple[str, str] | None: + """Resolve a bare base name to (class_id, resolution), or None if unbindable. + + Same-module class wins first (exact); otherwise an import binding that names + a real class in the target module resolves cross_module. A base that binds to + no in-repo class definition returns None and is never guessed into an edge. + """ + same = class_by_name_in_module.get(module_id, {}).get(base_name) + if same is not None: + return same, "exact" + binding = imports.get(base_name) + if binding is not None: + target_mod, imported = binding + cand = class_by_name_in_module.get(target_mod, {}).get(imported) + if cand is not None: + return cand, "cross_module" + return None + + +def _method_index( + definitions: list[SymbolDefinition], +) -> tuple[dict[str, dict[str, str]], dict[str, dict[str, str]]]: + """(class name -> id per module, class id -> {method name -> method id}).""" + class_by_name: dict[str, dict[str, str]] = {} + methods_of: dict[str, dict[str, str]] = {} + for d in definitions: + if d.kind == "class": + class_by_name.setdefault(d.module_id, {})[d.name] = d.id + class_ids = {i for m in class_by_name.values() for i in m.values()} + for d in definitions: + if d.parent in class_ids and d.kind in ("method", "async_method"): + methods_of.setdefault(d.parent, {})[d.name] = d.id + return class_by_name, methods_of + + +def extract_inheritance_edges( + repo_root: Path, + definitions: list[SymbolDefinition], + ids: set[str], + import_tables: dict[str, dict[str, tuple[str, str]]], +) -> list[InheritanceEdge]: + """Return resolved subclass + override edges, sorted deterministically. + + Overrides are computed against the direct resolved parent only; a diamond or + multi-level chain still yields one override edge per resolved parent that + declares the same method name, which is the honest, re-checkable claim. + """ + class_by_name, methods_of = _method_index(definitions) + class_line_to_id = {(d.file, d.line): d.id + for d in definitions if d.kind == "class"} + site_of = {d.id: (d.file, d.line) for d in definitions} + edges: list[InheritanceEdge] = [] + for py in walk_files(repo_root, suffixes=(".py",)): + rel = py.relative_to(repo_root).as_posix() + module_id = _strip_suffix(rel) + if module_id not in ids: + continue + try: + tree = ast.parse(py.read_text(encoding="utf-8-sig")) + except (OSError, SyntaxError, ValueError): + continue + imports = import_tables.get(module_id, {}) + for line, base_names in _class_bases(tree).items(): + child_id = class_line_to_id.get((rel, line)) + if child_id is None: + continue + for base_name in base_names: + resolved = _resolve_base(base_name, module_id, class_by_name, imports) + if resolved is None: + continue # external or unbindable base: never guessed + parent_id, resolution = resolved + edges.append(InheritanceEdge( + kind="subclass", child=child_id, parent=parent_id, + name=base_name, file=rel, line=line, resolution=resolution)) + edges.extend(_override_edges( + child_id, parent_id, resolution, methods_of, site_of)) + edges.sort(key=lambda e: (e.kind, e.child, e.parent, e.name, e.file, e.line)) + return _dedupe(edges) + + +def _override_edges( + child_id: str, parent_id: str, resolution: str, + methods_of: dict[str, dict[str, str]], + site_of: dict[str, tuple[str, int]], +) -> list[InheritanceEdge]: + """Override edges for every child method whose name the parent also defines. + + The edge's file:line is the override method's own definition site, taken + from the definition table so every hop stays evidence-backed. + """ + child_methods = methods_of.get(child_id, {}) + parent_methods = methods_of.get(parent_id, {}) + out: list[InheritanceEdge] = [] + for mname, mid in sorted(child_methods.items()): + parent_mid = parent_methods.get(mname) + if parent_mid is None: + continue + file, line = site_of.get(mid, ("", 0)) + out.append(InheritanceEdge( + kind="override", child=mid, parent=parent_mid, name=mname, + file=file, line=line, resolution=resolution)) + return out + + +def _dedupe(edges: list[InheritanceEdge]) -> list[InheritanceEdge]: + seen: set[tuple] = set() + out: list[InheritanceEdge] = [] + for e in edges: + key = (e.kind, e.child, e.parent, e.name) + if key in seen: + continue + seen.add(key) + out.append(e) + return out diff --git a/client-plugin/server/src/index_graph/symbols/model.py b/client-plugin/server/src/index_graph/symbols/model.py new file mode 100644 index 0000000..f02967f --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/model.py @@ -0,0 +1,62 @@ +"""Frozen data model for the symbol-level call/reference graph. + +A SymbolDefinition is a function/class/method the AST saw defined. A SymbolCall +is a call site, carrying file:line evidence and an honest resolution label: a +resolved edge names a real definition; an unresolved reference names only the +bare name it could not statically bind, and is never guessed into an edge. +""" +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class SymbolDefinition: + """A function, class, or method the AST recorded as defined.""" + id: str # "module_id::name" or "module_id::Class::method" + name: str # bare name (unqualified) + kind: str # "function" | "async_function" | "class" | "method" | "async_method" + module_id: str # parent module id (matches InternalGraph module ids) + file: str # repo-relative path + line: int # 1-indexed definition line + parent: str | None # "module_id::Class" for a method; None for a top-level symbol + is_public: bool # name does not start with "_" + + +@dataclass(frozen=True) +class SymbolCall: + """A call or reference site, with evidence and an honest resolution label.""" + from_symbol: str # caller symbol id + to_symbol: str | None # resolved target id, or None when unresolved + to_name: str # bare name as written at the call site + kind: str # "call" + evidence_file: str # repo-relative path + evidence_line: int # 1-indexed line of the call site + raw: str # source-line fragment + resolution: str # "exact" | "cross_module" | "cross_module_unresolved" + confidence: str # "high" | "moderate" | "low" + + +@dataclass(frozen=True) +class SymbolCoverage: + """What the symbol scan could and could not statically see.""" + symbols: int + resolved_calls: int + unresolved_calls: int + parse_errors: tuple[str, ...] + dynamic_calls: tuple[tuple[str, int], ...] # (file, line) of getattr/variable dispatch + + @property + def complete(self) -> bool: + return not self.parse_errors and not self.dynamic_calls + + +@dataclass(frozen=True) +class SymbolGraph: + """Per-repository symbol graph, analogous to InternalGraph.""" + repo: str + symbols: tuple[SymbolDefinition, ...] + calls: tuple[SymbolCall, ...] + coverage: SymbolCoverage + fan_in: dict[str, int] # resolved-target id -> number of distinct callers + fan_out: dict[str, int] # caller id -> number of distinct resolved targets diff --git a/client-plugin/server/src/index_graph/symbols/navigate.py b/client-plugin/server/src/index_graph/symbols/navigate.py new file mode 100644 index 0000000..c8ce008 --- /dev/null +++ b/client-plugin/server/src/index_graph/symbols/navigate.py @@ -0,0 +1,89 @@ +"""Symbol-graph navigation: go-to-definition, find-references, find-implementations. + +A pure query layer over the wave-1 ``SymbolGraph`` plus resolved inheritance +edges. Every result carries file:line evidence, and every verdict defers to the +graph's own resolution so nothing is guessed: + + - definitions: symbols whose id or bare name matches the query. + - references: resolved callers of the matched symbols, plus a separate, + honestly-labeled list of unresolved same-name references (never callers). + - implementations: for a class query, its resolved subclasses; for a method + query, the in-repo overrides of that method. A bare method name resolves + against every matching definition so ``Base.method`` and ``method`` both work. + +The query is a symbol id (``module::name`` or ``Class::method``) or a bare name. +""" +from __future__ import annotations + +from .inheritance import InheritanceEdge +from .model import SymbolDefinition, SymbolGraph + + +def _matches(sym: SymbolDefinition, query: str) -> bool: + """A symbol matches a query by exact id, bare name, or a ``Class::method`` tail.""" + if sym.id == query or sym.name == query: + return True + # A `Class::method` or `module::name` tail lets a partial id resolve. + return sym.id.endswith("::" + query) + + +def find_definitions(graph: SymbolGraph, query: str) -> list[dict]: + """Go-to-definition: every matching symbol with its file:line site.""" + return [ + {"id": s.id, "name": s.name, "kind": s.kind, "module_id": s.module_id, + "file": s.file, "line": s.line, "is_public": s.is_public} + for s in graph.symbols if _matches(s, query) + ] + + +def find_references(graph: SymbolGraph, query: str) -> dict: + """Find-references: resolved callers (evidence-backed) + unresolved refs. + + A resolved caller names a real call edge whose target is a matched symbol. + An unresolved reference shares the bare name but bound to no edge; it is + reported separately and never counted as a caller. + """ + targets = {s.id for s in graph.symbols if _matches(s, query)} + names = {s.name for s in graph.symbols if _matches(s, query)} + references = [ + {"from_symbol": c.from_symbol, "to_symbol": c.to_symbol, + "to_name": c.to_name, "file": c.evidence_file, "line": c.evidence_line, + "raw": c.raw, "resolution": c.resolution, "confidence": c.confidence} + for c in graph.calls if c.to_symbol in targets + ] + references.sort(key=lambda r: (r["file"], r["line"], r["from_symbol"])) + unresolved = [ + {"from_symbol": c.from_symbol, "to_name": c.to_name, + "file": c.evidence_file, "line": c.evidence_line, "raw": c.raw} + for c in graph.calls if c.to_symbol is None and c.to_name in names + ] + unresolved.sort(key=lambda r: (r["file"], r["line"], r["from_symbol"])) + return {"references": references, "unresolved": unresolved} + + +def find_implementations( + graph: SymbolGraph, edges: list[InheritanceEdge], query: str, +) -> dict: + """Find-implementations: subclasses of a class, or overrides of a method. + + Resolution is by the matched definition's kind. A class query returns the + ``subclass`` edges whose parent is a matched class. A method query returns + the ``override`` edges whose parent is a matched method. When a query + matches both (rare bare-name collisions), both lists are populated honestly. + """ + class_targets = {s.id for s in graph.symbols + if _matches(s, query) and s.kind == "class"} + method_targets = {s.id for s in graph.symbols + if _matches(s, query) and s.kind in ("method", "async_method")} + subclasses = [_edge_row(e) for e in edges + if e.kind == "subclass" and e.parent in class_targets] + overrides = [_edge_row(e) for e in edges + if e.kind == "override" and e.parent in method_targets] + subclasses.sort(key=lambda r: (r["file"], r["line"], r["child"])) + overrides.sort(key=lambda r: (r["file"], r["line"], r["child"])) + return {"subclasses": subclasses, "overrides": overrides} + + +def _edge_row(e: InheritanceEdge) -> dict: + return {"child": e.child, "parent": e.parent, "name": e.name, + "file": e.file, "line": e.line, "resolution": e.resolution} diff --git a/client-plugin/server/src/index_graph/verify.py b/client-plugin/server/src/index_graph/verify.py new file mode 100644 index 0000000..af95885 --- /dev/null +++ b/client-plugin/server/src/index_graph/verify.py @@ -0,0 +1,87 @@ +"""Ground a structural claim against the verified graph: MATCH / REFUTED / UNVERIFIABLE. + +The deterministic anti-hallucination oracle. An agent asserts a dependency or an +existence claim about the code; index confirms or refutes it from the real graph with +file:line evidence, instead of the agent trusting its own memory. This artifact has its +own honest triad (a false claim is REFUTED, not DRIFT) and is re-checkable: a consumer +re-runs `recheck` and recomputes the hash rather than trusting the verdict. The content +hash covers the whole graph (graph-scoped provenance, matching the certificate +convention), so it binds the verdict to the exact graph state it was computed against. +""" +from __future__ import annotations + +from .certify import canonical_sha + + +def _evidence(rel: dict) -> str | None: + sigs = rel.get("signals") or [] + if not sigs: + return None + s = sigs[0] + f = s.get("file") + if not f: + return None + line = s.get("line") + return f"{f}:{line}" if line is not None else f + + +def _names(pack: dict) -> set[str]: + names = set(pack.get("roles", {}).keys()) + for r in pack.get("relations", []): + if r.get("from"): + names.add(r["from"]) + if r.get("to"): + names.add(r["to"]) + return names + + +def verify_claim(pack: dict, claim: dict) -> dict: + """Return {verdict, evidence, detail} for a claim against the pack.""" + kind = claim.get("kind") + names = _names(pack) + + if kind == "exists": + name = claim.get("name") + if name in names: + return {"verdict": "MATCH", "evidence": None, + "detail": f"{name} is a repo in the workspace"} + return {"verdict": "REFUTED", "evidence": None, + "detail": f"{name} is not a repo in the workspace"} + + if kind == "depends": + frm, to = claim.get("from"), claim.get("to") + if frm not in names or to not in names: + missing = frm if frm not in names else to + return {"verdict": "UNVERIFIABLE", "evidence": None, + "detail": f"{missing} is not a repo in the workspace"} + matches = [r for r in pack.get("relations", []) + if not r.get("external") and r.get("from") == frm and r.get("to") == to] + if matches: + # return the strongest edge's evidence (high > moderate > low), file as a + # deterministic tiebreak, so the oracle hands back its best witness, not the + # alphabetically-first one. + rank = {"high": 0, "moderate": 1, "low": 2} + best = min(matches, key=lambda r: (rank.get(r.get("confidence"), 3), + (r.get("signals") or [{}])[0].get("file", ""))) + detail = f"{frm} depends on {to}" + if len(matches) > 1: + detail += f" ({len(matches)} edges agree)" + return {"verdict": "MATCH", "evidence": _evidence(best), "detail": detail} + return {"verdict": "REFUTED", "evidence": None, + "detail": f"no dependency {frm} -> {to} in the graph"} + + return {"verdict": "UNVERIFIABLE", "evidence": None, "detail": "unknown claim kind"} + + +def build_verification(pack: dict, claim: dict, *, tool_version: str, recheck: str) -> dict: + result = verify_claim(pack, claim) + return { + "schema": "index.verification/1", + "tool_version": tool_version, + "claim": claim, + "verdict": result["verdict"], + "evidence": result["evidence"], + "detail": result["detail"], + "content_sha256": canonical_sha(pack), + "recheck": recheck, + } diff --git a/client-plugin/server/src/index_graph/viz/__init__.py b/client-plugin/server/src/index_graph/viz/__init__.py new file mode 100644 index 0000000..113fc30 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/__init__.py @@ -0,0 +1,16 @@ +"""Zero-dependency renderers for the dependency-graph context pack.""" +from .layout import build_layout +from .svg import render_svg +from .mermaid import render_mermaid +from .charts import render_charts +from .html import render_html +from .manifest import render_manifest +from .atlas_layout import build_atlas_layout +from .atlas_svg import render_atlas_svg +from .atlas_html import render_atlas_html + +__all__ = [ + "build_layout", "render_svg", "render_mermaid", + "render_charts", "render_html", "render_manifest", + "build_atlas_layout", "render_atlas_svg", "render_atlas_html", +] diff --git a/client-plugin/server/src/index_graph/viz/atlas_assets.py b/client-plugin/server/src/index_graph/viz/atlas_assets.py new file mode 100644 index 0000000..626f10d --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/atlas_assets.py @@ -0,0 +1,127 @@ +"""Atlas dashboard CSS + JS (string constants embedded into the self-contained HTML).""" +from __future__ import annotations + +ATLAS_CSS = """ +:root{--bg:#f4f3ef;--panel:rgba(255,255,255,.55);--ink:#0b0c0e;--soft:#2f3238;--muted:#585c64;--hairline:rgba(11,12,14,.14);--accent:#4636e8;--gold:#0b0c0e;--font-body:Arial,Helvetica,sans-serif;--font-mono:ui-monospace,SFMono-Regular,Consolas,monospace} +*{box-sizing:border-box}body{margin:0;background:var(--bg);color:var(--ink);font-family:var(--font-body)} +main{display:grid;grid-template-columns:1fr 360px;min-height:100vh} +.skip-link{position:absolute;left:1rem;top:1rem;transform:translateY(-140%);background:var(--bg);border:1px solid var(--hairline);padding:.55rem .85rem;z-index:10} +.skip-link:focus{transform:none;outline:2px solid var(--accent);outline-offset:3px} +#stage{overflow:hidden;padding:1rem;position:relative}#stage svg{max-width:100%;height:auto;cursor:grab;background:var(--bg)} +#stage.grabbing svg{cursor:grabbing} +aside{border-left:1px solid var(--hairline);padding:1rem;font-family:var(--font-mono);font-size:.82rem;overflow:auto;background:var(--panel)} +.controls{display:flex;flex-wrap:wrap;gap:.4rem;margin-bottom:.6rem;align-items:center} +.chip{cursor:pointer;border:1px solid var(--hairline);border-radius:6px;padding:.2em .5em;background:transparent;color:var(--ink)} +.chip[aria-pressed=true]{background:var(--accent);color:var(--bg)} +input[type=search]{flex:1;min-width:8rem;padding:.4em;background:transparent;color:var(--ink);border:1px solid var(--hairline);border-radius:6px} +#trail{font-family:var(--font-mono);font-size:.72rem;opacity:.8;margin:.2rem 0 .6rem;min-height:1.2em} +#trail a{color:var(--gold);cursor:pointer;text-decoration:underline} +#detail h3{margin:.2rem 0;color:var(--gold)}#detail h4{margin:.6rem 0 .2rem;color:var(--gold)} +#detail .md{font-family:var(--font-body);font-size:.95rem;line-height:1.5;border-top:1px solid var(--hairline);margin-top:.6rem;padding-top:.6rem} +#detail .md pre{background:rgba(255,255,255,.55);border:1px solid var(--hairline);padding:.5em;overflow:auto}#detail .md table{border-collapse:collapse} +#detail .md th,#detail .md td{border:1px solid var(--hairline);padding:.2em .5em} +#detail .md .wikilink{color:var(--accent);cursor:pointer}#detail .md .md-img{opacity:.6;font-style:italic} +a.wikilink{color:var(--accent)} +@media(max-width:820px){main{grid-template-columns:1fr}aside{border-left:none}} +""" + +ATLAS_SKIP_LINK = '' + +ATLAS_JS = r""" +const $=s=>document.querySelector(s),$$=s=>[...document.querySelectorAll(s)]; +const esc=s=>String(s).replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); +const norm=s=>String(s).trim().toLowerCase().replace(/_/g,'-').replace(/ /g,'-'); +const repos={};(DATA.repos||[]).forEach(r=>repos[r.name]=r); +const docs={};(DATA.docs||[]).forEach(d=>docs[d.id]=d); +const tgt={}; // normalized name -> {kind,id} +(DATA.repos||[]).forEach(r=>{if(!(norm(r.name)in tgt))tgt[norm(r.name)]={kind:'repo',id:r.name};}); +(DATA.docs||[]).forEach(d=>{[d.title,d.id.split('/').pop().replace(/\.[^.]+$/,'')].forEach(c=>{if(!(norm(c)in tgt))tgt[norm(c)]={kind:'doc',id:d.id};});}); +function kedgesFrom(id){return (DATA.knowledge_edges||[]).filter(e=>e.from===id);} +function selectClear(){$$('.node,.docnode').forEach(n=>n.classList.remove('sel'));} +function detailRepo(name){selectClear(); + const g=$(`.node[data-name="${cssEsc(name)}"]`);if(g)g.classList.add('sel'); + const outs=(DATA.relations||[]).filter(e=>e.from===name&&!e.external); + const descBy=(DATA.knowledge_edges||[]).filter(e=>e.type==='describes'&&e.to===name); + $('#detail').innerHTML=`

    ${esc(name)} repo

    `+ + `
    roles: ${esc((DATA.roles[name]||[]).join(', '))||'none'}
    `+ + `

    depends on

    `+(outs.map(e=>`
    ${esc(e.to)} [${esc(e.confidence)}]
    `).join('')||'none')+ + `

    documented by

    `+(descBy.map(e=>linkNode(e.from,'doc')).join('')||'none'); + pushTrail({kind:'repo',id:name});} +function detailDoc(id){selectClear(); + const g=$(`.docnode[data-doc="${cssEsc(id)}"]`);if(g)g.classList.add('sel'); + const d=docs[id]||{title:id}; + const out=kedgesFrom(id); + const desc=out.filter(e=>e.type==='describes').map(e=>esc(e.to)).join(', '); + const links=out.filter(e=>e.type!=='describes').map(e=>linkNode(e.to,e.to_kind)).join('')||'none'; + const back=(DATA.backlinks&&DATA.backlinks[id]||[]).map(b=>linkNode(b.from,'doc')).join('')||'none'; + $('#detail').innerHTML=`

    ${esc(d.title)} doc

    `+ + (desc?`
    describes ${desc}
    `:'')+ + `

    links

    ${links}

    linked from

    ${back}`+ + `
    ${DATA.doc_html[id]||''}
    `; + wireWikilinks();pushTrail({kind:'doc',id});} +function linkNode(id,kind){const label=kind==='repo'?id:(docs[id]?docs[id].title:id); + return ``;} +function cssEsc(s){return String(s).replace(/["\\]/g,'\\$&');} +function go(kind,id){kind==='repo'?detailRepo(id):detailDoc(id); + const sel=kind==='repo'?`.node[data-name="${cssEsc(id)}"]`:`.docnode[data-doc="${cssEsc(id)}"]`; + const el=$(sel);if(el&&el.scrollIntoView)el.scrollIntoView({block:'center',inline:'center'});} +function wireWikilinks(){$$('#detail .wikilink,#detail .navlink').forEach(a=>a.addEventListener('click',ev=>{ + ev.preventDefault();const t=a.dataset.atlasTarget?tgt[a.dataset.atlasTarget]:{kind:a.dataset.kind,id:a.dataset.id}; + if(t)go(t.kind,t.id);}));} +let trail=[]; +function pushTrail(node){if(trail.length&&trail[trail.length-1].id===node.id)return;trail.push(node);renderTrail();} +function renderTrail(){$('#trail').innerHTML=trail.map((n,i)=>`${esc(n.id)}`).join(' > '); + $$('#trail a').forEach(a=>a.addEventListener('click',()=>{const n=trail[+a.dataset.i];trail=trail.slice(0,+a.dataset.i);go(n.kind,n.id);}));} +let view={k:1,tx:0,ty:0}; +function applyView(){const vp=$('#viewport');if(vp)vp.setAttribute('transform',`translate(${view.tx},${view.ty}) scale(${view.k})`);} +function svgPt(svg,cx,cy){const r=svg.getBoundingClientRect();const vb=svg.viewBox.baseVal; + return {x:(cx-r.left)/r.width*vb.width,y:(cy-r.top)/r.height*vb.height};} +function wireZoom(){const stage=$('#stage'),svg=stage&&stage.querySelector('svg');if(!svg)return; + svg.addEventListener('wheel',ev=>{ev.preventDefault();const p=svgPt(svg,ev.clientX,ev.clientY); + const f=ev.deltaY<0?1.1:1/1.1,nk=Math.min(8,Math.max(.2,view.k*f)); + view.tx=p.x-(p.x-view.tx)*(nk/view.k);view.ty=p.y-(p.y-view.ty)*(nk/view.k);view.k=nk;applyView();},{passive:false}); + let drag=null; + svg.addEventListener('pointerdown',ev=>{drag={x:ev.clientX,y:ev.clientY,tx:view.tx,ty:view.ty}; + stage.classList.add('grabbing');svg.setPointerCapture(ev.pointerId);}); + svg.addEventListener('pointermove',ev=>{if(!drag)return;const r=svg.getBoundingClientRect(),vb=svg.viewBox.baseVal; + view.tx=drag.tx+(ev.clientX-drag.x)*vb.width/r.width;view.ty=drag.ty+(ev.clientY-drag.y)*vb.height/r.height;applyView();}); + svg.addEventListener('pointerup',()=>{drag=null;stage.classList.remove('grabbing');}); + $('#zoom-reset').addEventListener('click',()=>{view={k:1,tx:0,ty:0};applyView();});} +function searchApply(){const q=$('#search').value.trim().toLowerCase();const on=new Set(); + $$('.node').forEach(g=>{const m=!q||g.dataset.name.toLowerCase().includes(q); + g.classList.toggle('dim',!m);if(m)on.add('repo:'+g.dataset.name);}); + $$('.docnode').forEach(g=>{const d=docs[g.dataset.doc]; + const m=!q||g.dataset.doc.toLowerCase().includes(q)||(d&&d.title.toLowerCase().includes(q)); + g.classList.toggle('dim',!m);if(m)on.add('doc:'+g.dataset.doc);}); + $$('.edge').forEach(p=>p.classList.toggle('dim',!!q&&!(on.has('repo:'+p.dataset.from)&&on.has('repo:'+p.dataset.to)))); + $$('.kedge').forEach(l=>{const t=on.has('repo:'+l.dataset.to)||on.has('doc:'+l.dataset.to); + l.classList.toggle('dim',!!q&&!(on.has('doc:'+l.dataset.from)&&t));});} +function wireMentions(){const b=$('#toggle-mentions');b.addEventListener('click',()=>{ + const on=b.getAttribute('aria-pressed')==='true';b.setAttribute('aria-pressed',String(!on)); + $$('.kedge-mentions').forEach(l=>{l.style.display=on?'none':'';});});} +function neighborhood(kind,id){const keep=new Set([kind+':'+id]); + (DATA.relations||[]).forEach(e=>{if(e.external)return; + if(kind==='repo'&&e.from===id)keep.add('repo:'+e.to); + if(kind==='repo'&&e.to===id)keep.add('repo:'+e.from);}); + (DATA.knowledge_edges||[]).forEach(e=>{ + if(kind==='doc'&&e.from===id)keep.add(e.to_kind+':'+e.to); + if(e.to===id&&((kind==='repo'&&e.to_kind==='repo')||(kind==='doc'&&e.to_kind==='doc')))keep.add('doc:'+e.from);}); + return keep;} +function focusOn(kind,id){const keep=neighborhood(kind,id); + $$('.node').forEach(g=>g.classList.toggle('dim',!keep.has('repo:'+g.dataset.name))); + $$('.docnode').forEach(g=>g.classList.toggle('dim',!keep.has('doc:'+g.dataset.doc))); + $$('.edge').forEach(p=>p.classList.toggle('dim',!(keep.has('repo:'+p.dataset.from)&&keep.has('repo:'+p.dataset.to)))); + $$('.kedge').forEach(l=>l.classList.toggle('dim',!(keep.has('doc:'+l.dataset.from)&&(keep.has('repo:'+l.dataset.to)||keep.has('doc:'+l.dataset.to)))));} +function clearFocus(){$$('.dim').forEach(e=>e.classList.remove('dim'));} +function wire(){ + $$('.node').forEach(g=>g.addEventListener('click',()=>detailRepo(g.dataset.name))); + $$('.docnode').forEach(g=>g.addEventListener('click',()=>detailDoc(g.dataset.doc))); + $$('.node').forEach(g=>g.addEventListener('dblclick',()=>focusOn('repo',g.dataset.name))); + $$('.docnode').forEach(g=>g.addEventListener('dblclick',()=>focusOn('doc',g.dataset.doc))); + $('#focus-clear').addEventListener('click',clearFocus); + $('#search').addEventListener('input',searchApply); + wireMentions(); + wireZoom(); +} +document.addEventListener('DOMContentLoaded',wire); +""" diff --git a/client-plugin/server/src/index_graph/viz/atlas_html.py b/client-plugin/server/src/index_graph/viz/atlas_html.py new file mode 100644 index 0000000..88b7044 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/atlas_html.py @@ -0,0 +1,43 @@ +"""Assemble the self-contained atlas dashboard document.""" +from __future__ import annotations + +import json + +from ..knowledge.markdown import render_markdown +from .theme import css_variables +from .atlas_assets import ATLAS_CSS, ATLAS_JS, ATLAS_SKIP_LINK + + +def _backlinks(pack: dict) -> dict: + out: dict[str, list] = {} + for e in sorted(pack.get("knowledge_edges", []), key=lambda e: (e["to"], e["from"], e["type"])): + out.setdefault(e["to"], []).append({"from": e["from"], "type": e["type"]}) + return out + + +def render_atlas_html(pack: dict, docs: list, *, svg: str, include_external: bool = True) -> str: + rendered = {d.rel_path: render_markdown(d.body) for d in sorted(docs, key=lambda d: d.rel_path)} + data = dict(pack) + data["doc_html"] = rendered + data["backlinks"] = _backlinks(pack) + blob = json.dumps(data, sort_keys=True, separators=(",", ":")).replace("<", "\\u003c") + return ( + "" + '' + '' + "index | atlas" + f"" + f"{ATLAS_SKIP_LINK}" + '
    ' + '
    ' + '' + '' + '' + '' + "
    " + '
    ' + f"{svg}
    " + '
    ' + f"" + "" + ) diff --git a/client-plugin/server/src/index_graph/viz/atlas_layout.py b/client-plugin/server/src/index_graph/viz/atlas_layout.py new file mode 100644 index 0000000..40f608f --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/atlas_layout.py @@ -0,0 +1,103 @@ +"""Two-layer atlas layout: repo positions (reused) + doc satellites + knowledge band.""" +from __future__ import annotations + +from dataclasses import dataclass + +from .layout import build_layout, LayoutModel, MARGIN + +_DOC_W, _DOC_H, _ROW_GAP, _COL_GAP = 160.0, 30.0, 10.0, 24.0 + + +@dataclass(frozen=True) +class DocNode: + id: str + title: str + x: float + y: float + w: float + h: float + describes: str | None # repo name, or None for a band (cross-cutting) doc + + +@dataclass(frozen=True) +class KEdge: + type: str # describes | links-to | mentions + frm: str + to: str + to_kind: str # repo | doc + points: tuple[tuple[float, float], tuple[float, float]] + + +@dataclass(frozen=True) +class AtlasLayout: + repo_layout: LayoutModel + docs: tuple[DocNode, ...] + kedges: tuple[KEdge, ...] + width: float + height: float + + +def build_atlas_layout(pack: dict, *, include_external: bool = True) -> AtlasLayout: + repo_layout = build_layout(pack, include_external=include_external) + repo_by_name = {n.name: n for n in repo_layout.nodes} + + # describes target per doc (first by sorted edge order wins -> deterministic) + describes: dict[str, str] = {} + for e in sorted(pack.get("knowledge_edges", []), key=lambda e: (e["from"], e["type"], e["to_kind"], e["to"])): + if e["type"] == "describes" and e["from"] not in describes: + describes[e["from"]] = e["to"] + + docs_meta = sorted(pack.get("docs", []), key=lambda d: d["id"]) + region_top = repo_layout.height + 50.0 + placed: dict[str, DocNode] = {} + + # described docs -> one column per repo, columns flow left-to-right without overlapping + by_repo: dict[str, list[dict]] = {} + for d in docs_meta: + target = describes.get(d["id"]) + if target in repo_by_name: + by_repo.setdefault(target, []).append(d) + cursor_x = MARGIN + col_depth = 0 + for repo in sorted(by_repo, key=lambda r: (repo_by_name[r].x, r)): + x = max(repo_by_name[repo].x, cursor_x) # under its repo, never overlapping the prior column + for k, d in enumerate(by_repo[repo]): + y = region_top + k * (_DOC_H + _ROW_GAP) + placed[d["id"]] = DocNode(d["id"], d["title"], x, y, _DOC_W, _DOC_H, repo) + cursor_x = x + _DOC_W + _COL_GAP + col_depth = max(col_depth, len(by_repo[repo])) + + # band docs (describe nothing / unknown repo) -> a wrapping row beneath the columns + band_top = region_top + max(col_depth, 1) * (_DOC_H + _ROW_GAP) + 40.0 + width_guess = max(repo_layout.width, cursor_x) + bx = MARGIN + for d in docs_meta: + if d["id"] in placed: + continue + if bx > MARGIN and bx + _DOC_W + MARGIN > width_guess: + bx = MARGIN + band_top += _DOC_H + _ROW_GAP + placed[d["id"]] = DocNode(d["id"], d["title"], bx, band_top, _DOC_W, _DOC_H, None) + bx += _DOC_W + _COL_GAP + + docs = tuple(placed[d["id"]] for d in docs_meta) + + def _center(node_id: str, kind: str): + if kind == "repo" and node_id in repo_by_name: + r = repo_by_name[node_id] + return (r.x + r.w / 2.0, r.y + r.h / 2.0) + if node_id in placed: + d = placed[node_id] + return (d.x + d.w / 2.0, d.y + d.h / 2.0) + return None + + kedges: list[KEdge] = [] + for e in sorted(pack.get("knowledge_edges", []), key=lambda e: (e["from"], e["type"], e["to_kind"], e["to"])): + a = _center(e["from"], "doc") + b = _center(e["to"], e["to_kind"]) + if a is not None and b is not None: + kedges.append(KEdge(e["type"], e["from"], e["to"], e["to_kind"], (a, b))) + + width = max([d.x + d.w for d in docs] + [repo_layout.width]) + MARGIN + height = max([d.y + d.h for d in docs] + [repo_layout.height]) + MARGIN + return AtlasLayout(repo_layout, docs, tuple(kedges), width, height) diff --git a/client-plugin/server/src/index_graph/viz/atlas_svg.py b/client-plugin/server/src/index_graph/viz/atlas_svg.py new file mode 100644 index 0000000..aa5c24e --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/atlas_svg.py @@ -0,0 +1,63 @@ +"""Render an AtlasLayout to a self-contained two-layer SVG (repos + docs).""" +from __future__ import annotations + +from xml.sax.saxutils import escape, quoteattr + +from .atlas_layout import AtlasLayout, DocNode, KEdge +from .svg import _node_svg, _edge_svg # reuse repo node + dependency edge renderers +from .theme import THEME, svg_style + + +def _atlas_style() -> str: + t = THEME + return ( + f".docnode rect{{fill:{t.bg};stroke:{t.gold};stroke-dasharray:3 2;}}" + f".docnode text{{font-family:{t.font_mono};fill:{t.ink};font-size:11px;}}" + f".docnode.band rect{{stroke:{t.teal};}}" + f".docnode.sel rect{{stroke:{t.accent};stroke-width:2;stroke-dasharray:none;}}" + f".kedge{{fill:none;stroke-width:1;}}" + f".kedge-describes{{stroke:{t.gold};}}" + f".kedge-links-to{{stroke:{t.ok};stroke-dasharray:4 3;}}" + f".kedge-mentions{{stroke:{t.muted};stroke-dasharray:1 4;opacity:.35;}}" + f".kedge.dim,.node.dim,.docnode.dim,.edge.dim{{opacity:.08;}}" + ) + + +def _doc_svg(d: DocNode) -> str: + cls = "docnode" + ("" if d.describes is not None else " band") + return ( + f"' + f'' + f'{escape(d.title)}' + ) + + +def _kedge_svg(k: KEdge) -> str: + (a, b) = k.points + cls = "kedge kedge-" + k.type + return ( + f"' + ) + + +def render_atlas_svg(atlas: AtlasLayout) -> str: + rl = atlas.repo_layout + defs = ( + '' + f'' + ) + style = f"" + kedges = "".join(_kedge_svg(k) for k in atlas.kedges) + repo_edges = "".join(_edge_svg(e) for e in rl.edges) + repo_nodes = "".join(_node_svg(n) for n in rl.nodes) + doc_nodes = "".join(_doc_svg(d) for d in atlas.docs) + return ( + f'' + f'{defs}{style}{kedges}{repo_edges}{repo_nodes}{doc_nodes}' + ) diff --git a/client-plugin/server/src/index_graph/viz/charts.py b/client-plugin/server/src/index_graph/viz/charts.py new file mode 100644 index 0000000..df8a082 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/charts.py @@ -0,0 +1,49 @@ +"""Three compact HTML+CSS bar charts derived purely by counting the pack.""" +from __future__ import annotations + +from collections import Counter +from xml.sax.saxutils import escape + +_PRECEDENCE = ("entrypoint", "orchestrator", "hub", "library", "leaf", "isolated") + + +def _primary(roles: list[str]) -> str: + for r in _PRECEDENCE: + if r in roles: + return r + return "isolated" + + +def _bars(pairs: list[tuple[str, int]]) -> str: + top = max((v for _, v in pairs), default=0) or 1 + rows = [] + for label, value in pairs: + pct = round(100 * value / top) + rows.append( + f'
    {escape(label)}' + f'{value}
    ' + ) + return '
    ' + "".join(rows) + "
    " + + +def render_charts(pack: dict, *, include_external: bool = True) -> dict[str, str]: + rels = pack.get("relations", []) + counted_rels = rels if include_external else [r for r in rels if not r.get("external")] + conf = Counter(r.get("confidence", "low") for r in counted_rels) + confidence = _bars([(k, conf.get(k, 0)) for k in ("high", "moderate", "low")]) + + roles = pack.get("roles", {}) + role_counts = Counter(_primary(list(roles.get(r["name"], ()))) for r in pack.get("repos", [])) + role_chart = _bars([(k, role_counts[k]) for k in _PRECEDENCE if role_counts[k]]) + + sal = pack.get("salience", {}) + fan_in = sorted(sal.items(), key=lambda kv: (-kv[1].get("in_degree", 0), kv[0]))[:5] + fan_out = sorted(sal.items(), key=lambda kv: (-kv[1].get("out_degree", 0), kv[0]))[:5] + fanio = ( + '

    Most depended-on

    ' + + _bars([(k, v.get("in_degree", 0)) for k, v in fan_in]) + + '

    Most dependencies

    ' + + _bars([(k, v.get("out_degree", 0)) for k, v in fan_out]) + ) + return {"confidence": confidence, "roles": role_chart, "fanio": fanio} diff --git a/client-plugin/server/src/index_graph/viz/html.py b/client-plugin/server/src/index_graph/viz/html.py new file mode 100644 index 0000000..ca11c08 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/html.py @@ -0,0 +1,136 @@ +"""Render the hero: a single self-contained interactive dashboard.""" +from __future__ import annotations + +import json +from xml.sax.saxutils import escape + +from .theme import css_variables, ROLE_COLOR, THEME + +_CSS = """ +*{box-sizing:border-box}body{margin:0;background:var(--bg);color:var(--ink); +font-family:var(--font-body)}main{display:grid;grid-template-columns:1fr 340px; +min-height:100vh}#stage{overflow:auto;padding:1rem}aside{border-left:1px solid +var(--hairline);padding:1rem;font-family:var(--font-mono);font-size:.82rem} +.controls{display:flex;flex-wrap:wrap;gap:.4rem;margin-bottom:.8rem} +.chip{cursor:pointer;border:1px solid var(--hairline);border-radius:6px; +padding:.2em .5em;background:transparent;color:var(--ink)} +.chip[aria-pressed=true]{background:var(--accent);color:var(--bg)} +input[type=search]{width:100%;padding:.4em;background:transparent;color:var(--ink); +border:1px solid var(--hairline);border-radius:6px;margin-bottom:.6rem} +.node.dim{opacity:.12}.edge.dim{opacity:.05} +.node:focus rect,.node.sel rect{stroke:var(--accent);stroke-width:2} +.row{display:flex;align-items:center;gap:.4rem;margin:.15rem 0} +.lbl{width:6.5rem;opacity:.85}.bar{height:.7rem;background:var(--teal);border-radius:3px} +.num{opacity:.7}h4{margin:.6rem 0 .2rem;color:var(--gold)} +@media(prefers-reduced-motion:reduce){*{transition:none!important}} +@media(max-width:820px){main{grid-template-columns:1fr}aside{border-left:none}} +.tip{position:fixed;pointer-events:none;z-index:9;max-width:24rem;background:var(--bg); +border:1px solid var(--accent);border-radius:6px;padding:.4em .6em;font-family:var(--font-mono); +font-size:.74rem;line-height:1.35}.tip[hidden]{display:none} +.legend{display:flex;flex-wrap:wrap;align-items:center;gap:.5rem;margin:.2rem 0 .8rem; +font-family:var(--font-mono);font-size:.72rem;opacity:.85}.legend b{color:var(--gold)} +.leg{display:inline-flex;align-items:center;gap:.3rem}.leg i{width:.8rem;height:.8rem; +border-radius:2px;display:inline-block}.leg i.ln{height:0;width:1.2rem;border-top:2px solid} +.leg i.ln.d5{border-top-style:dashed}.leg i.ln.d2{border-top-style:dotted} +""" + +_JS = """ +const $=s=>document.querySelector(s),$$=s=>[...document.querySelectorAll(s)]; +const esc=s=>String(s).replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); +const state={q:'',roles:new Set(),conf:new Set(),ext:true}; +const idx={};DATA.repos.forEach(r=>idx[r.name]=r); +function match(name){const r=idx[name];if(!r)return state.ext; + if(state.q&&!name.toLowerCase().includes(state.q))return false; + const role=(DATA.roles[name]||['isolated'])[0]; + if(state.roles.size&&!state.roles.has(role))return false;return true;} +function apply(){$$('.node').forEach(g=>{g.classList.toggle('dim',!match(g.dataset.name));}); + $$('.edge').forEach(p=>{const on=match(p.dataset.from)&&match(p.dataset.to)&& + (!state.conf.size||state.conf.has(p.className.baseVal.match(/edge-(high|moderate|low)/)?.[1])); + p.classList.toggle('dim',!on);});} +function detail(name){const r=idx[name]||{name,ecosystems:[],markers:[]}; + const outs=DATA.relations.filter(e=>e.from===name); + const ins=DATA.relations.filter(e=>e.to===name); + const sig=e=>(e.signals||[]).map(s=>`${esc(s.file)}${s.line?':'+esc(s.line):''} ${esc(s.kind)}`).join('; '); + $('#detail').innerHTML=`

    ${esc(name)}

    roles: ${esc((DATA.roles[name]||[]).join(', '))||'none'}
    +
    in ${ (DATA.salience[name]||{}).in_degree||0 } · out ${ (DATA.salience[name]||{}).out_degree||0 }
    +

    depends on

    ${outs.map(e=>`
    ${esc(e.target_name)} [${esc(e.confidence)}] ${sig(e)}
    `).join('')||'none'} +

    depended on by

    ${ins.map(e=>`
    ${esc(e.from)} [${esc(e.confidence)}]
    `).join('')||'none'}`;} +const tip=Object.assign(document.createElement('div'),{className:'tip',hidden:true}); +function edgeTip(p,x,y){const sg=JSON.parse(p.getAttribute('data-signals')||'[]'); + const conf=(p.className.baseVal.match(/edge-(high|moderate|low)/)||[])[1]||'declared'; + const ev=sg.map(s=>`${esc(s.file)}${s.line?':'+esc(s.line):''} (${esc(s.kind)})`).join('
    ')||'manifest'; + tip.innerHTML=`${esc(p.dataset.from)} → ${esc(p.dataset.to)} · ${esc(conf)}
    ${ev}`; + tip.hidden=false;tip.style.left=(x+12)+'px';tip.style.top=(y+12)+'px';} +function nbrs(name){const s=new Set([name]);DATA.relations.forEach(e=>{ + if(e.from===name&&e.to)s.add(e.to);if(e.to===name)s.add(e.from);});return s;} +function highlight(name){const s=nbrs(name); + $$('.node').forEach(g=>g.classList.toggle('dim',!s.has(g.dataset.name))); + $$('.edge').forEach(p=>p.classList.toggle('dim',!(s.has(p.dataset.from)&&s.has(p.dataset.to))));} +function wire(){ + $('#search').addEventListener('input',e=>{state.q=e.target.value.toLowerCase();apply();}); + $$('.chip[data-role]').forEach(c=>c.addEventListener('click',()=>{ + const r=c.dataset.role;state.roles.has(r)?state.roles.delete(r):state.roles.add(r); + c.setAttribute('aria-pressed',state.roles.has(r));apply();})); + $$('.node').forEach(g=>{const pick=()=>{$$('.node').forEach(n=>n.classList.remove('sel')); + g.classList.add('sel');detail(g.dataset.name);}; + g.addEventListener('click',pick);g.addEventListener('keydown',e=>{if(e.key==='Enter')pick();});}); + document.body.appendChild(tip); + $$('.edge').forEach(p=>{p.addEventListener('mousemove',e=>edgeTip(p,e.clientX,e.clientY)); + p.addEventListener('mouseleave',()=>tip.hidden=true);}); + $$('.node').forEach(g=>{g.addEventListener('mouseenter',()=>highlight(g.dataset.name)); + g.addEventListener('mouseleave',apply);}); + apply();} +document.addEventListener('DOMContentLoaded',wire); +""" + + +def _legend() -> str: + roles = "".join( + f'{escape(r)}' + for r, c in ROLE_COLOR.items() if r != "external") + edges = ( + f'high' + f'moderate' + f'low / external' + f'cycle') + return (f'
    roles{roles}' + f'edges{edges}
    ') + + +def _salience_audit_panel(pack: dict) -> str: + entries = pack.get("salience_audit", []) + rows = "".join( + f'
    {escape(str(e.get("node", "")))}' + f' [{escape(str(e.get("kind", "")))}]' + f': {escape(str(e.get("note", "")))}
    ' + for e in entries + ) + body = rows if rows else '
    none
    ' + return f"

    salience audit

    {body}" + + +def render_html(pack: dict, *, svg: str, charts: dict[str, str]) -> str: + data = json.dumps(pack, sort_keys=True, separators=(",", ":")).replace("<", "\\u003c") + roles = sorted({(rs or ["isolated"])[0] for rs in pack.get("roles", {}).values()}) + chips = "".join( + f'' for r in roles + ) + audit_panel = _salience_audit_panel(pack) + return ( + "" + '' + '' + "index · context" + f"" + '
    ' + f'
    {chips}
    ' + f"{_legend()}{svg}
    " + '
    ' + f"" + "" + ) diff --git a/client-plugin/server/src/index_graph/viz/layout.py b/client-plugin/server/src/index_graph/viz/layout.py new file mode 100644 index 0000000..0f956bb --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/layout.py @@ -0,0 +1,215 @@ +"""Deterministic layered-by-role layout for the dependency graph.""" +from __future__ import annotations + +from dataclasses import dataclass, replace + +ROLE_PRECEDENCE = ("entrypoint", "orchestrator", "hub", "library", "leaf", "isolated") +_EXTERNAL_LAYER = len(ROLE_PRECEDENCE) +_SWEEPS = 3 + +LAYER_GAP = 140.0 +NODE_H = 44.0 +NODE_GAP = 28.0 +CHAR_W = 8.5 +PAD_X = 24.0 +MARGIN = 40.0 +BACK_BOW = 60.0 # right-side headroom a back-edge bows into; reserved in the canvas + + +@dataclass(frozen=True) +class LaidNode: + name: str + role: str + roles: tuple[str, ...] + layer: int + order: int = 0 + x: float = 0.0 + y: float = 0.0 + w: float = 0.0 + h: float = 0.0 + external: bool = False + in_degree: int = 0 + out_degree: int = 0 + hub: bool = False + in_cycle: bool = False + + +@dataclass(frozen=True) +class LaidEdge: + from_repo: str + to_repo: str + confidence: str + external: bool + back_edge: bool = False + in_cycle: bool = False + points: tuple[tuple[float, float], ...] = () + signals: tuple[dict, ...] = () + + +@dataclass(frozen=True) +class LayoutModel: + nodes: tuple[LaidNode, ...] + edges: tuple[LaidEdge, ...] + # Lists only PRESENT layers in ascending order; NOT index-aligned to ROLE_PRECEDENCE. + layers: tuple[str, ...] + width: float = 0.0 + height: float = 0.0 + + +def _primary_role(roles: list[str]) -> str: + for role in ROLE_PRECEDENCE: + if role in roles: + return role + return "isolated" + + +def _build_nodes(pack: dict, include_external: bool) -> list[LaidNode]: + roles = pack.get("roles", {}) + salience = pack.get("salience", {}) + cycle_members = {n for c in pack.get("cycles", []) for n in c} + nodes: list[LaidNode] = [] + for repo in pack.get("repos", []): + name = repo["name"] + rs = tuple(roles.get(name, ())) + primary = _primary_role(list(rs)) + sal = salience.get(name, {}) + nodes.append( + LaidNode( + name=name, + role=primary, + roles=rs, + layer=ROLE_PRECEDENCE.index(primary), + in_degree=int(sal.get("in_degree", 0)), + out_degree=int(sal.get("out_degree", 0)), + hub=bool(sal.get("hub", False)), + in_cycle=name in cycle_members, + ) + ) + if include_external: + seen = set() + for rel in pack.get("relations", []): + if rel.get("external") and rel["target_name"] not in seen: + seen.add(rel["target_name"]) + nodes.append( + LaidNode( + name=rel["target_name"], + role="external", + roles=("external",), + layer=_EXTERNAL_LAYER, + external=True, + ) + ) + return nodes + + +def _build_edges(pack: dict, names: set[str], include_external: bool) -> list[LaidEdge]: + edges: list[LaidEdge] = [] + for rel in pack.get("relations", []): + external = bool(rel.get("external")) + target = rel["target_name"] if external else rel["to"] + if external and not include_external: + continue + if rel["from"] not in names: + continue + if target not in names: + continue + edges.append( + LaidEdge( + from_repo=rel["from"], + to_repo=target, + confidence=rel.get("confidence", "low"), + external=external, + in_cycle=rel.get("in_cycle", False), + signals=tuple(rel.get("signals", ())), + ) + ) + return edges + + +def _order_within_layers(nodes: list[LaidNode], edges: list[LaidEdge]) -> list[LaidNode]: + by_layer: dict[int, list[LaidNode]] = {} + for n in nodes: + by_layer.setdefault(n.layer, []).append(n) + # initial stable order: alphabetical by name + for layer in by_layer.values(): + layer.sort(key=lambda n: n.name) + # adjacency for barycentre + nbrs: dict[str, list[str]] = {n.name: [] for n in nodes} + for e in edges: + nbrs.setdefault(e.from_repo, []).append(e.to_repo) + nbrs.setdefault(e.to_repo, []).append(e.from_repo) + for _ in range(_SWEEPS): + pos = {n.name: i for layer in by_layer.values() for i, n in enumerate(layer)} + for layer in by_layer.values(): + def bary(n: LaidNode) -> tuple[float, str]: + ns = [pos[m] for m in nbrs.get(n.name, []) if m in pos] + return (sum(ns) / len(ns) if ns else pos[n.name], n.name) + layer.sort(key=bary) + ordered: list[LaidNode] = [] + for layer_idx in sorted(by_layer): + for order, n in enumerate(by_layer[layer_idx]): + ordered.append(replace(n, order=order)) + return ordered + + +def _place(nodes: list[LaidNode]) -> tuple[list[LaidNode], float, float]: + by_layer: dict[int, list[LaidNode]] = {} + for n in nodes: + by_layer.setdefault(n.layer, []).append(n) + for layer in by_layer.values(): + layer.sort(key=lambda n: n.order) + widths = { + idx: sum(len(n.name) * CHAR_W + PAD_X for n in layer) + NODE_GAP * (len(layer) - 1) + for idx, layer in by_layer.items() + } + content_w = max(widths.values(), default=0.0) + canvas_w = content_w + 2 * MARGIN + BACK_BOW # reserve right headroom for back-edge bows + placed: dict[str, LaidNode] = {} + for layer_idx, layer in by_layer.items(): + cursor = MARGIN + (content_w - widths[layer_idx]) / 2 # centre within the content band + y = MARGIN + layer_idx * LAYER_GAP + for n in layer: + w = len(n.name) * CHAR_W + PAD_X + placed[n.name] = replace(n, x=cursor, y=y, w=w, h=NODE_H) + cursor += w + NODE_GAP + ordered = [placed[n.name] for n in nodes] + layer_count = (max(by_layer) + 1) if by_layer else 0 + canvas_h = MARGIN * 2 + max(layer_count - 1, 0) * LAYER_GAP + NODE_H + return ordered, canvas_w, canvas_h + + +def _route(edges: list[LaidEdge], nodes: list[LaidNode]) -> list[LaidEdge]: + by_name = {n.name: n for n in nodes} + routed: list[LaidEdge] = [] + for e in edges: + src, dst = by_name.get(e.from_repo), by_name.get(e.to_repo) + if src is None or dst is None: + continue + back = dst.layer <= src.layer + sx, sy = src.x + src.w / 2, src.y + src.h + tx, ty = dst.x + dst.w / 2, dst.y + if back: # route the upward/lateral return through the reserved right headroom + sy = src.y + src.h / 2 + ty = dst.y + dst.h / 2 + pts = ((sx, sy), (sx + BACK_BOW, sy), (tx + BACK_BOW, ty), (tx, ty)) + else: + dy = (ty - sy) * 0.4 + pts = ((sx, sy), (sx, sy + dy), (tx, ty - dy), (tx, ty)) + routed.append(replace(e, back_edge=back, points=pts)) + return routed + + +def build_layout(pack: dict, *, include_external: bool = True) -> LayoutModel: + nodes = _build_nodes(pack, include_external) + names = {n.name for n in nodes} + edges = _build_edges(pack, names, include_external) + nodes = _order_within_layers(nodes, edges) + nodes, width, height = _place(nodes) + edges = _route(edges, nodes) + present = sorted({n.layer for n in nodes}) + labels = tuple( + (ROLE_PRECEDENCE[i] if i < len(ROLE_PRECEDENCE) else "external") for i in present + ) + return LayoutModel( + nodes=tuple(nodes), edges=tuple(edges), layers=labels, width=width, height=height + ) diff --git a/client-plugin/server/src/index_graph/viz/lens_html.py b/client-plugin/server/src/index_graph/viz/lens_html.py new file mode 100644 index 0000000..167ffae --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/lens_html.py @@ -0,0 +1,219 @@ +"""The Context Lens: a self-contained page that shows what an agent's context +budget actually holds, and what it drops and why, as you move the budget. + +No other map tool renders this. Obsidian shows notes; graph tools show nodes. +This shows a SELECTION under a token budget: one ordered stack of repos in the +exact rank the greedy fill considers them, with a budget frontier that slides +as you drag the budget. Repos above the frontier are RETAINED; the moment one +crosses below it, it is OMITTED and carries its failure code +(budget_exceeded / outside_focus_or_budget). The replay runs the exact greedy +rule from context/lens.py in the browser over the pack's own numbers, so the +frontier the slider draws is the frontier the CLI would emit. Zero runtime +deps, deterministic, offline; the receipt hash is printed on the page. + +Brand: ceramic bg, ink text, one iris accent, mono for receipts — cohesive +with the index atlas. Motion conveys state change only (a repo crossing the +frontier), with a reduced-motion fallback. +""" +from __future__ import annotations + +import html +import json + +_CSS = """ +:root{ + --bg:#f4f3ef;--panel:#fbfaf7;--ink:#0b0c0e;--soft:#2f3238;--muted:#565a62; + --hair:rgba(11,12,14,.12);--hair2:rgba(11,12,14,.07); + --iris:#4636e8;--iris-wash:rgba(70,54,232,.08);--iris-line:rgba(70,54,232,.35); + --drop:#8a5a12;--in-ink:#0b0c0e; + --mono:ui-monospace,SFMono-Regular,Consolas,"SF Mono",monospace; + --body:Arial,Helvetica,"Helvetica Neue",sans-serif; + --e:cubic-bezier(.22,1,.36,1)} +*{box-sizing:border-box} +html{-webkit-text-size-adjust:100%} +body{margin:0;background:var(--bg);color:var(--ink);font-family:var(--body); + font-size:16px;line-height:1.5;-webkit-font-smoothing:antialiased} +.wrap{max-width:1000px;margin:0 auto;padding:2.2rem 1.5rem 3rem} +header{margin-bottom:1.6rem} +.eyebrow{font:600 .72rem/1 var(--mono);letter-spacing:.14em;color:var(--iris); + text-transform:uppercase;margin:0 0 .55rem} +h1{font:600 1.7rem/1.15 var(--body);margin:0 0 .5rem;letter-spacing:-.02em;text-wrap:balance; + display:flex;align-items:baseline;gap:.6rem;flex-wrap:wrap} +.sub{color:var(--soft);font-size:.92rem;max-width:68ch;margin:0} +.sub code{font:.85em var(--mono);background:var(--hair2);padding:.05em .35em;border-radius:4px} +.verdict{font:600 .68rem/1 var(--mono);letter-spacing:.06em;padding:.35em .6em;border-radius:999px; + border:1px solid;vertical-align:middle;white-space:nowrap;transition:color .2s,border-color .2s,background .2s} +.verdict[data-v=MATCH]{color:#1f6b45;border-color:rgba(31,107,69,.4);background:rgba(31,107,69,.07)} +.verdict[data-v=UNVERIFIABLE]{color:var(--drop);border-color:rgba(138,90,18,.4);background:rgba(138,90,18,.08)} + +.panel{background:var(--panel);border:1px solid var(--hair);border-radius:14px; + box-shadow:0 1px 0 rgba(11,12,14,.03),0 12px 30px -22px rgba(11,12,14,.35);overflow:hidden} +.rail{display:grid;grid-template-columns:1fr auto;gap:1rem 1.4rem;align-items:center; + padding:1.1rem 1.3rem;border-bottom:1px solid var(--hair2)} +.slider{grid-column:1/-1;display:flex;align-items:center;gap:1rem} +.slider label{font:600 .74rem/1 var(--mono);letter-spacing:.04em;color:var(--muted);white-space:nowrap} +input[type=range]{flex:1;height:26px;accent-color:var(--iris);cursor:pointer} +input[type=range]:focus-visible{outline:2px solid var(--iris);outline-offset:4px;border-radius:4px} +.readout{display:flex;align-items:baseline;gap:.5rem;font:var(--mono)} +.readout .big{font:600 1.35rem/1 var(--mono);color:var(--iris);letter-spacing:-.01em; + font-variant-numeric:tabular-nums} +.readout .of{font:.8rem var(--mono);color:var(--muted)} +.legend{display:flex;gap:1.1rem;font:600 .72rem/1 var(--mono);letter-spacing:.03em;flex-wrap:wrap} +.legend b{font-weight:600} +.legend .k{display:inline-flex;align-items:center;gap:.4em;color:var(--muted)} +.legend .dot{width:.7em;height:.7em;border-radius:2px;display:inline-block} +.legend .dot.in{background:var(--iris)} +.legend .dot.out{background:transparent;border:1px solid var(--muted)} +.legend .n{color:var(--ink);font-variant-numeric:tabular-nums} + +.stack{list-style:none;margin:0;padding:.3rem 0} +.frontier{position:relative;height:0;margin:0 1.3rem;border-top:1.5px dashed var(--iris-line); + transition:margin-top .28s var(--e)} +.frontier::after{content:"budget frontier — " attr(data-tok) " tok";position:absolute;right:0;top:-.55rem; + font:600 .64rem/1 var(--mono);letter-spacing:.05em;color:var(--iris);background:var(--panel); + padding:.2em .5em;border:1px solid var(--iris-line);border-radius:999px} +.item{display:grid;grid-template-columns:1.7rem 1fr auto;align-items:center;gap:.2rem .9rem; + padding:.62rem 1.3rem;position:relative;transition:background .24s var(--e),opacity .24s var(--e)} +.item::before{content:"";position:absolute;left:0;top:0;bottom:0;width:3px;background:var(--iris); + transform:scaleY(0);transform-origin:top;transition:transform .26s var(--e)} +.item.in::before{transform:scaleY(1)} +.item+.item{border-top:1px solid var(--hair2)} +.item .rank{font:600 .72rem/1 var(--mono);color:var(--muted);font-variant-numeric:tabular-nums;text-align:right} +.item .name{font:600 .98rem/1.25 var(--body);letter-spacing:-.01em;transition:color .24s} +.item .meta{grid-column:2;font:.74rem/1.4 var(--mono);color:var(--muted); + display:flex;gap:.85rem;flex-wrap:wrap;margin-top:.12rem} +.item .desc{grid-column:2;font:.82rem/1.4 var(--body);color:var(--soft);margin-top:.15rem; + max-width:64ch;overflow:hidden;text-overflow:ellipsis;white-space:nowrap} +.item .cost{font:600 .82rem/1 var(--mono);color:var(--ink);font-variant-numeric:tabular-nums;white-space:nowrap;text-align:right} +.item .code{color:var(--drop);font-weight:600} +.item.in{background:var(--iris-wash)} +.item.out{opacity:.62} +.item.out .name{color:var(--soft)} +.item.cross{animation:cross .5s var(--e)} +@keyframes cross{0%{background:rgba(70,54,232,.22)}100%{background:var(--iris-wash)}} + +footer{margin-top:1.3rem;display:flex;flex-direction:column;gap:.35rem; + font:.72rem/1.5 var(--mono);color:var(--muted);word-break:break-all} +footer .lbl{color:var(--soft);font-weight:600} +@media(max-width:640px){ + .wrap{padding:1.5rem 1rem 2rem}h1{font-size:1.4rem} + .item .desc{white-space:normal} + .rail{grid-template-columns:1fr}.legend{justify-content:flex-start} +} +@media(prefers-reduced-motion:reduce){ + *{transition:none!important;animation:none!important} +} +""" + +_JS = r""" +const P=LENS,ORDER=P.replay.order,BASE=P.replay.base_tokens; +const $=s=>document.querySelector(s); +const esc=s=>String(s).replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); +const OUTSIDE={};(P.envelope.omitted||[]).forEach(o=>{if(o.reason!=='budget_exceeded')OUTSIDE[o.name]=o.reason;}); +function replay(budget){ // MIRRORS context/lens.py replay_retained + const keep=new Set();let approx=BASE; + for(const o of ORDER){if(approx+o.cost<=budget||keep.size===0){keep.add(o.name);approx+=o.cost;}} + return{keep,approx}; +} +// build every row ONCE, in rank order; state is a class toggle so rows glide, never re-mount +const stack=$('#stack'); +ORDER.forEach((o,i)=>{ + const li=document.createElement('li');li.className='item';li.dataset.n=o.name; + const meta=[`in ${o.salience.in_degree}`,`out ${o.salience.out_degree}`, + (o.roles&&o.roles.length?o.roles.join(' · '):'')].filter(Boolean).map(esc).join(''); + li.innerHTML=`${String(i+1).padStart(2,'0')}`+ + `${esc(o.name)}`+ + `~${o.cost}`+ + `${meta}`+ + (o.description?`${esc(o.description)}`:''); + stack.appendChild(li); + // the frontier divider lives right after the last row; moved via order-index + o._el=li; +}); +const frontier=document.createElement('div');frontier.className='frontier';frontier.id='frontier'; +let prevIn=new Set(); +function setCode(el,name,inCtx){ + const cost=el.querySelector('.cost'); + if(inCtx){cost.textContent='~'+cost.dataset.cost;cost.classList.remove('code');} + else{cost.innerHTML=`${esc(OUTSIDE[name]||'budget_exceeded')}`;} +} +function render(budget){ + const{keep,approx}=replay(budget); + let lastIn=-1; + ORDER.forEach((o,i)=>{ + const inCtx=keep.has(o.name),el=o._el; + el.classList.toggle('in',inCtx);el.classList.toggle('out',!inCtx); + if(inCtx){lastIn=i;if(!prevIn.has(o.name)){el.classList.remove('cross');void el.offsetWidth;el.classList.add('cross');}} + setCode(el,o.name,inCtx); + }); + // slide the frontier to sit just below the last retained row + const anchor=lastIn>=0?ORDER[lastIn]._el:null; + if(anchor&&anchor.nextSibling!==frontier)stack.insertBefore(frontier,anchor.nextSibling); + else if(!anchor&&stack.firstChild!==frontier)stack.insertBefore(frontier,stack.firstChild); + frontier.dataset.tok=budget; + prevIn=keep; + const inN=keep.size,outN=ORDER.length-inN; + $('#in-n').textContent=inN;$('#out-n').textContent=outN; + $('#used').textContent=approx;$('#bud').textContent=budget;$('#bval').textContent=budget; + // MIRRORS context/lens.py replay_verdict: any scoped candidate dropped -> UNVERIFIABLE + const verdict=inNrender(+slider.value)); +render(+slider.value); +""" + + +def render_lens_html(lens: dict) -> str: + env = lens["envelope"] + order = lens["replay"]["order"] + costs = [o["cost"] for o in order] + total = sum(costs) + lens["replay"]["base_tokens"] + built = env["budget"]["token_budget"] + lo = max(1, min(costs) if costs else 1) + hi = max(total, built) + focus = env["focus"]["repo"] + scope = (f"focused on {html.escape(str(focus))}" + if focus else "across the whole workspace") + root = html.escape(str(env["root"])) + data = json.dumps(lens, separators=(",", ":")) + return f""" + +Context Lens — {root} + +
    +
    +

    index · context lens

    +

    What fits in the context budget +MATCH

    +

    Every repo {root} maps, ranked in the order an agent's context would +fill, {scope}. Drag the budget and the frontier slides: repos above it are RETAINED, +the ones below are OMITTED and carry their failure code. The slider replays the exact +selection index context-envelope emits, so what you see is what the agent gets.

    +
    + +
    +
    +
    + + +{built} +
    +
    0/ {built} tokens used
    +
    +RETAINED 0 +OMITTED 0 +
    +
    +
      +
      + +
      +
      receipt sha256  {html.escape(lens['receipt_sha256'])}
      +
      re-check  {html.escape(env['recheck']['command'])}  ·  graph-pack {html.escape(env['recheck']['graph_pack_sha256'][:16])}…
      +
      +
      + +""" diff --git a/client-plugin/server/src/index_graph/viz/manifest.py b/client-plugin/server/src/index_graph/viz/manifest.py new file mode 100644 index 0000000..7f845c2 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/manifest.py @@ -0,0 +1,43 @@ +"""The context manifest: the defined handoff seam for downstream consumers.""" +from __future__ import annotations + +import hashlib +import json +from collections import Counter + + +def _sha(b: bytes) -> str: + return hashlib.sha256(b).hexdigest() + + +def render_manifest(pack: dict, *, artifacts: dict, meta: dict) -> dict: + snapshot = json.dumps(pack, sort_keys=True, separators=(",", ":")).encode("utf-8") + role_counts = Counter( + (rs or ["isolated"])[0] for rs in pack.get("roles", {}).values() + ) + renders = { + key: {"path": path, "sha256": _sha(data)} + for key, (path, data) in artifacts.items() + if key in ("mermaid", "svg", "html") + } + out = { + "schema_version": "1", + "generated": { + "tool": "index", + "version": meta.get("version", ""), + "commit": meta.get("commit"), + "root": meta.get("root", ""), + }, + "graph": { + "node_count": len(pack.get("repos", [])), + "edge_count": len(pack.get("relations", [])), + "roles": dict(role_counts), + "snapshot_sha256": _sha(snapshot), + }, + "renders": renders, + "receipts": {"present": False}, + } + if "context" in artifacts: + path, data = artifacts["context"] + out["context_pack"] = {"path": path, "sha256": _sha(data)} + return out diff --git a/client-plugin/server/src/index_graph/viz/mermaid.py b/client-plugin/server/src/index_graph/viz/mermaid.py new file mode 100644 index 0000000..2eab495 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/mermaid.py @@ -0,0 +1,70 @@ +"""Render the pack to a Mermaid flowchart (Mermaid performs its own layout).""" +from __future__ import annotations + +import re + +from .theme import ROLE_COLOR + +_PRECEDENCE = ("entrypoint", "orchestrator", "hub", "library", "leaf", "isolated") + + +def _primary(roles: list[str]) -> str: + for r in _PRECEDENCE: + if r in roles: + return r + return "isolated" + + +def _id_map(names): + out, used = {}, set() + for name in names: # names arrive in a stable (sorted) order + base = "n_" + re.sub(r"[^0-9A-Za-z]", "_", name) + ident, i = base, 1 + while ident in used: + ident, i = f"{base}_{i}", i + 1 + used.add(ident) + out[name] = ident + return out + + +def render_mermaid(pack: dict, *, include_external: bool = True) -> str: + roles = pack.get("roles", {}) + lines = ["flowchart TD"] + # deterministic node declarations + internal = sorted(r["name"] for r in pack.get("repos", [])) + externals = ( + sorted({r["target_name"] for r in pack.get("relations", []) if r.get("external")}) + if include_external + else [] + ) + ids = _id_map(internal + externals) + node_role: dict[str, str] = {} + for name in internal: + primary = _primary(list(roles.get(name, ()))) + node_role[name] = primary + lines.append(f' {ids[name]}["{name}"]') + for name in externals: + node_role[name] = "external" + lines.append(f' {ids[name]}(("{name}"))') + # edges, deterministic order (skip external edges when include_external=False) + rels = sorted( + pack.get("relations", []), + key=lambda r: (r["from"], (r["target_name"] if r.get("external") else r["to"]), r.get("confidence", "")), + ) + for r in rels: + if r.get("external") and not include_external: + continue + target = r["target_name"] if r.get("external") else r["to"] + kinds = {s["kind"] for s in r.get("signals", []) if s.get("kind")} + conf = r.get("confidence", "low") + if kinds: + label = f'{conf} ({"+".join(sorted(kinds))})' + lines.append(f' {ids[r["from"]]} -->|{label}| {ids[target]}') + else: + lines.append(f' {ids[r["from"]]} -->|{conf}| {ids[target]}') + # classDefs + assignments + for role, color in ROLE_COLOR.items(): + lines.append(f" classDef {role} fill:{color},stroke:#0d1b1c,color:#e9e2d0;") + for name in internal + externals: + lines.append(f" class {ids[name]} {node_role[name]};") + return "\n".join(lines) + "\n" diff --git a/client-plugin/server/src/index_graph/viz/svg.py b/client-plugin/server/src/index_graph/viz/svg.py new file mode 100644 index 0000000..3574513 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/svg.py @@ -0,0 +1,73 @@ +"""Render a LayoutModel to a self-contained SVG network graph.""" +from __future__ import annotations + +import json +from xml.sax.saxutils import escape, quoteattr + +from .layout import LayoutModel +from .theme import THEME, svg_style + + +def _path_d(points: tuple[tuple[float, float], ...]) -> str: + (m, c1, c2, end) = points + return ( + f"M{m[0]:.2f},{m[1]:.2f} " + f"C{c1[0]:.2f},{c1[1]:.2f} {c2[0]:.2f},{c2[1]:.2f} {end[0]:.2f},{end[1]:.2f}" + ) + + +def _edge_svg(edge) -> str: + classes = ["edge", f"edge-{edge.confidence}"] + if edge.external: + classes.append("edge-external") + if edge.back_edge: + classes.append("edge-back") + if getattr(edge, "in_cycle", False): + classes.append("edge-cycle") + return ( + (lambda sig: ( + f'' + ))(quoteattr(json.dumps(list(edge.signals), sort_keys=True))) + if edge.points + else "" + ) + + +def _node_svg(node) -> str: + label = escape(node.name) + node_classes = "node role-" + node.role + (" cycle" if getattr(node, "in_cycle", False) else "") + return ( + f'' + f'' + f'{label}' + f"{label}: {escape(node.role)}" + f"" + ) + + +def render_svg(layout: LayoutModel) -> str: + defs = ( + '' + f'' + ) + style = f"" + bg = f'' + edges = "".join(_edge_svg(e) for e in layout.edges) + nodes = "".join(_node_svg(n) for n in layout.nodes) + return ( + f'' + f"{defs}{style}{bg}{edges}{nodes}" + ) diff --git a/client-plugin/server/src/index_graph/viz/theme.py b/client-plugin/server/src/index_graph/viz/theme.py new file mode 100644 index 0000000..53e6a67 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/theme.py @@ -0,0 +1,65 @@ +"""Project Telos palette + font tokens, shared by every renderer.""" +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True) +class Theme: + bg: str = "#f4f3ef" + ink: str = "#0b0c0e" + accent: str = "#4636e8" + teal: str = "#585c64" + gold: str = "#0b0c0e" + ok: str = "#2f3238" + muted: str = "#585c64" + alert: str = "#d9544d" + hairline: str = "rgba(11,12,14,.14)" + font_body: str = "Arial,Helvetica,sans-serif" + font_mono: str = "ui-monospace,SFMono-Regular,Consolas,monospace" + + +THEME = Theme() + +# role -> fill colour (deterministic mapping; unknown roles fall back to muted) +ROLE_COLOR = { + "entrypoint": THEME.accent, + "orchestrator": THEME.gold, + "hub": THEME.ok, + "library": THEME.teal, + "leaf": THEME.muted, + "isolated": THEME.muted, + "external": THEME.hairline, +} + + +def css_variables() -> str: + t = THEME + return ( + ":root{" + f"--bg:{t.bg};--ink:{t.ink};--accent:{t.accent};--teal:{t.teal};" + f"--gold:{t.gold};--ok:{t.ok};--muted:{t.muted};--hairline:{t.hairline};" + f"--font-body:{t.font_body};--font-mono:{t.font_mono};" + "}" + ) + + +def svg_style() -> str: + t = THEME + roles = "".join( + f".role-{role} rect{{fill:{color};}}" for role, color in ROLE_COLOR.items() + ) + return ( + f"svg{{background-color:{t.bg};}}" + f"text{{font-family:{t.font_mono};fill:{t.ink};}} /* labels are identifiers, use monospace not font-body */" + f"rect{{stroke:{t.hairline};}}" + f"{roles}" + f".edge{{fill:none;stroke:{t.muted};}}" + f".edge-high{{stroke:{t.ok};}}" + f".edge-moderate{{stroke:{t.gold};stroke-dasharray:5 3;}}" + f".edge-low{{stroke:{t.muted};stroke-dasharray:2 3;}}" + f".edge-external{{stroke:{t.hairline};}}" + f".edge-back{{stroke:{t.accent};}}" + f".edge-cycle{{stroke:{t.alert};stroke-width:2.4;stroke-dasharray:none;}}" + f".node.cycle rect{{stroke:{t.alert};stroke-width:2.4;}}" + ) diff --git a/client-plugin/server/src/index_graph/viz/workbench_assets.py b/client-plugin/server/src/index_graph/viz/workbench_assets.py new file mode 100644 index 0000000..8b60387 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/workbench_assets.py @@ -0,0 +1,174 @@ +"""Workbench CSS (string constant embedded into the self-contained HTML).""" +from __future__ import annotations + +WB_CSS = """ +:root{ + --bg:#f4f3ef;--panel:#fbfaf7;--ink:#0b0c0e;--soft:#2f3238;--muted:#565a62; + --hair:rgba(11,12,14,.12);--hair2:rgba(11,12,14,.07); + --iris:#4636e8;--iris-wash:rgba(70,54,232,.08);--iris-line:rgba(70,54,232,.35); + --warn:#8a5a12;--ok:#1f6b45; + --mono:ui-monospace,SFMono-Regular,Consolas,"SF Mono",monospace; + --body:Arial,Helvetica,"Helvetica Neue",sans-serif; + --e:cubic-bezier(.22,1,.36,1); + --z-drop:10;--z-palette:40;--z-toast:50} +*{box-sizing:border-box} +body{margin:0;background:var(--bg);color:var(--ink);font-family:var(--body); + font-size:15px;line-height:1.5;-webkit-font-smoothing:antialiased;overflow:hidden} +button{font:inherit;color:inherit;background:none;border:none;cursor:pointer} +.app{display:grid;grid-template-rows:auto 1fr;height:100vh} + +.top{display:flex;align-items:center;gap:.9rem;padding:.6rem 1rem; + border-bottom:1px solid var(--hair);background:var(--panel)} +.brand{font:600 .8rem/1 var(--mono);letter-spacing:.12em;color:var(--iris); + text-transform:uppercase;white-space:nowrap} +.modes{display:flex;gap:.15rem;background:var(--hair2);border-radius:8px;padding:.18rem} +.mode{font:600 .74rem/1 var(--mono);letter-spacing:.03em;padding:.42em .8em;border-radius:6px; + color:var(--muted);transition:color .15s,background .15s} +.mode[aria-pressed=true]{background:var(--panel);color:var(--ink); + box-shadow:0 1px 2px rgba(11,12,14,.12)} +.mode:focus-visible{outline:2px solid var(--iris);outline-offset:2px} +.searchbtn{display:flex;align-items:center;gap:.6rem;margin-left:auto; + font:.78rem var(--mono);color:var(--muted);border:1px solid var(--hair); + border-radius:8px;padding:.42em .7em;min-width:14rem;justify-content:space-between} +.searchbtn kbd{font:600 .68rem var(--mono);border:1px solid var(--hair); + border-radius:4px;padding:.1em .4em;background:var(--hair2)} +.basketbtn{position:relative;font:600 .74rem var(--mono);border:1px solid var(--hair); + border-radius:8px;padding:.45em .8em;transition:border-color .15s} +.basketbtn[data-n]:not([data-n="0"]){border-color:var(--iris-line);color:var(--iris)} +.basketbtn .n{font-variant-numeric:tabular-nums} + +.main{display:grid;grid-template-columns:1fr 380px;min-height:0} +.stage{overflow:auto;position:relative;min-width:0} +.stage:focus-visible{outline:2px solid var(--iris);outline-offset:-2px} +.view{display:none;padding:1.1rem 1.3rem} +.view.on{display:block} +#view-map.on{display:block;padding:0;height:100%;overflow:hidden} +#map-stage{height:100%;overflow:hidden;position:relative} +#map-stage svg{cursor:grab;display:block;width:100%;height:100%} +#map-stage.grabbing svg{cursor:grabbing} +.maptools{position:absolute;top:.7rem;left:.9rem;display:flex;gap:.4rem;z-index:var(--z-drop)} +.maptools .chip,.chip{font:600 .72rem var(--mono);border:1px solid var(--hair); + background:var(--panel);border-radius:6px;padding:.32em .6em;color:var(--soft)} +.chip[aria-pressed=true]{background:var(--iris);border-color:var(--iris);color:#fff} +.legendbar{position:absolute;bottom:.7rem;left:.9rem;display:flex;gap:.9rem;z-index:var(--z-drop); + font:600 .68rem var(--mono);color:var(--muted);background:var(--panel); + border:1px solid var(--hair);border-radius:6px;padding:.35em .7em} +.lg{display:inline-flex;align-items:center;gap:.35em} +.lg i{width:.65em;height:.65em;border-radius:2px;display:inline-block} + +aside{border-left:1px solid var(--hair);background:var(--panel);overflow:auto; + padding:1rem 1.1rem;font-size:.88rem} +aside h3{font:600 1.05rem/1.2 var(--body);margin:.1rem 0 .1rem;letter-spacing:-.01em; + display:flex;align-items:baseline;gap:.5rem;flex-wrap:wrap} +aside h3 small{font:600 .66rem var(--mono);color:var(--muted);letter-spacing:.08em;text-transform:uppercase} +aside h4{font:600 .7rem var(--mono);letter-spacing:.08em;text-transform:uppercase; + color:var(--muted);margin:1rem 0 .35rem} +.pinrow{margin:.4rem 0 .2rem} +.pin{font:600 .72rem var(--mono);border:1px solid var(--hair);border-radius:6px; + padding:.3em .65em;transition:all .15s} +.pin[aria-pressed=true]{background:var(--iris);border-color:var(--iris);color:#fff} +.kv{font:.76rem var(--mono);color:var(--soft);display:flex;gap:.9rem;flex-wrap:wrap;margin:.2rem 0} +.kv b{color:var(--ink);font-variant-numeric:tabular-nums} +.dep{border:1px solid var(--hair2);border-radius:8px;padding:.4rem .55rem;margin:.3rem 0} +.dep .to{font:600 .85rem var(--body)} +.dep .conf{font:.68rem var(--mono);color:var(--muted);margin-left:.4em} +.dep .cyc{font:600 .66rem var(--mono);color:var(--warn);margin-left:.4em} +.ev{font:.72rem var(--mono);color:var(--muted);margin-top:.15rem} +.ev a,.navlink{color:var(--iris);cursor:pointer;text-decoration:none} +.navlink:hover,.ev a:hover{text-decoration:underline} +.fp{font:.7rem var(--mono);color:var(--muted);word-break:break-all} +.md{font-size:.9rem;line-height:1.55;border-top:1px solid var(--hair2);margin-top:.7rem;padding-top:.7rem} +.md pre{background:var(--hair2);border-radius:6px;padding:.6em;overflow:auto;font-size:.8rem} +.md table{border-collapse:collapse}.md th,.md td{border:1px solid var(--hair);padding:.25em .55em} +.md .wikilink{color:var(--iris);cursor:pointer} +.md img{max-width:100%} +.crumbs{font:.7rem var(--mono);color:var(--muted);margin-bottom:.5rem;min-height:1.1em} +.crumbs a{color:var(--iris);cursor:pointer} + +.cards{display:grid;grid-template-columns:repeat(auto-fit,minmax(210px,1fr));gap:.8rem;margin:.9rem 0} +.stat{border:1px solid var(--hair);border-radius:10px;padding:.7rem .85rem;background:var(--panel)} +.stat .n{font:600 1.5rem/1.1 var(--mono);color:var(--ink);font-variant-numeric:tabular-nums} +.stat .l{font:600 .68rem var(--mono);letter-spacing:.06em;text-transform:uppercase;color:var(--muted)} +.stat.iris .n{color:var(--iris)} +.hero{max-width:72ch} +.hero h2{font:600 1.5rem/1.2 var(--body);letter-spacing:-.02em;margin:.2rem 0 .4rem;text-wrap:balance} +.hero p{color:var(--soft);margin:.3rem 0} +.hero code{font:.85em var(--mono);background:var(--hair2);padding:.06em .35em;border-radius:4px} +.rowlist{list-style:none;margin:.4rem 0;padding:0} +.rowlist li{display:flex;align-items:baseline;gap:.7rem;padding:.5rem .6rem;border-radius:8px; + border:1px solid transparent} +.rowlist li:hover,.rowlist li.cursor{background:var(--iris-wash);border-color:var(--iris-line)} +.rowlist .nm{font-weight:600;cursor:pointer} +.rowlist .mt{font:.72rem var(--mono);color:var(--muted)} +.verdictchip{font:600 .66rem var(--mono);letter-spacing:.05em;padding:.25em .55em; + border-radius:999px;border:1px solid} +.verdictchip.UNVERIFIABLE{color:var(--warn);border-color:rgba(138,90,18,.4)} +.verdictchip.MATCH{color:var(--ok);border-color:rgba(31,107,69,.4)} +.warnrow{font:.78rem var(--mono);color:var(--warn);padding:.2rem 0} +.cycchip{display:inline-block;font:600 .72rem var(--mono);color:var(--warn); + border:1px solid rgba(138,90,18,.4);border-radius:6px;padding:.25em .6em;margin:.15rem .25rem .15rem 0} + +.palette{position:fixed;inset:0;z-index:var(--z-palette);display:none; + background:rgba(11,12,14,.28);backdrop-filter:blur(2px)} +.palette.on{display:block} +.pbox{max-width:620px;margin:9vh auto 0;background:var(--panel);border:1px solid var(--hair); + border-radius:12px;box-shadow:0 24px 60px -24px rgba(11,12,14,.5);overflow:hidden} +.pbox input{width:100%;border:none;outline:none;background:transparent;color:var(--ink); + font:1rem var(--body);padding:.9rem 1.1rem;border-bottom:1px solid var(--hair2)} +.presults{max-height:46vh;overflow:auto;padding:.35rem} +.pr{display:flex;align-items:baseline;gap:.7rem;padding:.5rem .75rem;border-radius:8px;cursor:pointer} +.pr.sel{background:var(--iris-wash)} +.pr .k{font:600 .62rem var(--mono);letter-spacing:.06em;text-transform:uppercase; + color:var(--muted);min-width:3.6rem} +.pr .t{font-weight:600} +.pr .s{font:.72rem var(--mono);color:var(--muted);margin-left:auto;white-space:nowrap} +.phint{font:.68rem var(--mono);color:var(--muted);padding:.5rem .9rem;border-top:1px solid var(--hair2); + display:flex;gap:1rem;flex-wrap:wrap} +.phint kbd{border:1px solid var(--hair);border-radius:4px;padding:.05em .35em;background:var(--hair2)} + +.toast{position:fixed;bottom:1.2rem;left:50%;transform:translate(-50%,.6rem);opacity:0; + z-index:var(--z-toast);background:var(--ink);color:var(--bg);font:600 .78rem var(--mono); + border-radius:8px;padding:.6em 1em;transition:opacity .2s var(--e),transform .2s var(--e); + pointer-events:none;max-width:90vw;word-break:break-all} +.toast.on{opacity:1;transform:translate(-50%,0)} + +.basketbar{border:1px solid var(--iris-line);background:var(--iris-wash);border-radius:10px; + padding:.6rem .8rem;margin:.7rem 0} +.basketbar .bl{font:600 .7rem var(--mono);letter-spacing:.06em;text-transform:uppercase;color:var(--iris)} +.basketbar .items{display:flex;gap:.35rem;flex-wrap:wrap;margin:.4rem 0} +.bitem{font:600 .72rem var(--mono);background:var(--panel);border:1px solid var(--iris-line); + border-radius:6px;padding:.25em .55em;cursor:pointer} +.bitem:hover{text-decoration:line-through} +.basketbar .sum{font:.76rem var(--mono);color:var(--soft)} +.basketbar .sum b{font-variant-numeric:tabular-nums} +.cta{font:600 .74rem var(--mono);background:var(--iris);color:#fff;border-radius:7px; + padding:.45em .8em;margin-top:.35rem} +.cta:focus-visible{outline:2px solid var(--ink);outline-offset:2px} + +/* lens view reuses the stack look */ +.lensrail{display:flex;align-items:center;gap:1rem;margin:.6rem 0 .9rem;flex-wrap:wrap} +.lensrail label{font:600 .74rem var(--mono);color:var(--muted);white-space:nowrap} +.lensrail input[type=range]{flex:1;min-width:12rem;accent-color:var(--iris)} +.lensrail .big{font:600 1.25rem var(--mono);color:var(--iris);font-variant-numeric:tabular-nums} +.item{display:grid;grid-template-columns:1.7rem 1fr auto;gap:.15rem .8rem;padding:.5rem .6rem; + position:relative;border-radius:8px;transition:background .22s var(--e),opacity .22s var(--e)} +.item::before{content:"";position:absolute;left:0;top:6px;bottom:6px;width:3px;border-radius:2px; + background:var(--iris);transform:scaleY(0);transition:transform .22s var(--e)} +.item.in::before{transform:scaleY(1)} +.item.in{background:var(--iris-wash)} +.item.out{opacity:.6} +.item .rank{font:600 .7rem var(--mono);color:var(--muted);text-align:right;font-variant-numeric:tabular-nums} +.item .nm{font-weight:600;cursor:pointer} +.item .cost{font:600 .78rem var(--mono);text-align:right;font-variant-numeric:tabular-nums} +.item .code{color:var(--warn);font-weight:600;font-size:.74rem} +.frontier{height:0;border-top:1.5px dashed var(--iris-line);margin:.1rem .3rem;position:relative; + transition:all .25s var(--e)} +.frontier::after{content:"frontier";position:absolute;right:0;top:-.62rem;font:600 .6rem var(--mono); + color:var(--iris);background:var(--panel);border:1px solid var(--iris-line); + border-radius:999px;padding:.15em .5em} + +footer.receipt{padding:.5rem 1rem;border-top:1px solid var(--hair); + font:.68rem var(--mono);color:var(--muted);white-space:nowrap;overflow:hidden;text-overflow:ellipsis} +@media(max-width:900px){.main{grid-template-columns:1fr}aside{display:none}} +@media(prefers-reduced-motion:reduce){*{transition:none!important;animation:none!important}} +""" diff --git a/client-plugin/server/src/index_graph/viz/workbench_html.py b/client-plugin/server/src/index_graph/viz/workbench_html.py new file mode 100644 index 0000000..4d34324 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/workbench_html.py @@ -0,0 +1,225 @@ +"""Assemble the self-contained workbench document. + +One page, five views over one sealed pack: Overview (the workspace at a +glance + the context basket), Map (the interactive two-layer atlas with a +live context-lens overlay), Docs (the derived knowledge layer), Context Lens +(the budget frontier), Health (cycles, warnings, salience audit, freshness +fingerprints). Command palette on Ctrl/Cmd-K, full state in the URL hash, +zero runtime dependencies, receipt printed in the footer. +""" +from __future__ import annotations + +import html +import json + +from .workbench_assets import WB_CSS +from .workbench_js import WB_JS_CORE +from .workbench_js2 import WB_JS_UX + +_MAP_EXTRA_CSS = ( + "#map-stage .node.lens-out,#map-stage .docnode.lens-out{opacity:.14}" + "#map-stage .node.sel rect{stroke:#4636e8;stroke-width:2.5}" + "#map-stage .docnode.sel rect{stroke:#4636e8;stroke-width:2}" +) + + +def _stat(n, label, cls="") -> str: + return (f'
      {n}
      ' + f'
      {label}
      ') + + +def _overview(wb: dict) -> str: + s = wb["summary"] + roles = " · ".join(f"{k} {v}" for k, v in s["role_counts"].items()) + tops = ", ".join(s["top_salience"][:5]) + return f"""
      +

      {s['repos']} repos, {s['docs']} docs, one verified map.

      +

      Everything below is derived from file:line evidence at {html.escape(wb['root'])} +and sealed under receipt {html.escape(wb['receipt_sha256'][:16])}…. +Press Ctrl+K for anything; pin repos into the context basket and the +exact index context-envelope command is written for you.

      +
      +{_stat(s['repos'], 'repos')}{_stat(s['docs'], 'docs')} +{_stat(s['relations'], 'dependency edges')}{_stat(s['knowledge_edges'], 'knowledge edges')} +{_stat(s['cycles'], 'cycles', 'iris' if s['cycles'] else '')} +{_stat(s['warnings'], 'warnings', 'iris' if s['warnings'] else '')} +
      +
      roles: {html.escape(roles)}
      +
      highest-salience: {html.escape(tops)}
      +
      +

      all repos

      +
        """ + "".join( + f'
      • {html.escape(r["name"])}' + f'in {r["in_degree"]} · out {r["out_degree"]}' + f'{" · " + html.escape(", ".join(r["roles"])) if r["roles"] else ""}
      • ' + for r in wb["repos"]) + "
      " + + +def _docs_view(wb: dict) -> str: + rows = "".join( + f'
    • ' + f'{html.escape(d["title"])}{html.escape(d["id"])}
    • ' + for d in wb["docs"]) + dm = wb["doc_meta"] + budget_line = "" + if dm["bodies_embedded"] < dm["total"]: + budget_line = (f' Page-weight budget: {dm["bodies_embedded"]} of ' + f'{dm["total"]} bodies embedded (most-connected first); ' + f'the list and search below cover all of them.') + return (f'

      The knowledge layer

      ' + f'

      {len(wb["docs"])} docs discovered and linked to the map by ' + f'describes / links-to / mentions edges — derived, not hand-filed. ' + f'Select one to read it with backlinks in the panel.{budget_line}

      ' + f'
        {rows or "
      • no docs discovered
      • "}
      ') + + +def _lens_view(wb: dict) -> str: + order = wb["lens"]["replay"]["order"] + total = sum(o["cost"] for o in order) + wb["lens"]["replay"]["base_tokens"] + built = wb["lens"]["budget"]["token_budget"] + return f"""

      What fits in the context budget +{wb['lens']['verdict']}

      +

      Repos in the exact rank the envelope fills them. Drag the budget: the frontier slides and +rows cross between retained and omitted, replaying the same greedy rule +index context-envelope runs — what you see is what an agent gets. The Map view +can overlay this frontier onto the graph.

      +
      + + +{built} + +0 used incl. base {wb['lens']['replay']['base_tokens']} +
      +
      """ + + +def _health(wb: dict) -> str: + f = wb["freshness"] + cycles = "".join(f'{html.escape(" → ".join(c + [c[0]]))}' + for c in wb["cycles"]) or 'none — the graph is acyclic' + warns = "".join(f'
      {html.escape(w)}
      ' + for w in (wb["warnings"] + wb["knowledge_warnings"])[:40]) \ + or '
      none
      ' + audit = "".join( + f'
      [{html.escape(a.get("kind", ""))}] {html.escape(a.get("node", ""))}: ' + f'{html.escape(a.get("note", ""))}
      ' + for a in wb["salience_audit"]) or '
      no salience-faithfulness warnings
      ' + fps = "".join(f'
    • {html.escape(n)}' + f'{html.escape(h)}
    • ' + for n, h in sorted(f["repos"].items())) + return f"""

      Structural health & freshness

      +

      What one pass can honestly verify: cycles, warnings, and the salience audit are structural +facts of the current graph. The fingerprints below are the baseline a later run diffs +against — a FRESH/STALE verdict needs a prior snapshot, so none is claimed here.

      +

      re-check: {html.escape(f['recheck'])}

      +

      dependency cycles

      {cycles}
      +

      warnings

      {warns} +

      salience audit

      {audit} +

      workspace fingerprint

      +
      root {html.escape(f['root_sha256'])}
      +
        {fps}
      """ + + +def _spine(wb: dict) -> str: + sp = wb["spine"] + if not sp["tools"]: + return (f'

      The operator spine

      ' + f'

      No flagship envelopes were supplied to this build, so nothing is ' + f'claimed about peers. {html.escape(sp["capture"])}.

      ') + ring = "".join( + f'
      {html.escape(e["from"])} ' + f'→ {html.escape(e["to"])}' + f'{html.escape(e["action"])}' + f'
      {html.escape(e["reason"])}
      ' + for e in sp["ring"]) or '
      no next_actions declared
      ' + tools = "".join( + f'
      {html.escape(t["tool"])}' + f'v{html.escape(t["version"])} · {html.escape(t["command"])}' + f'' + f'{html.escape(t["status"])}' + + "".join(f'
      {html.escape(c["name"])} — {html.escape(c["status"])}
      ' + for c in t["checks"]) + "
      " + for t in sp["tools"]) + peers = "".join( + f'
      {html.escape(s["peer"])}' + f'{html.escape(s["surface"])} · {s["entries"]} entries' + f'
      inspect: {html.escape(s["inspect"])}
      ' + for s in sp["peer_surfaces"]) or '
      none detected under .telos/
      ' + skipped = "".join( + f'
      {html.escape(s["file"])}: {html.escape(s["reason"])}
      ' + for s in sp["skipped"]) + return (f'

      The operator spine

      ' + f'

      The flagships are aware of each other by protocol: every doctor/status ' + f'envelope declares next_actions handing off to a peer. This view ' + f'renders the captured envelopes and the awareness ring they form — receipts ' + f'from the tools themselves, not this page’s opinion.

      ' + f'

      awareness ring (who hands off to whom)

      {ring}' + f'

      captured envelopes

      {tools}' + f'

      peer surfaces on disk

      {peers}' + + (f'

      skipped inputs (fail closed)

      {skipped}' if skipped else "")) + + +def render_workbench_html(wb: dict) -> str: + svg = wb["svg"] + data = {k: v for k, v in wb.items() if k != "svg"} + blob = json.dumps(data, sort_keys=True, separators=(",", ":")).replace("<", "\\u003c") + root = html.escape(str(wb["root"])) + return f""" + +index workbench — {root} + +
      +
      +index · workbench + + + +
      +
      +
      +
      {_overview(wb)}
      +
      +
      + + + +
      +
      +in budget (overlay) +click: inspectdouble-click: neighborhood +wheel: zoom +map draws {wb['doc_meta']['map_docs']} of {wb['doc_meta']['total']} docs · internal deps only ({wb['summary']['external_relations']} external edges counted) +
      +{svg}
      +
      {_docs_view(wb)}
      +
      {_lens_view(wb)}
      +
      {_health(wb)}
      +
      {_spine(wb)}
      +
      + +
      +
      + +
      +
      receipt sha256 {html.escape(wb['receipt_sha256'])} · root {root} · +re-derive: index workbench --root . --out workbench.html
      + +""" diff --git a/client-plugin/server/src/index_graph/viz/workbench_js.py b/client-plugin/server/src/index_graph/viz/workbench_js.py new file mode 100644 index 0000000..107b49c --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/workbench_js.py @@ -0,0 +1,163 @@ +"""Workbench JS part 1: state, routing, detail panel, basket, lens replay. + +Everything renders from the sealed DATA pack; every visual state is a replay +of real numbers (lens costs, evidence lines, fingerprints). URL hash carries +the full UI state (mode/selection/budget/basket) so any view is deep-linkable +and restorable offline. +""" +from __future__ import annotations + +WB_JS_CORE = r""" +const D=DATA,$=s=>document.querySelector(s),$$=s=>[...document.querySelectorAll(s)]; +const esc=s=>String(s).replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); +const REPO={};D.repos.forEach(r=>REPO[r.name]=r); +const DOC={};D.docs.forEach(d=>DOC[d.id]=d); +const ORDER=D.lens.replay.order,BASE=D.lens.replay.base_tokens; +const COST={};ORDER.forEach(o=>COST[o.name]=o.cost); +// backlinks are a projection of knowledge_edges — rebuilt here instead of +// shipped twice (a 6MB saving on large workspaces) +const BL={};(D.knowledge_edges||[]).forEach(e=>{ + (BL[e.to]=BL[e.to]||[]).push({from:e.from,type:e.type});}); +const S={mode:'overview',sel:null,budget:D.lens.budget.token_budget,basket:[],trail:[]}; + +/* ---------- hash state (deep-linkable, offline) ---------- */ +function writeHash(){ + const h='#m='+S.mode+(S.sel?'&s='+encodeURIComponent(S.sel.kind+':'+S.sel.id):'') + +'&b='+S.budget+(S.basket.length?'&k='+S.basket.map(encodeURIComponent).join(','):''); + if(location.hash!==h)history.replaceState(null,'',h); +} +function readHash(){ + const q=new URLSearchParams(location.hash.slice(1).replace(/&/g,'&')); + const m=q.get('m');if(m&&$('#mode-'+m))S.mode=m; + const b=parseInt(q.get('b'),10);if(b>0)S.budget=b; + const k=q.get('k');if(k)S.basket=k.split(',').map(decodeURIComponent).filter(n=>REPO[n]); + const s=q.get('s');if(s){const i=s.indexOf(':');const kind=s.slice(0,i),id=decodeURIComponent(s.slice(i+1)); + if((kind==='repo'&&REPO[id])||(kind==='doc'&&DOC[id]))S.sel={kind,id};} +} + +/* ---------- toast ---------- */ +let toastT=null; +function toast(msg){const t=$('#toast');t.textContent=msg;t.classList.add('on'); + clearTimeout(toastT);toastT=setTimeout(()=>t.classList.remove('on'),2600);} +function copy(text,label){ + (navigator.clipboard?navigator.clipboard.writeText(text):Promise.reject()) + .then(()=>toast('copied: '+label)).catch(()=>toast(text)); +} + +/* ---------- modes ---------- */ +function setMode(m){S.mode=m; + $$('.mode').forEach(b=>b.setAttribute('aria-pressed',String(b.id==='mode-'+m))); + $$('.view').forEach(v=>v.classList.toggle('on',v.id==='view-'+m)); + if(m==='map')fitMap(); + writeHash(); +} + +/* ---------- selection + detail panel ---------- */ +function pushTrail(sel){const last=S.trail[S.trail.length-1]; + if(!last||last.id!==sel.id)S.trail.push(sel); + $('#crumbs').innerHTML=S.trail.slice(-6).map((n,i)=>`${esc(n.id.split('/').pop())}`).join(' › '); + $$('#crumbs a').forEach(a=>a.onclick=()=>{const n=S.trail[+a.dataset.i];if(n)select(n.kind,n.id);}); +} +function select(kind,id){S.sel={kind,id};pushTrail(S.sel); + if(kind==='repo')detailRepo(id);else detailDoc(id); + highlightMap();writeHash(); +} +function evLine(e){ + return `
      ${esc(e.kind)} · ${esc(e.file)}${e.line?':'+e.line:''}
      `; +} +function detailRepo(name){const r=REPO[name];if(!r)return; + const inB=S.basket.includes(name); + const deps=(r.depends_on||[]).map(d=>`
      ${esc(d.to)}`+ + `${esc(d.confidence)}${d.in_cycle?'cycle':''}`+ + (d.evidence||[]).map(evLine).join('')+`
      `).join('')||'
      none
      '; + const docs=(r.documented_by||[]).map(id=>``).join('')||'
      none
      '; + const back=(BL[name]||[]).map(b=>``).join('')||'
      none
      '; + $('#detail').innerHTML=`

      ${esc(name)} repo

      `+ + `
      roles ${esc((r.roles||[]).join(', ')||'none')}`+ + `in ${r.in_degree}out ${r.out_degree}`+ + `~${COST[name]||'?'} tok
      `+ + (r.description?`

      ${esc(r.description)}

      `:'')+ + `
      `+ + `

      depends on

      ${deps}

      documented by

      ${docs}

      linked from

      ${back}`+ + `

      freshness fingerprint

      ${esc(r.fingerprint||'n/a')}
      `; + $('#pinbtn').onclick=()=>{togglePin(name);detailRepo(name);}; + wireNav(); +} +function detailDoc(id){const d=DOC[id];if(!d)return; + const back=(BL[id]||[]).map(b=>``).join('')||'
      none
      '; + const body=D.doc_html[id]||('body not embedded (page-weight budget: '+ + D.doc_meta.bodies_embedded+' of '+D.doc_meta.total+' bodies). Open the file at '+ + esc(id)+' or rebuild with --max-doc-bodies.'); + $('#detail').innerHTML=`

      ${esc(d.title)} doc

      `+ + `
      ${esc(id)}
      `+ + `

      linked from

      ${back}`+ + `
      ${body}
      `; + wireNav(); +} +function wireNav(){$$('#detail .navlink,#detail .wikilink').forEach(a=>a.onclick=ev=>{ + ev.preventDefault(); + if(a.dataset.atlasTarget){const t=resolveWiki(a.dataset.atlasTarget);if(t)select(t.kind,t.id);return;} + select(a.dataset.kind,a.dataset.id);});} +const norm=s=>String(s).trim().toLowerCase().replace(/_/g,'-').replace(/ /g,'-'); +const WIKI={};D.repos.forEach(r=>{WIKI[norm(r.name)]={kind:'repo',id:r.name};}); +D.docs.forEach(d=>{[d.title,d.id.split('/').pop().replace(/\.[^.]+$/,'')] + .forEach(c=>{if(!(norm(c)in WIKI))WIKI[norm(c)]={kind:'doc',id:d.id};});}); +function resolveWiki(t){return WIKI[norm(t)];} + +/* ---------- context basket ---------- */ +function togglePin(name){const i=S.basket.indexOf(name); + if(i>=0)S.basket.splice(i,1);else S.basket.push(name); + renderBasket();writeHash(); +} +function basketTokens(){return S.basket.reduce((a,n)=>a+(COST[n]||0),BASE);} +function envelopeCmd(){ + return 'index context-envelope --root . --budget '+Math.max(basketTokens(),1) + +(S.basket.length===1?' --focus '+S.basket[0]:'')+' --json'; +} +function renderBasket(){ + $('#basket-n').textContent=S.basket.length; + $('#basketbtn').dataset.n=S.basket.length; + const bar=$('#basketbar'); + if(!S.basket.length){bar.innerHTML='context basket'+ + '
      Pin repos from the detail panel (or press p) to assemble a working set; the exact envelope command is generated for you.
      ';return;} + bar.innerHTML=`context basket`+ + `
      ${S.basket.map(n=>``).join('')}
      `+ + `
      ~${basketTokens()} tokens incl. base ${BASE} · ${S.basket.length} repos
      `+ + ``; + $$('#basketbar .bitem').forEach(b=>b.onclick=()=>{togglePin(b.dataset.n); + if(S.sel&&S.sel.kind==='repo')detailRepo(S.sel.id);}); + $('#copy-env').onclick=()=>copy(envelopeCmd(),'envelope command'); +} + +/* ---------- lens (budget replay — mirrors context/lens.py) ---------- */ +function replay(budget){const keep=new Set();let approx=BASE; + for(const o of ORDER){if(approx+o.cost<=budget||keep.size===0){keep.add(o.name);approx+=o.cost;}} + return{keep,approx};} +function renderLens(){ + const{keep,approx}=replay(S.budget); + let lastIn=-1; + ORDER.forEach((o,i)=>{const el=o._el;if(!el)return; + const inC=keep.has(o.name);el.classList.toggle('in',inC);el.classList.toggle('out',!inC); + if(inC)lastIn=i; + el.querySelector('.cost').innerHTML=inC?('~'+o.cost):'budget_exceeded';}); + const fr=$('#frontier'),stack=$('#lens-stack'); + const anchor=lastIn>=0?ORDER[lastIn]._el:null; + if(anchor&&anchor.nextSibling!==fr)stack.insertBefore(fr,anchor.nextSibling); + $('#lens-used').textContent=approx;$('#lens-bval').textContent=S.budget; + const v=keep.size{const li=document.createElement('div');li.className='item'; + li.innerHTML=`${String(i+1).padStart(2,'0')}`+ + `${esc(o.name)}`+ + `~${o.cost}`; + li.querySelector('.nm').onclick=()=>select('repo',o.name); + stack.appendChild(li);o._el=li;}); + const fr=document.createElement('div');fr.className='frontier';fr.id='frontier';stack.appendChild(fr); + const sl=$('#lens-slider');sl.value=S.budget; + sl.oninput=()=>{S.budget=+sl.value;renderLens();writeHash();}; +} +""" diff --git a/client-plugin/server/src/index_graph/viz/workbench_js2.py b/client-plugin/server/src/index_graph/viz/workbench_js2.py new file mode 100644 index 0000000..54102e9 --- /dev/null +++ b/client-plugin/server/src/index_graph/viz/workbench_js2.py @@ -0,0 +1,126 @@ +"""Workbench JS part 2: map interactions, command palette, keyboard, boot.""" +from __future__ import annotations + +WB_JS_UX = r""" +/* ---------- map (pan/zoom/select/focus + lens overlay) ---------- */ +let view={k:1,tx:0,ty:0}; +function applyView(){const vp=$('#map-stage #viewport'); + if(vp)vp.setAttribute('transform',`translate(${view.tx},${view.ty}) scale(${view.k})`);} +function fitMap(){view={k:1,tx:0,ty:0};applyView();} +function svgPt(svg,cx,cy){const r=svg.getBoundingClientRect(),vb=svg.viewBox.baseVal; + return{x:(cx-r.left)/r.width*vb.width,y:(cy-r.top)/r.height*vb.height};} +function wireMap(){const stage=$('#map-stage'),svg=stage&&stage.querySelector('svg');if(!svg)return; + svg.removeAttribute('width');svg.removeAttribute('height'); + svg.addEventListener('wheel',ev=>{ev.preventDefault(); + const p=svgPt(svg,ev.clientX,ev.clientY),f=ev.deltaY<0?1.12:1/1.12, + nk=Math.min(9,Math.max(.15,view.k*f)); + view.tx=p.x-(p.x-view.tx)*(nk/view.k);view.ty=p.y-(p.y-view.ty)*(nk/view.k);view.k=nk;applyView();},{passive:false}); + let drag=null; + svg.addEventListener('pointerdown',ev=>{drag={x:ev.clientX,y:ev.clientY,tx:view.tx,ty:view.ty}; + stage.classList.add('grabbing');svg.setPointerCapture(ev.pointerId);}); + svg.addEventListener('pointermove',ev=>{if(!drag)return; + const r=svg.getBoundingClientRect(),vb=svg.viewBox.baseVal; + view.tx=drag.tx+(ev.clientX-drag.x)*vb.width/r.width; + view.ty=drag.ty+(ev.clientY-drag.y)*vb.height/r.height;applyView();}); + svg.addEventListener('pointerup',()=>{drag=null;stage.classList.remove('grabbing');}); + $$('#map-stage .node').forEach(g=>{g.style.cursor='pointer'; + g.addEventListener('click',()=>select('repo',g.dataset.name)); + g.addEventListener('dblclick',()=>focusHood('repo',g.dataset.name));}); + $$('#map-stage .docnode').forEach(g=>{g.style.cursor='pointer'; + g.addEventListener('click',()=>select('doc',g.dataset.doc));}); + $('#zoom-fit').onclick=fitMap; + $('#focus-clear').onclick=()=>{$$('#map-stage .dim').forEach(e=>e.classList.remove('dim'));}; + $('#toggle-lens-overlay').onclick=e=>{const on=e.currentTarget.getAttribute('aria-pressed')==='true'; + e.currentTarget.setAttribute('aria-pressed',String(!on));highlightMap();}; +} +function focusHood(kind,id){const keep=new Set([kind+':'+id]); + D.repos.forEach(r=>{(r.depends_on||[]).forEach(d=>{ + if(r.name===id)keep.add('repo:'+d.to); + if(d.to===id)keep.add('repo:'+r.name);});}); + (D.knowledge_edges||[]).forEach(e=>{ + if(kind==='doc'&&e.from===id)keep.add(e.to_kind+':'+e.to); + if(e.to===id&&e.to_kind===kind)keep.add('doc:'+e.from);}); + $$('#map-stage .node').forEach(g=>g.classList.toggle('dim',!keep.has('repo:'+g.dataset.name))); + $$('#map-stage .docnode').forEach(g=>g.classList.toggle('dim',!keep.has('doc:'+g.dataset.doc))); + $$('#map-stage .edge').forEach(p=>p.classList.toggle('dim', + !(keep.has('repo:'+p.dataset.from)&&keep.has('repo:'+p.dataset.to)))); + $$('#map-stage .kedge').forEach(l=>l.classList.toggle('dim', + !(keep.has('doc:'+l.dataset.from)&&(keep.has('repo:'+l.dataset.to)||keep.has('doc:'+l.dataset.to))))); +} +function highlightMap(keepSet){ + // lens overlay: when ON, repos outside the current budget replay get dimmed + const btn=$('#toggle-lens-overlay'); + const on=btn&&btn.getAttribute('aria-pressed')==='true'; + const keep=keepSet||replay(S.budget).keep; + $$('#map-stage .node').forEach(g=>{ + g.classList.toggle('lens-out',on&&!keep.has(g.dataset.name)); + g.classList.toggle('sel',!!S.sel&&S.sel.kind==='repo'&&S.sel.id===g.dataset.name);}); + $$('#map-stage .docnode').forEach(g=> + g.classList.toggle('sel',!!S.sel&&S.sel.kind==='doc'&&S.sel.id===g.dataset.doc)); +} + +/* ---------- command palette (ctrl/cmd-K) ---------- */ +const ACTIONS=[ + {k:'mode',t:'Go to Overview',run:()=>setMode('overview')}, + {k:'mode',t:'Go to Map',run:()=>setMode('map')}, + {k:'mode',t:'Go to Docs',run:()=>setMode('docs')}, + {k:'mode',t:'Go to Context Lens',run:()=>setMode('lens')}, + {k:'mode',t:'Go to Health',run:()=>setMode('health')}, + {k:'mode',t:'Go to Spine (flagship interop)',run:()=>setMode('spine')}, + {k:'copy',t:'Copy envelope command (basket)',run:()=>copy(envelopeCmd(),'envelope command')}, + {k:'copy',t:'Copy re-check: freshness',run:()=>copy(D.freshness.recheck,'freshness re-check')}, + {k:'copy',t:'Copy receipt sha256',run:()=>copy(D.receipt_sha256,'receipt sha256')}, +]; +let pIdx=0,pHits=[]; +function pOpen(){$('#palette').classList.add('on');const i=$('#pinput');i.value='';pQuery('');i.focus();} +function pClose(){$('#palette').classList.remove('on');} +function fuzzy(q,s){q=q.toLowerCase();s=s.toLowerCase();let i=0; + for(const c of s){if(c===q[i])i++;if(i===q.length)return true;}return q.length===0;} +function pQuery(q){ + const hits=[]; + ACTIONS.forEach(a=>{if(fuzzy(q,a.t))hits.push({k:a.k,t:a.t,s:'action',run:a.run});}); + D.repos.forEach(r=>{if(fuzzy(q,r.name))hits.push({k:'repo',t:r.name, + s:(r.roles||[]).join('/')||'repo',run:()=>{select('repo',r.name);}});}); + D.docs.forEach(d=>{if(fuzzy(q,d.title)||fuzzy(q,d.id))hits.push({k:'doc',t:d.title, + s:d.id,run:()=>{setMode('docs');select('doc',d.id);}});}); + pHits=hits.slice(0,40);pIdx=0; + $('#presults').innerHTML=pHits.map((h,i)=> + `
      ${esc(h.k)}`+ + `${esc(h.t)}${esc(h.s)}
      `).join('') + ||'
      no matches
      '; + $$('#presults .pr').forEach(el=>el.onclick=()=>{const h=pHits[+el.dataset.i];if(h){pClose();h.run();}}); +} +function pMove(d){if(!pHits.length)return;pIdx=(pIdx+d+pHits.length)%pHits.length; + $$('#presults .pr').forEach((el,i)=>el.classList.toggle('sel',i===pIdx)); + const el=$$('#presults .pr')[pIdx];if(el)el.scrollIntoView({block:'nearest'});} + +/* ---------- keyboard ---------- */ +document.addEventListener('keydown',ev=>{ + const pal=$('#palette').classList.contains('on'); + if((ev.ctrlKey||ev.metaKey)&&ev.key.toLowerCase()==='k'){ev.preventDefault();pal?pClose():pOpen();return;} + if(pal){ + if(ev.key==='Escape'){pClose();} + else if(ev.key==='ArrowDown'){ev.preventDefault();pMove(1);} + else if(ev.key==='ArrowUp'){ev.preventDefault();pMove(-1);} + else if(ev.key==='Enter'){const h=pHits[pIdx];if(h){pClose();h.run();}} + return;} + if(ev.target.tagName==='INPUT')return; + const m={'1':'overview','2':'map','3':'docs','4':'lens','5':'health','6':'spine'}[ev.key]; + if(m){setMode(m);return;} + if(ev.key==='/'){ev.preventDefault();pOpen();return;} + if(ev.key==='p'&&S.sel&&S.sel.kind==='repo'){togglePin(S.sel.id);detailRepo(S.sel.id);return;} + if(ev.key==='Escape'){$$('#map-stage .dim').forEach(e=>e.classList.remove('dim'));} +}); +$('#pinput')&&($('#pinput').oninput=e=>pQuery(e.target.value)); +$('#palette').addEventListener('click',ev=>{if(ev.target.id==='palette')pClose();}); + +/* ---------- boot ---------- */ +readHash(); +$$('.mode').forEach(b=>b.onclick=()=>setMode(b.id.slice(5))); +$('#searchbtn').onclick=pOpen; +$('#basketbtn').onclick=()=>{setMode('overview'); + $('#basketbar').scrollIntoView({block:'center'});}; +buildLens();renderLens();renderBasket();wireMap(); +setMode(S.mode); +if(S.sel)select(S.sel.kind,S.sel.id); +""" diff --git a/client-plugin/server/src/index_graph/wiki/__init__.py b/client-plugin/server/src/index_graph/wiki/__init__.py new file mode 100644 index 0000000..9372b89 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/__init__.py @@ -0,0 +1,19 @@ +"""The verified wiki: single-repo pages derived from the real module graph, +joined with authored docs, sealed with per-page hashes, pinned to a commit, +and re-checkable with `index wiki --verify` (MATCH/DRIFT/UNVERIFIABLE).""" +from __future__ import annotations + +from .pack import CLUSTER_THRESHOLD, build_wiki_pack +from .html import render_wiki_html +from .seal import (WIKI_SCHEMA, extract_embedded_pack, load_artifact, + run_verify, verify_wiki) +from .cli import CloneError, add_wiki_parser, clone_repo, cmd_wiki +from . import serve +from .serve import make_server, parse_route, serve_forever + +__all__ = [ + "WIKI_SCHEMA", "CLUSTER_THRESHOLD", "build_wiki_pack", "render_wiki_html", + "verify_wiki", "run_verify", "load_artifact", "extract_embedded_pack", + "add_wiki_parser", "cmd_wiki", "CloneError", "clone_repo", + "serve", "make_server", "parse_route", "serve_forever", +] diff --git a/client-plugin/server/src/index_graph/wiki/cli.py b/client-plugin/server/src/index_graph/wiki/cli.py new file mode 100644 index 0000000..2bd28f1 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/cli.py @@ -0,0 +1,121 @@ +"""CLI face for `index wiki`: generate the sealed artifact, or verify one. + +A `source` positional accepts a git URL or a local path: a URL is shallow-cloned +to a temp dir, run, and cleaned up, so `index wiki https://github.com/org/repo` +gives the paste-a-repo experience without a server. A local path (or --root) +reads in place. The clone is a normal git clone: nothing is fetched that a plain +`git clone --depth 1` would not fetch, and the temp dir is always removed. +""" +from __future__ import annotations + +import json +import shutil +import subprocess +import tempfile +from pathlib import Path + +_URL_SCHEMES = ("http://", "https://", "git://", "ssh://", "file://") + + +class CloneError(RuntimeError): + """A shallow clone could not be completed. Carries a plain reason string so + callers (the CLI, the on-demand server) can surface it without a traceback.""" + + +def clone_repo(source: str, dest: Path) -> None: + """Shallow-clone `source` into `dest`. Raises CloneError with a plain message + on failure. Shared by `index wiki` and `index serve` so there is one clone + code path, not two.""" + try: + subprocess.run(["git", "clone", "--depth", "1", source, str(dest)], + check=True, capture_output=True, text=True) + except (subprocess.CalledProcessError, OSError) as exc: + raise CloneError(f"could not clone {source}: {exc}") from exc + + +def add_wiki_parser(sub) -> None: + w = sub.add_parser( + "wiki", + help="Single-repo verified wiki: pages derived from the module graph, " + "sealed and commit-pinned; --verify re-checks a sealed artifact.") + w.add_argument("source", nargs="?", default=None, + help="a git URL (shallow-cloned then removed) or a local path " + "to derive the wiki from; overrides --root when given") + w.add_argument("--root", type=Path, default=Path.cwd(), + help="the single repo to derive the wiki from") + w.add_argument("--out", default=None, help="write the artifact to this path") + w.add_argument("--format", choices=["html", "json"], default="html") + w.add_argument("--verify", type=Path, default=None, + help="verify a sealed wiki artifact against the current tree " + "(MATCH/DRIFT/UNVERIFIABLE, exit 0/1/2)") + w.add_argument("--json", action="store_true", + help="with --verify, emit the verification report as JSON") + + +def _emit_report(report: dict, as_json: bool) -> int: + from .seal import VERDICT_EXIT + if as_json: + print(json.dumps(report, indent=2, sort_keys=True)) + else: + print(f"verdict={report['verdict']} pages={report['pages_checked']} " + f"edges={report['edges_checked']}") + for finding in report["findings"]: + print(f" [{finding['rule']}] {finding['detail']}") + return VERDICT_EXIT[report["verdict"]] + + +def _is_url(source: str) -> bool: + return source.startswith(_URL_SCHEMES) or source.startswith("git@") + + +def _clone(source: str) -> Path: + dest = Path(tempfile.mkdtemp(prefix="index-wiki-")) + try: + clone_repo(source, dest) + except CloneError as exc: + shutil.rmtree(dest, ignore_errors=True) + raise SystemExit(str(exc)) from exc + return dest + + +def _resolve_source(source: str) -> tuple[Path, Path | None]: + """Return (root, tempdir-to-clean). A URL is cloned; a local path is read in + place; anything else is rejected rather than silently treated as a path.""" + if _is_url(source): + clone = _clone(source) + return clone, clone + path = Path(source) + if path.is_dir(): + return path.resolve(), None + raise SystemExit(f"source is not a git URL or an existing directory: {source}") + + +def cmd_wiki(args) -> int: + if args.verify is not None: + from .seal import run_verify + return _emit_report(run_verify(args.verify, args.root.resolve()), args.json) + root, tmp = (_resolve_source(args.source) if args.source + else (args.root.resolve(), None)) + try: + if not root.is_dir(): + raise SystemExit(f"root not found: {root}") + return _generate(root, args) + finally: + if tmp is not None: + shutil.rmtree(tmp, ignore_errors=True) + + +def _generate(root: Path, args) -> int: + from .pack import build_wiki_pack + pack = build_wiki_pack(root) + if args.format == "json": + text = json.dumps(pack, indent=2, sort_keys=True) + else: + from .html import render_wiki_html + text = render_wiki_html(pack) + if args.out: + Path(args.out).write_text(text, encoding="utf-8") + print(f"wrote {args.out}") + else: + print(text) + return 0 diff --git a/client-plugin/server/src/index_graph/wiki/html.py b/client-plugin/server/src/index_graph/wiki/html.py new file mode 100644 index 0000000..f6adfd3 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/html.py @@ -0,0 +1,191 @@ +"""Render the wiki pack to one self-contained HTML file with client-side nav. + +Follows the atlas posture: no external scripts, styles, or fonts; every +untrusted string is escaped server-side; doc markdown arrives pre-rendered +by the escaping-safe renderer; the sealed pack is embedded as a JSON data +island (with ``<`` escaped) so ``index wiki --verify`` can read it back. +""" +from __future__ import annotations + +import json +from html import escape + +from ..viz.theme import css_variables +from .seal import EMBED_CLOSE, EMBED_OPEN + +_CSS = ( + "*{box-sizing:border-box}body{margin:0;background:var(--bg);color:var(--ink);" + "font-family:var(--font-body)}" + "header{padding:.7rem 1rem;border-bottom:1px solid var(--hairline);" + "font-family:var(--font-mono);font-size:.85rem}" + "main{display:grid;grid-template-columns:260px 1fr;min-height:100vh}" + "nav{border-right:1px solid var(--hairline);padding:1rem;font-family:var(--font-mono);" + "font-size:.8rem;overflow:auto}" + "nav a{display:block;color:var(--ink);text-decoration:none;padding:.15rem .3rem;" + "border-radius:4px;word-break:break-all}" + "nav a.active{background:var(--accent);color:var(--bg)}" + "#pages{padding:1rem 1.5rem;overflow:auto}" + "section.page footer{margin-top:1.2rem;padding-top:.5rem;border-top:1px solid " + "var(--hairline);font-family:var(--font-mono);font-size:.72rem;color:var(--muted)}" + "table{border-collapse:collapse}th,td{border:1px solid var(--hairline);" + "padding:.2em .6em;text-align:left}" + ".ev{font-family:var(--font-mono);font-size:.78em;color:var(--muted)}" + ".label{font-family:var(--font-mono);font-size:.78rem;color:var(--muted)}" + "article{border-top:1px solid var(--hairline);margin-top:1rem;padding-top:.5rem}" + ".md pre{background:rgba(255,255,255,.55);border:1px solid var(--hairline);" + "padding:.5em;overflow:auto}" + "svg{max-width:100%;height:auto}" + "@media(max-width:820px){main{grid-template-columns:1fr}nav{border-right:none}}" +) + +_JS = ( + "const links=[...document.querySelectorAll('[data-target]')];" + "function show(t){document.querySelectorAll('.page').forEach(" + "s=>{s.hidden=(s.id!==t);});" + "links.forEach(a=>{a.classList.toggle('active',a.dataset.target===t);});}" + "links.forEach(a=>a.addEventListener('click'," + "ev=>{ev.preventDefault();show(a.dataset.target);}));" + "show('page-0');" +) + + +def _loc(e: dict) -> str: + line = e.get("line") + return f"{e['file']}:{line}" if line else str(e["file"]) + + +def _footer(page: dict) -> str: + n = page["boundary"]["evidence_count"] + return ("
      structure derived from the dependency graph; no generated " + f"prose; evidence shown: {n} file:line reference(s)
      ") + + +def _overview_section(p: dict) -> str: + rows = [("repo", p["repo"]), ("commit", p["commit"]), + ("ecosystems", ", ".join(p["ecosystems"]) or "none"), + ("modules", str(p["module_count"])), + ("internal edges", str(p["internal_edge_count"])), + ("cycles", str(p["cycle_count"])), + ("entry points (no internal importer)", + ", ".join(p["entry_points"]) or "none")] + table = "".join(f"{escape(k)}{escape(v)}" + for k, v in rows) + inventory = "".join(f"
    • {escape(d)}
    • " + for d in p["doc_paths"]) or "
    • none
    • " + cov = p["coverage"] + coverage = ("complete" if cov["complete"] else + f"{len(cov['parse_errors'])} unparsed file(s), " + f"{len(cov['dynamic_imports'])} dynamic import(s) not followed") + return (f"

      {escape(p['title'])}

      {table}
      " + f"

      doc inventory ({p['doc_count']})

        {inventory}
      " + f"

      graph coverage: {escape(coverage)}

      ") + + +def _edge_list(items: list[dict], key: str) -> str: + lis = "".join( + f"
    • {escape(e[key])} " + f"{escape(_loc(e))}
    • " for e in items) + return f"
        {lis}
      " if lis else "

      none

      " + + +def _module_section(p: dict) -> str: + cycles = "".join(f"
    • {escape(' -> '.join(c))}
    • " + for c in p["cycles"]) + return (f"

      {escape(p['title'])}

      " + f"

      {escape(p['path'])} ({escape(p['language'])})

      " + f"

      imports

      {_edge_list(p['imports'], 'to')}" + f"

      dependents

      {_edge_list(p['dependents'], 'from')}" + + (f"

      cycle membership

        {cycles}
      " if cycles else "")) + + +def _via_list(groups: list[dict], key: str) -> str: + out = [] + for group in groups: + vias = "".join( + f"
    • {escape(v['from'])} -> {escape(v['to'])} " + f"{escape(_loc(v))}
    • " for v in group["via"]) + out.append(f"
    • {escape(group[key])}
        {vias}
    • ") + return f"
        {''.join(out)}
      " if out else "

      none

      " + + +def _package_section(p: dict) -> str: + mods = "".join(f"
    • {escape(m)}
    • " for m in p["modules"]) + return (f"

      {escape(p['title'])}

      " + f"

      modules ({len(p['modules'])})

        {mods}
      " + f"

      imports

      {_via_list(p['imports'], 'to')}" + f"

      dependents

      {_via_list(p['dependents'], 'from')}") + + +def _symbol_section(p: dict) -> str: + d = p["definition"] + defloc = f"{d['file']}:{d['line']}" + callers = "".join( + f"
    • {escape(e['from_symbol'])} " + f"{escape(_loc(e))}
    • " for e in p["callers"]) + callees = "".join( + f"
    • {escape(e['to_symbol'])} " + f"{escape(_loc(e))}
    • " for e in p["callees"]) + unresolved = "".join( + f"
    • {escape(e['to_name'])} " + f"{escape(_loc(e))} " + f"{escape(e['reason'])}
    • " + for e in p["unresolved_calls"]) + parent = f" in {escape(p['parent'])}" if p.get("parent") else "" + return (f"

      {escape(p['title'])}

      " + f"

      {escape(p['symbol_kind'])} " + f"{escape(p['symbol'])}{parent}, defined at " + f"{escape(defloc)}

      " + f"

      called by ({len(p['callers'])})

      " + + (f"
        {callers}
      " if callers else "

      none

      ") + + f"

      calls ({len(p['callees'])})

      " + + (f"
        {callees}
      " if callees else "

      none

      ") + + (f"

      unresolved references ({len(p['unresolved_calls'])})

      " + f"
        {unresolved}
      " if unresolved else "")) + + +def _architecture_section(p: dict) -> str: + return (f"

      {escape(p['title'])}

      " + f"

      rendered from the real {escape(p['granularity'])} " + f"graph, {p['edge_count']} evidence-backed edge(s); never inferred

      " + f"{p['svg']}" + f"
      mermaid source" + f"
      {escape(p['mermaid'])}
      ") + + +def _docs_section(p: dict) -> str: + articles = "".join( + f"

      {escape(d['title'])} {escape(d['path'])}

      " + f"
      {d['html']}
      " for d in p["docs"]) + return (f"

      {escape(p['title'])}

      " + "

      authored by humans; joined in verbatim and " + "rendered offline, never rewritten

      " + (articles or "

      none

      ")) + + +_SECTIONS = {"overview": _overview_section, "module": _module_section, + "package": _package_section, "architecture": _architecture_section, + "symbol": _symbol_section, "docs": _docs_section} + + +def render_wiki_html(pack: dict) -> str: + nav, sections = [], [] + for i, page in enumerate(pack["pages"]): + target = f"page-{i}" + nav.append(f'{escape(page["title"])}') + body = _SECTIONS[page["kind"]](page) + sections.append(f'
      {body}{_footer(page)}
      ') + blob = json.dumps(pack, sort_keys=True, + separators=(",", ":")).replace("<", "\\u003c") + return ( + "" + '' + '' + f"index | wiki | {escape(pack['repo'])}" + f"" + f"
      index wiki: {escape(pack['repo'])} " + f"pinned to {escape(pack['commit'])}
      " + f'
      ' + f'
      {"".join(sections)}
      ' + f"{EMBED_OPEN}{blob}{EMBED_CLOSE}" + f"" + "" + ) diff --git a/client-plugin/server/src/index_graph/wiki/pack.py b/client-plugin/server/src/index_graph/wiki/pack.py new file mode 100644 index 0000000..b416f22 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/pack.py @@ -0,0 +1,253 @@ +"""Derive the verified-wiki pack from the real module graph of one repo. + +Tier 1 by construction: zero-dependency, model-free, deterministic. Every +page is a projection of the internals graph (Python AST-exact, other +languages best-effort per docs/PROTOCOL.md) or of authored markdown joined +in verbatim. No prose is generated, and every edge shown carries the +file:line evidence the graph recorded for it. +""" +from __future__ import annotations + +from pathlib import Path + +from .. import __version__ +from ..internals import InternalGraph, build_internals +from ..knowledge.docs import discover_docs +from ..knowledge.markdown import render_markdown +from ..symbols import SymbolGraph, build_symbol_graph +from ..viz import build_layout, render_mermaid, render_svg +from .seal import WIKI_SCHEMA, build_manifest, head_commit + +# Above this module count the wiki collapses module pages into package pages. +CLUSTER_THRESHOLD = 120 +# Above this symbol count the wiki omits per-symbol pages to avoid bloat; the +# symbol graph is still derived and sealable via `index internals-symbols`. +SYMBOL_PAGE_LIMIT = 800 +_HUB_FAN_IN = 4 + + +def _boundary(evidence_count: int) -> dict: + return {"derived_from": "dependency-graph", "generated_prose": False, + "evidence_count": evidence_count} + + +def _role(fan_in: int, fan_out: int) -> str: + if fan_in == 0 and fan_out > 0: + return "entrypoint" + if fan_in >= _HUB_FAN_IN: + return "hub" + if fan_in == 0 and fan_out == 0: + return "isolated" + if fan_out == 0: + return "leaf" + return "library" + + +def _graph_pack(ids: list[str], languages: dict[str, str], + pair_signals: dict[tuple[str, str], list[dict]], + cycles: list[list[str]]) -> dict: + """Shape a module or package graph like a context pack so the existing + layout/SVG/mermaid renderers draw it unchanged.""" + fan_in: dict[str, int] = {} + fan_out: dict[str, int] = {} + for frm, to in pair_signals: + fan_out[frm] = fan_out.get(frm, 0) + 1 + fan_in[to] = fan_in.get(to, 0) + 1 + roles, salience = {}, {} + for node in ids: + fi, fo = fan_in.get(node, 0), fan_out.get(node, 0) + roles[node] = [_role(fi, fo)] + salience[node] = {"in_degree": fi, "out_degree": fo, "hub": fi >= _HUB_FAN_IN} + cycle_sets = [set(c) for c in cycles] + relations = [] + for (frm, to), signals in sorted(pair_signals.items()): + confidence = "high" if languages.get(frm) == "python" else "moderate" + relations.append({"from": frm, "to": to, "target_name": to, "external": False, + "confidence": confidence, "signals": signals, + "in_cycle": any(frm in c and to in c for c in cycle_sets)}) + return {"repos": [{"name": n} for n in sorted(ids)], "relations": relations, + "roles": roles, "salience": salience, "cycles": [list(c) for c in cycles]} + + +def _module_graph_pack(g: InternalGraph) -> dict: + pairs: dict[tuple[str, str], list[dict]] = {} + for e in g.edges: + pairs.setdefault((e.from_id, e.to_id), []).append( + {"kind": "import", "file": e.evidence_file, + "line": e.evidence_line, "raw": e.raw}) + languages = {m.id: m.language for m in g.modules} + return _graph_pack([m.id for m in g.modules], languages, pairs, + [list(c) for c in g.cycles]) + + +def _package_of(module_id: str) -> str: + return module_id.rsplit("/", 1)[0] if "/" in module_id else "(root)" + + +def _package_graph_pack(g: InternalGraph) -> dict: + pairs: dict[tuple[str, str], list[dict]] = {} + for e in g.edges: + frm, to = _package_of(e.from_id), _package_of(e.to_id) + if frm != to: + pairs.setdefault((frm, to), []).append( + {"kind": "import", "file": e.evidence_file, + "line": e.evidence_line, "raw": e.raw}) + members: dict[str, set[str]] = {} + for m in g.modules: + members.setdefault(_package_of(m.id), set()).add(m.language) + languages = {pkg: ("python" if langs == {"python"} else "mixed") + for pkg, langs in members.items()} + return _graph_pack(sorted(members), languages, pairs, []) + + +def _overview_page(g: InternalGraph, docs: list, commit: str) -> dict: + entry_points = sorted(m.id for m in g.modules + if g.fan_in.get(m.id, 0) == 0 and g.fan_out.get(m.id, 0) > 0) + return {"id": "overview", "kind": "overview", "title": f"{g.repo} overview", + "repo": g.repo, "commit": commit, + "ecosystems": sorted({m.language for m in g.modules}), + "entry_points": entry_points, + "module_count": len(g.modules), "internal_edge_count": len(g.edges), + "cycle_count": len(g.cycles), + "doc_count": len(docs), "doc_paths": [d.rel_path for d in docs], + "coverage": {"complete": g.coverage.complete, + "parse_errors": list(g.coverage.parse_errors), + "dynamic_imports": [{"file": f, "line": ln} + for f, ln in g.coverage.dynamic_imports]}, + "boundary": _boundary(0)} + + +def _module_pages(g: InternalGraph) -> list[dict]: + imports_by: dict[str, list[dict]] = {} + dependents_by: dict[str, list[dict]] = {} + for e in g.edges: + imports_by.setdefault(e.from_id, []).append( + {"to": e.to_id, "file": e.evidence_file, "line": e.evidence_line, "raw": e.raw}) + dependents_by.setdefault(e.to_id, []).append( + {"from": e.from_id, "file": e.evidence_file, "line": e.evidence_line, "raw": e.raw}) + pages = [] + for m in g.modules: + imports = imports_by.get(m.id, []) + dependents = dependents_by.get(m.id, []) + pages.append({"id": f"module/{m.id}", "kind": "module", "title": m.id, + "module": m.id, "path": m.path, "language": m.language, + "imports": imports, "dependents": dependents, + "cycles": [list(c) for c in g.cycles if m.id in c], + "boundary": _boundary(len(imports) + len(dependents))}) + return pages + + +def _package_pages(g: InternalGraph) -> list[dict]: + members: dict[str, list[str]] = {} + for m in g.modules: + members.setdefault(_package_of(m.id), []).append(m.id) + outgoing: dict[str, dict[str, list[dict]]] = {} + incoming: dict[str, dict[str, list[dict]]] = {} + for e in g.edges: + frm, to = _package_of(e.from_id), _package_of(e.to_id) + if frm == to: + continue + via = {"from": e.from_id, "to": e.to_id, + "file": e.evidence_file, "line": e.evidence_line} + outgoing.setdefault(frm, {}).setdefault(to, []).append(via) + incoming.setdefault(to, {}).setdefault(frm, []).append(via) + pages = [] + for pkg in sorted(members): + imports = [{"to": t, "via": v} for t, v in sorted(outgoing.get(pkg, {}).items())] + dependents = [{"from": f, "via": v} for f, v in sorted(incoming.get(pkg, {}).items())] + evidence = (sum(len(i["via"]) for i in imports) + + sum(len(d["via"]) for d in dependents)) + pages.append({"id": f"package/{pkg}", "kind": "package", "title": pkg, + "package": pkg, "modules": sorted(members[pkg]), + "imports": imports, "dependents": dependents, + "boundary": _boundary(evidence)}) + return pages + + +def _architecture_page(pack: dict, granularity: str, edge_count: int) -> dict: + return {"id": "architecture", "kind": "architecture", "title": "Architecture", + "granularity": granularity, + "svg": render_svg(build_layout(pack, include_external=False)), + "mermaid": render_mermaid(pack, include_external=False), + "edge_count": edge_count, "cycles": pack["cycles"], + "boundary": _boundary(edge_count)} + + +def _docs_page(docs: list) -> dict: + entries = [{"path": d.rel_path, "title": d.title, "html": render_markdown(d.body)} + for d in docs] + return {"id": "docs", "kind": "docs", "title": "Docs", + "provenance": "authored-by-humans", "docs": entries, + "boundary": _boundary(0)} + + +def _symbol_boundary(evidence_count: int) -> dict: + return {"derived_from": "symbol-call-graph", "generated_prose": False, + "evidence_count": evidence_count} + + +def _symbol_pages(sg: SymbolGraph) -> list[dict]: + """One page per symbol: definition, resolved callers/callees with evidence, + and any unresolved references surfaced honestly (never a guessed edge). + + Skipped entirely above SYMBOL_PAGE_LIMIT so a huge repo does not bloat the + pack; the symbol graph itself is still derivable and sealable on the CLI. + """ + if len(sg.symbols) > SYMBOL_PAGE_LIMIT: + return [] + callers_by: dict[str, list[dict]] = {} + callees_by: dict[str, list[dict]] = {} + unresolved_by: dict[str, list[dict]] = {} + for c in sg.calls: + if c.to_symbol is not None: + callees_by.setdefault(c.from_symbol, []).append( + {"to_symbol": c.to_symbol, "file": c.evidence_file, + "line": c.evidence_line, "raw": c.raw}) + callers_by.setdefault(c.to_symbol, []).append( + {"from_symbol": c.from_symbol, "file": c.evidence_file, + "line": c.evidence_line, "raw": c.raw}) + else: + unresolved_by.setdefault(c.from_symbol, []).append( + {"to_name": c.to_name, "file": c.evidence_file, + "line": c.evidence_line, "raw": c.raw, + "reason": "unresolved: dynamic or cross-module target not in this repo"}) + pages = [] + for d in sg.symbols: + callers = callers_by.get(d.id, []) + callees = callees_by.get(d.id, []) + unresolved = unresolved_by.get(d.id, []) + pages.append({ + "id": f"symbol/{d.id}", "kind": "symbol", "title": d.name, + "symbol": d.id, "symbol_kind": d.kind, "module": d.module_id, + "parent": d.parent, "is_public": d.is_public, + "definition": {"file": d.file, "line": d.line}, + "callers": callers, "callees": callees, "unresolved_calls": unresolved, + "boundary": _symbol_boundary(len(callers) + len(callees))}) + return pages + + +def build_wiki_pack(root: Path | str, repo_name: str | None = None) -> dict: + """The whole wiki as one sealed, portable, deterministic JSON pack.""" + root = Path(root).resolve() + g = build_internals(root, repo_name) + sg = build_symbol_graph(root, repo_name) + docs = discover_docs(root) + commit = head_commit(root) + clustered = len(g.modules) > CLUSTER_THRESHOLD + if clustered: + body = _package_pages(g) + arch = _architecture_page(_package_graph_pack(g), "package", len(g.edges)) + else: + body = _module_pages(g) + arch = _architecture_page(_module_graph_pack(g), "module", len(g.edges)) + symbol_pages = _symbol_pages(sg) + pages = [_overview_page(g, docs, commit), arch, *body, *symbol_pages, _docs_page(docs)] + inputs = {"modules": len(g.modules), "internal_edges": len(g.edges), + "docs": len(docs), "clustered": clustered, + "symbols": len(sg.symbols), "symbol_pages": len(symbol_pages), + "resolved_symbol_calls": sg.coverage.resolved_calls, + "coverage_complete": g.coverage.complete} + manifest = build_manifest(pages, repo=g.repo, commit=commit, + inputs=inputs, tool_version=__version__) + return {"schema": WIKI_SCHEMA, "repo": g.repo, "commit": commit, + "pages": pages, "manifest": manifest} diff --git a/client-plugin/server/src/index_graph/wiki/seal.py b/client-plugin/server/src/index_graph/wiki/seal.py new file mode 100644 index 0000000..e9614d7 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/seal.py @@ -0,0 +1,213 @@ +"""Seal and verify the wiki artifact: page hashes, graph re-derivation, commit pin. + +A page hash is the canonical SHA-256 of the page object (docs/PROTOCOL.md). +The verdict is one of three words, MATCH, DRIFT, or UNVERIFIABLE, never a +fourth, with exit codes 0/1/2 to match the existing verify contract. DRIFT +means a page was tampered with, the wiki claims a module edge the real graph +does not contain, or the repo moved off the pinned commit. UNVERIFIABLE means +the artifact cannot be evaluated at all. A verifier that cannot fail on a +known-bad input is not a verifier; the negative fixtures live in the tests. +""" +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +from ..certify.certificate import canonical_sha + +WIKI_SCHEMA = "index.wiki/1" +VERIFICATION_SCHEMA = "index.wiki-verification/1" +VERDICT_EXIT = {"MATCH": 0, "DRIFT": 1, "UNVERIFIABLE": 2} + +# The data island the HTML artifact embeds; render and extract share one truth. +EMBED_OPEN = '" + + +def head_commit(root: Path | str) -> str: + """Full HEAD sha of the repo at root, or "unversioned" for a non-git root.""" + try: + out = subprocess.run(["git", "-C", str(root), "rev-parse", "HEAD"], + capture_output=True, text=True, timeout=10) + except (OSError, subprocess.TimeoutExpired): + return "unversioned" + sha = out.stdout.strip() + return sha if out.returncode == 0 and sha else "unversioned" + + +def build_manifest(pages: list[dict], *, repo: str, commit: str, + inputs: dict, tool_version: str) -> dict: + return { + "schema": WIKI_SCHEMA, + "repo": repo, + "commit": commit, + "tool_version": tool_version, + "pages": [{"id": p["id"], "sha256": canonical_sha(p)} for p in pages], + "inputs": inputs, + } + + +def claimed_module_edges(pages: list[dict]) -> set[tuple[str, str]]: + """Every module-level edge the wiki asserts, from module and package pages.""" + edges: set[tuple[str, str]] = set() + for page in pages: + if page.get("kind") == "module": + for imp in page.get("imports", []): + edges.add((page.get("module", ""), imp.get("to", ""))) + for dep in page.get("dependents", []): + edges.add((dep.get("from", ""), page.get("module", ""))) + elif page.get("kind") == "package": + for group in list(page.get("imports", [])) + list(page.get("dependents", [])): + for via in group.get("via", []): + edges.add((via.get("from", ""), via.get("to", ""))) + return edges + + +def claimed_symbol_calls(pages: list[dict]) -> set[tuple[str, str]]: + """Every resolved symbol-call edge (caller, callee) the wiki asserts, drawn + from the callers/callees of each symbol page. An unresolved reference (which + the pages surface separately, without a to_symbol) is never a claimed edge.""" + edges: set[tuple[str, str]] = set() + for page in pages: + if page.get("kind") != "symbol": + continue + sym = page.get("symbol", "") + for callee in page.get("callees", []): + edges.add((sym, callee.get("to_symbol", ""))) + for caller in page.get("callers", []): + edges.add((caller.get("from_symbol", ""), sym)) + return edges + + +def _structural_gap(artifact: object) -> str | None: + if not isinstance(artifact, dict): + return "artifact is not a JSON object" + if artifact.get("schema") != WIKI_SCHEMA: + return f"artifact schema must be {WIKI_SCHEMA}" + pages, manifest = artifact.get("pages"), artifact.get("manifest") + if not isinstance(pages, list) or not pages: + return "artifact carries no pages" + if any(not isinstance(p, dict) or "id" not in p for p in pages): + return "artifact carries a page without an id" + if not isinstance(manifest, dict) or not isinstance(manifest.get("pages"), list): + return "artifact carries no sealing manifest" + return None + + +def _hash_findings(artifact: dict) -> list[dict]: + listed = {e.get("id"): e.get("sha256") + for e in artifact["manifest"]["pages"] if isinstance(e, dict)} + findings: list[dict] = [] + seen: set = set() + for page in artifact["pages"]: + pid = page["id"] + seen.add(pid) + if pid not in listed: + findings.append({"rule": "page-unlisted", + "detail": f"page {pid!r} is not sealed by the manifest"}) + elif canonical_sha(page) != listed[pid]: + findings.append({"rule": "page-tampered", + "detail": f"page {pid!r} does not match its sealed hash"}) + for pid in sorted(set(listed) - seen, key=str): + findings.append({"rule": "page-missing", + "detail": f"manifest seals page {pid!r} " + "but the artifact does not carry it"}) + return findings + + +def _report(verdict: str, findings: list[dict], *, pages_checked: int, + edges_checked: int, recheck: str) -> dict: + return {"schema": VERIFICATION_SCHEMA, "verdict": verdict, "findings": findings, + "pages_checked": pages_checked, "edges_checked": edges_checked, + "recheck": recheck} + + +def verify_wiki(artifact: object, root: Path | str, *, recheck: str = "") -> dict: + """Recompute page hashes and the graph derivation against the current tree.""" + root = Path(root) + recheck = recheck or f'index wiki --verify --root "{root}"' + gap = _structural_gap(artifact) + if gap is None and not root.is_dir(): + gap = f"root not found: {root}" + if gap: + return _report("UNVERIFIABLE", [{"rule": "artifact", "detail": gap}], + pages_checked=0, edges_checked=0, recheck=recheck) + findings = _hash_findings(artifact) + from ..internals import build_internals + real = {(e.from_id, e.to_id) for e in build_internals(root).edges} + claimed = claimed_module_edges(artifact["pages"]) + for frm, to in sorted(claimed - real): + findings.append({"rule": "edge-not-in-graph", + "detail": f"wiki claims module edge {frm} -> {to} but the " + "graph derived from the current tree does not contain it"}) + from ..symbols import build_symbol_graph, symbol_graph_to_claims + real_symbol = symbol_graph_to_claims(build_symbol_graph(root)) + claimed_symbol = claimed_symbol_calls(artifact["pages"]) + for frm, to in sorted(claimed_symbol - real_symbol): + findings.append({"rule": "symbol-call-not-in-graph", + "detail": f"wiki claims symbol call {frm} -> {to} but the " + "symbol graph derived from the current tree does " + "not contain it"}) + pinned, current = artifact["manifest"].get("commit"), head_commit(root) + if pinned != current: + findings.append({"rule": "commit-moved", + "detail": f"wiki is pinned to {pinned} " + f"but the current tree is at {current}"}) + else: + # Several pages assert content that no structured edge/hash check + # re-derives once the manifest is resealed: the architecture DIAGRAM + # (rendered svg/mermaid strings), the OVERVIEW facts (coverage, counts, + # entry points), and the DOCS prose (sealed but never re-read from + # source). Re-derive the whole pack from the tree (deterministic) and + # confirm each such page is exactly its fresh re-render. Only compared + # at the pinned commit, so a moved tree is named by commit-moved, not + # double-counted here. + from .pack import build_wiki_pack + fresh = {p.get("kind"): p for p in build_wiki_pack(root)["pages"]} + _CHECKS = (("architecture", "architecture-diagram-drift", + "the architecture diagram depicts structure the code does not have"), + ("overview", "overview-not-in-graph", + "the overview asserts facts the graph derived from the tree does not have"), + ("docs", "docs-not-in-source", + "a docs page asserts prose not re-derivable from the source markdown")) + for kind, rule, detail in _CHECKS: + stored = next((p for p in artifact["pages"] + if p.get("kind") == kind), None) + fresh_page = fresh.get(kind) + if stored is not None and fresh_page is not None and stored != fresh_page: + findings.append({"rule": rule, "detail": detail}) + return _report("MATCH" if not findings else "DRIFT", findings, + pages_checked=len(artifact["pages"]), + edges_checked=len(claimed) + len(claimed_symbol), recheck=recheck) + + +def extract_embedded_pack(html_text: str) -> dict: + """Pull the sealed pack back out of a rendered HTML artifact.""" + start = html_text.find(EMBED_OPEN) + if start < 0: + raise ValueError("no embedded wiki-data island found") + start += len(EMBED_OPEN) + end = html_text.find(EMBED_CLOSE, start) + if end < 0: + raise ValueError("embedded wiki-data island is unterminated") + return json.loads(html_text[start:end]) + + +def load_artifact(path: Path | str) -> dict: + text = Path(path).read_text(encoding="utf-8") + if text.lstrip().startswith("<"): + return extract_embedded_pack(text) + return json.loads(text) + + +def run_verify(artifact_path: Path | str, root: Path | str) -> dict: + """The shared CLI/MCP verification payload; an unreadable artifact is UNVERIFIABLE.""" + recheck = f'index wiki --verify "{artifact_path}" --root "{root}"' + try: + artifact = load_artifact(artifact_path) + except (OSError, ValueError) as exc: + return _report("UNVERIFIABLE", + [{"rule": "artifact", "detail": f"cannot read artifact: {exc}"}], + pages_checked=0, edges_checked=0, recheck=recheck) + return verify_wiki(artifact, root, recheck=recheck) diff --git a/client-plugin/server/src/index_graph/wiki/serve.py b/client-plugin/server/src/index_graph/wiki/serve.py new file mode 100644 index 0000000..fe65f32 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/serve.py @@ -0,0 +1,141 @@ +"""`index serve`: a stdlib http.server that serves the verified wiki on demand. + +The URL-swap experience: GET /// reconstructs the git +URL https:////, runs the EXISTING verified-wiki generation +(the same clone -> derive -> clean-up code path as `index wiki `), and +serves the self-contained HTML. Consent-clean by construction: generation is +on demand only, nothing is crawled or pre-indexed, robots.txt disallows +indexing, and every page states that the wiki derives structure and defers to +the repo owner's authored docs. This is the LOCAL server component; deploying +or hosting it anywhere is a separate operator decision. +""" +from __future__ import annotations + +import re +import shutil +import tempfile +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +from .cli import CloneError, clone_repo +from .html import render_wiki_html +from .pack import build_wiki_pack +from .serve_pages import ROBOTS_TXT, error_page, inject_banner, landing_page + +# A route segment: a plain forge/org/repo token. No slashes, no traversal, no +# leading dot, no scheme or userinfo. Deliberately strict; anything else is 400. +_SEGMENT = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$") +_ALLOWED_HOSTS = None # None = any host of the right shape; a set narrows it. + + +class RouteError(ValueError): + """A request path is not a valid /// route. The message is + a plain reason, safe to show a user, with no traceback.""" + + +def parse_route(path: str) -> tuple[str, str, str]: + """Parse /// into (host, org, repo), or raise + RouteError. Tolerates a trailing slash and a trailing `.git` on the repo.""" + trimmed = path.split("?", 1)[0].split("#", 1)[0].strip("/") + if trimmed.endswith(".git"): + trimmed = trimmed[:-4] + parts = trimmed.split("/") + if len(parts) != 3: + raise RouteError( + "a repo route is host/org/repo (e.g. /github.com/org/repo); " + "this path is not that shape") + host, org, repo = parts + for label, seg in (("host", host), ("org", org), ("repo", repo)): + if not _SEGMENT.match(seg): + raise RouteError(f"the {label} segment {seg!r} is not a plain " + "host/org/repo token") + if "." not in host: + raise RouteError(f"the host segment {host!r} is not a forge host") + if _ALLOWED_HOSTS is not None and host not in _ALLOWED_HOSTS: + raise RouteError(f"forge host {host!r} is not on the allow-list") + return host, org, repo + + +def build_git_url(host: str, org: str, repo: str) -> str: + """Reconstruct the https git URL from a parsed route. Always https, never a + scp-style or ssh URL, so only http(s) forge URLs are ever cloned.""" + return f"https://{host}/{org}/{repo}" + + +def wiki_for_route(path: str) -> str: + """Derive and render the verified wiki for a route, banner injected. Reuses + the `index wiki ` path: shallow clone, derive, always clean up.""" + host, org, repo = parse_route(path) + url = build_git_url(host, org, repo) + dest = Path(tempfile.mkdtemp(prefix="index-serve-")) + try: + clone_repo(url, dest) + pack = build_wiki_pack(dest, repo_name=repo) + return inject_banner(render_wiki_html(pack)) + finally: + shutil.rmtree(dest, ignore_errors=True) + + +class _WikiHandler(BaseHTTPRequestHandler): + server_version = "index-serve" + + def _send(self, status: int, body: bytes, ctype: str) -> None: + self.send_response(status) + self.send_header("Content-Type", ctype) + self.send_header("Content-Length", str(len(body))) + self.send_header("X-Robots-Tag", "noindex, nofollow") + self.end_headers() + if self.command != "HEAD": + self.wfile.write(body) + + def do_GET(self) -> None: # noqa: N802 (http.server naming) + route = self.path.split("?", 1)[0] + if route == "/": + return self._send(200, landing_page(), "text/html; charset=utf-8") + if route == "/robots.txt": + return self._send(200, ROBOTS_TXT.encode("utf-8"), + "text/plain; charset=utf-8") + if route == "/favicon.ico": + return self._send(204, b"", "image/x-icon") + self._serve_wiki(route) + + do_HEAD = do_GET + + def _serve_wiki(self, route: str) -> None: + try: + html = wiki_for_route(route) + except RouteError as exc: + return self._send(400, error_page(400, str(exc)), + "text/html; charset=utf-8") + except CloneError as exc: + reason = f"could not build the wiki: {exc}" + return self._send(502, error_page(502, reason), + "text/html; charset=utf-8") + self._send(200, html.encode("utf-8"), "text/html; charset=utf-8") + + def log_message(self, *args) -> None: # keep the server quiet in tests + return + + +def make_server(host: str = "127.0.0.1", port: int = 8000) -> ThreadingHTTPServer: + """Bind a threading HTTP server on (host, port). Default loopback; port 0 + picks an ephemeral port (used by the tests). The caller runs serve_forever.""" + return ThreadingHTTPServer((host, port), _WikiHandler) + + +def serve_forever(host: str = "127.0.0.1", port: int = 8000) -> int: + """Bind and run the server until interrupted. Prints the bound address and + the consent-clean posture, then blocks.""" + server = make_server(host, port) + bound_host, bound_port = server.server_address + print(f"index serve: http://{bound_host}:{bound_port}/ " + "(on-demand verified wiki; robots.txt disallows indexing)") + print("derive-not-generate, commit-pinned, re-checkable; " + "defers to the repo owner's authored docs. Ctrl-C to stop.") + try: + server.serve_forever() + except KeyboardInterrupt: + print("\nindex serve: stopped") + finally: + server.server_close() + return 0 diff --git a/client-plugin/server/src/index_graph/wiki/serve_pages.py b/client-plugin/server/src/index_graph/wiki/serve_pages.py new file mode 100644 index 0000000..e91e7c1 --- /dev/null +++ b/client-plugin/server/src/index_graph/wiki/serve_pages.py @@ -0,0 +1,95 @@ +"""Static pages and the consent-clean banner for the on-demand wiki server. + +Kept separate from serve.py so both stay under the file-size ceiling. Every +page states the same posture: the wiki DERIVES structure from the dependency +graph, generates no prose, is commit-pinned and re-checkable, and defers to +the repo owner's authored docs. Nothing here crawls, pre-indexes, or publishes. +""" +from __future__ import annotations + +from html import escape + +from ..viz.theme import css_variables + +# One sentence reused verbatim on the landing page and injected into every +# served wiki, so the derive-not-generate posture is never missing. +DERIVE_BANNER = ( + "This wiki derives structure from the dependency graph and does not " + "generate prose. It is commit-pinned and re-checkable with " + "index wiki --verify, and it defers to the repo owner's authored docs." +) + +_PAGE_CSS = ( + "body{margin:0;background:var(--bg);color:var(--ink);" + "font-family:var(--font-body);line-height:1.5}" + "main{max-width:44rem;margin:0 auto;padding:2rem 1.2rem}" + "h1{font-family:var(--font-mono);font-size:1.4rem}" + "code{font-family:var(--font-mono)}" + ".note{border:1px solid var(--hairline);border-radius:6px;padding:.8rem 1rem;" + "background:rgba(70,54,232,.05);font-size:.92rem}" + "a{color:var(--accent)}" + "footer{margin-top:2rem;font-family:var(--font-mono);font-size:.78rem;" + "color:var(--muted)}" +) + +# The consent-clean banner as an HTML fragment, injected into served wikis. +BANNER_HTML = ( + '
      ' + f"{escape(DERIVE_BANNER)}
      " +) + +ROBOTS_TXT = "User-agent: *\nDisallow: /\n" + + +def _shell(title: str, body: str) -> bytes: + return ( + "" + '' + '' + f"{escape(title)}" + f"" + f"
      {body}
      " + ).encode("utf-8") + + +def landing_page() -> bytes: + """The root page: how to use the server, stated in consent-clean terms.""" + body = ( + "

      index serve

      " + "

      A local server that derives a verified wiki for one repository " + "on demand. Request a repo by its forge path and index " + "shallow-clones it, derives the wiki from its dependency graph, serves " + "the self-contained page, and removes the clone.

      " + "

      Try a path of the shape " + "/<forge-host>/<org>/<repo>, for example " + "/github.com/org/repo.

      " + f'

      {escape(DERIVE_BANNER)}

      ' + "

      Generation is on demand only: nothing is crawled or pre-indexed, and " + "this local server publishes nothing. Deploying or hosting it anywhere is " + "a separate operator decision.

      " + "
      index wiki, derived from the module graph, never generated " + "prose. robots.txt disallows indexing.
      " + ) + return _shell("index serve", body) + + +def error_page(status: int, reason: str) -> bytes: + """A plain error page, never a stack trace.""" + body = ( + f"

      {status}

      " + f"

      {escape(reason)}

      " + '

      back to the landing page

      ' + ) + return _shell(f"index serve: {status}", body) + + +def inject_banner(wiki_html: str) -> str: + """Insert the consent-clean banner just inside the served wiki's .""" + marker = "" + idx = wiki_html.find(marker) + if idx < 0: + return BANNER_HTML + wiki_html + cut = idx + len(marker) + return wiki_html[:cut] + BANNER_HTML + wiki_html[cut:] diff --git a/client-plugin/server/src/index_graph/workbench.py b/client-plugin/server/src/index_graph/workbench.py new file mode 100644 index 0000000..64b4410 --- /dev/null +++ b/client-plugin/server/src/index_graph/workbench.py @@ -0,0 +1,291 @@ +"""The index workbench: every capability index derives, composed into ONE +self-contained page. + +Separately, index already ships a verified wiki, an interactive atlas, a +context lens, freshness fingerprints, and a symbol/dependency graph — each its +own artifact. The workbench folds the single-pass surfaces into one pack a +single HTML shell renders: a map you navigate, the docs that describe it, the +context lens over its token budget, and a health panel with the freshness +fingerprint, cycles, and salience audit. Every dependency edge carries its +file:line evidence; nothing is authored, everything re-derives. + +Honest scope (stated on the page, not hidden): freshness/drift VERDICTS need a +prior pinned snapshot to diff against. A single `--root` pass cannot emit a +drift verdict, so the workbench shows the current fingerprint (the baseline a +later `index drift` compares to) and the structural health that IS computable +in one pass — never a fabricated FRESH/STALE badge. +""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from .context.lens import build_lens_pack +from .freshness.fingerprint import workspace_fingerprint +from .graph.build import DependencyGraph +from .knowledge.atlas import build_atlas_pack +from .knowledge.markdown import render_markdown +from .viz.atlas_layout import build_atlas_layout +from .viz.atlas_svg import render_atlas_svg + +SCHEMA = "project-telos.workbench/v1" +TOOL = "index.workbench" +ACTION_SCHEMA = "project-telos.flagship-action/v1" + +# Page-weight budgets. Large workspaces (2k+ docs) produced a 60MB page that +# choked renderers — a robustness failure. Budgets are HONEST: what is dropped +# is counted and shown on the page ("N of M"), search metadata stays complete, +# and every cap is overridable at the CLI. Never a silent truncation. +MAX_DOC_BODIES = 200 # rendered markdown bodies embedded in the page +MAX_MAP_DOCS_PER_REPO = 6 # doc satellites drawn under one repo on the map +MAX_MAP_BAND_DOCS = 40 # cross-cutting (band) docs drawn on the map +MAX_MENTIONS = 2000 # weakest-tier edges embedded (describes/links-to + # always ship in full; mention drops are counted) + + +def load_spine(spine_dir: str | Path | None, root: Path) -> dict: + """Ingest captured flagship-action envelopes (the peers' own doctor/status + output) and detect on-disk peer surfaces. FAIL CLOSED: a file that is not a + valid flagship-action envelope is skipped WITH ITS REASON recorded, never + guessed at. Peers stay peers: the forum ledger is reported as present with + the forum command to inspect it — index never parses a peer's internals.""" + tools: list[dict] = [] + skipped: list[dict] = [] + ring: list[dict] = [] + if spine_dir is not None: + for p in sorted(Path(spine_dir).glob("*.json")): + try: + env = json.loads(p.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + skipped.append({"file": p.name, "reason": f"unreadable: {exc}"}) + continue + if env.get("schema") != ACTION_SCHEMA or "tool" not in env: + skipped.append({"file": p.name, + "reason": f"not a {ACTION_SCHEMA} envelope"}) + continue + checks = (env.get("native") or {}).get("checks", []) + tools.append({ + "tool": env["tool"], "version": env.get("tool_version", ""), + "command": env.get("command", ""), "status": env.get("status", ""), + "checks": [{"name": c.get("name", ""), "status": c.get("status", "")} + for c in checks], + "source_file": p.name, + }) + for na in env.get("next_actions", []): + ring.append({"from": env["tool"], "to": na.get("tool", ""), + "action": na.get("action", ""), + "reason": na.get("reason", "")}) + ledger = root / ".telos" / "forum-ledger" + surfaces = [] + if ledger.is_dir(): + surfaces.append({ + "peer": "forum", "surface": "causal ledger", + "path": str(ledger), "entries": sum(1 for _ in ledger.iterdir()), + "inspect": "forum ledger summary (or MCP forum.ledger.summary)"}) + return {"tools": sorted(tools, key=lambda t: t["tool"]), + "ring": sorted(ring, key=lambda e: (e["from"], e["to"])), + "skipped": skipped, "peer_surfaces": surfaces, + "capture": ("each flagship's doctor/status --json IS the envelope; " + "drop them in a directory and pass --spine-dir")} + + +def _evidence(rel: dict) -> list[dict]: + """The file:line signals behind a dependency edge — the 'why' of the map.""" + out = [] + for s in rel.get("signals", []): + if s.get("file"): + out.append({"kind": s.get("kind", ""), "file": s["file"], "line": s.get("line")}) + return out + + +def _repo_view(pack: dict, fresh: dict) -> list[dict]: + sal = pack.get("salience", {}) + roles = pack.get("roles", {}) + fresh_repos = fresh.get("repos", {}) + describes: dict[str, list[str]] = {} + for e in pack.get("knowledge_edges", []): + if e["type"] == "describes" and e["to_kind"] == "repo": + describes.setdefault(e["to"], []).append(e["from"]) + deps: dict[str, list[dict]] = {} + for r in pack.get("relations", []): + if r.get("external"): + continue + deps.setdefault(r["from"], []).append({ + "to": r["to"], "confidence": r.get("confidence", ""), + "in_cycle": r.get("in_cycle", False), "evidence": _evidence(r)}) + out = [] + for repo in pack.get("repos", []): + name = repo["name"] + s = sal.get(name, {}) + out.append({ + "name": name, + "roles": roles.get(name, []), + "ecosystems": repo.get("ecosystems", []), + "description": repo.get("description", ""), + "markers": repo.get("markers", []), + "in_degree": s.get("in_degree", 0), + "out_degree": s.get("out_degree", 0), + "depends_on": sorted(deps.get(name, []), key=lambda d: d["to"]), + "documented_by": sorted(describes.get(name, [])), + "fingerprint": fresh_repos.get(name, ""), + }) + return sorted(out, key=lambda r: r["name"]) + + +def _summary(pack: dict, repos: list[dict]) -> dict: + role_counts: dict[str, int] = {} + for r in repos: + for role in (r["roles"] or ["untyped"]): + role_counts[role] = role_counts.get(role, 0) + 1 + edge_types: dict[str, int] = {} + for e in pack.get("knowledge_edges", []): + edge_types[e["type"]] = edge_types.get(e["type"], 0) + 1 + return { + "repos": len(repos), + "docs": len(pack.get("docs", [])), + "relations": len([r for r in pack.get("relations", []) if not r.get("external")]), + "external_relations": len([r for r in pack.get("relations", []) if r.get("external")]), + "knowledge_edges": len(pack.get("knowledge_edges", [])), + "cycles": len(pack.get("cycles", [])), + "warnings": len(pack.get("warnings", [])) + len(pack.get("knowledge_warnings", [])), + "role_counts": dict(sorted(role_counts.items())), + "edge_types": dict(sorted(edge_types.items())), + "top_salience": [r["name"] for r in sorted( + repos, key=lambda r: (-r["in_degree"], -r["out_degree"], r["name"]))[:8]], + } + + +def _rank_docs(pack: dict) -> list[str]: + """Doc ids by connectivity (strongest knowledge edges first), then id. + Deterministic; drives which bodies are embedded and which nodes are drawn.""" + weight = {"describes": 3, "links-to": 2, "mentions": 1} + score: dict[str, int] = {d["id"]: 0 for d in pack.get("docs", [])} + for e in pack.get("knowledge_edges", []): + w = weight.get(e["type"], 0) + if e["from"] in score: + score[e["from"]] += w + if e.get("to_kind") == "doc" and e["to"] in score: + score[e["to"]] += w + return sorted(score, key=lambda i: (-score[i], i)) + + +def _map_pack(pack: dict, ranked: list[str]) -> dict: + """A copy of the pack with docs capped for the MAP ONLY (satellites per + repo + band docs), so the SVG stays renderable on huge workspaces. The + full doc list remains in the workbench data for search and the Docs view.""" + describes: dict[str, str] = {} + for e in sorted(pack.get("knowledge_edges", []), + key=lambda e: (e["from"], e["type"], e["to_kind"], e["to"])): + if e["type"] == "describes" and e["from"] not in describes: + describes[e["from"]] = e["to"] + rank = {rid: i for i, rid in enumerate(ranked)} + per_repo: dict[str, int] = {} + band = 0 + keep: set[str] = set() + for d in sorted(pack.get("docs", []), key=lambda d: rank.get(d["id"], 1 << 30)): + target = describes.get(d["id"]) + if target is not None: + if per_repo.get(target, 0) < MAX_MAP_DOCS_PER_REPO: + per_repo[target] = per_repo.get(target, 0) + 1 + keep.add(d["id"]) + elif band < MAX_MAP_BAND_DOCS: + band += 1 + keep.add(d["id"]) + capped = dict(pack) + capped["docs"] = [d for d in pack.get("docs", []) if d["id"] in keep] + capped["knowledge_edges"] = [ + e for e in pack.get("knowledge_edges", []) + if e["type"] != "mentions" + and e["from"] in keep and (e.get("to_kind") != "doc" or e["to"] in keep)] + return capped + + +def build_workbench_pack( + graph: DependencyGraph, + docs: list, + repo_dirs: dict[str, str], + *, + root: Path | str, + token_budget: int = 6000, + spine_dir: str | Path | None = None, + max_doc_bodies: int = MAX_DOC_BODIES, +) -> dict: + """Compose the atlas, docs, freshness fingerprint, lens, and health into one + deterministic, self-contained pack. The SVG (map markup) rides along under + 'svg'; everything else is JSON the shell renders and the receipt seals. + Page-weight budgets (MAX_*) keep huge workspaces renderable; every drop is + counted in `summary`/`doc_meta` and stated on the page.""" + pack = build_atlas_pack(graph, docs, repo_dirs) + ranked_docs = _rank_docs(pack) + structural = [e for e in pack.get("knowledge_edges", []) + if e["type"] != "mentions"] + mentions = [e for e in pack.get("knowledge_edges", []) + if e["type"] == "mentions"] + kept_mentions = sorted(mentions, key=lambda e: (e["from"], e["to"]))[:MAX_MENTIONS] + pack["knowledge_edges"] = structural + kept_mentions + embed = set(ranked_docs[:max_doc_bodies]) + body_of = {d.rel_path: d.body for d in docs} + pack["doc_html"] = {rid: render_markdown(body_of[rid]) + for rid in sorted(embed) if rid in body_of} + map_capped = _map_pack(pack, ranked_docs) + svg = render_atlas_svg(build_atlas_layout(map_capped, include_external=False)) + fresh = workspace_fingerprint({n.name: Path(n.path) for n in graph.repos}) + repos = _repo_view(pack, fresh) + lens = build_lens_pack(graph, root=root, token_budget=token_budget) + lens_order = [{k: o[k] for k in ("name", "cost", "roles", "salience")} + for o in lens["replay"]["order"]] # page never reads source_refs + wb = { + "schema": SCHEMA, + "tool": TOOL, + "root": str(root), + "summary": _summary(pack, repos), + "repos": repos, + "docs": sorted(pack.get("docs", []), key=lambda d: d["id"]), + "doc_html": pack["doc_html"], + "knowledge_edges": pack.get("knowledge_edges", []), + "cycles": [list(c) for c in pack.get("cycles", [])], + "warnings": list(pack.get("warnings", [])), + "knowledge_warnings": list(pack.get("knowledge_warnings", [])), + "salience_audit": pack.get("salience_audit", []), + "freshness": { + "schema": fresh.get("schema", ""), + "root_sha256": fresh.get("root", ""), + "repos": fresh.get("repos", {}), + "note": ("current fingerprint only; a FRESH/STALE drift verdict " + "requires a prior snapshot"), + "recheck": "index freshness --cert CERT --root ROOT; index drift --from-snap A --to-snap B", + }, + "lens": { + "verdict": lens["envelope"]["verification_verdict"], + "budget": lens["envelope"]["budget"], + "replay": {"rule": lens["replay"]["rule"], + "base_tokens": lens["replay"]["base_tokens"], + "order": lens_order}, + }, + "doc_meta": { + "total": len(pack.get("docs", [])), + "bodies_embedded": len(pack["doc_html"]), + "map_docs": len(map_capped["docs"]), + "mentions_total": len(mentions), + "mentions_embedded": len(kept_mentions), + "bodies_note": ("bodies beyond the budget are not embedded; the doc " + "list and search stay complete — open the file at " + "its id path, or raise --max-doc-bodies"), + }, + "spine": load_spine(spine_dir, Path(root)), + } + # seal the rendered map's hash into the receipt: the SVG is presented as + # sealed, so its markup must be bound. The large markup itself stays out of + # the sealed JSON body, but its content hash is inside it, so a stranger + # re-derives _sha(wb["svg"]) == wb["svg_sha256"] and tampering the map + # breaks the receipt. + wb["svg_sha256"] = _sha(svg) + wb["receipt_sha256"] = _sha(wb) + wb["svg"] = svg # markup, added after sealing the data + return wb + + +def _sha(value: object) -> str: + data = json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8") + return hashlib.sha256(data).hexdigest() diff --git a/scripts/build_client_package.py b/scripts/build_client_package.py index 6d6d7a1..7d86cf2 100644 --- a/scripts/build_client_package.py +++ b/scripts/build_client_package.py @@ -72,6 +72,30 @@ def entries(root): result[path.relative_to(root).as_posix()] = path.read_bytes() return result +VENDORED = "server/src/" + + +def vendored_payload(): + """The files the source ZIP places under server/src/, also committed in the plugin folder.""" + return {VENDORED + k: v for k, v in entries(ROOT / "src").items()} + + +def plugin_entries(): + """Plugin folder inputs without the vendored copy, which is rebuilt from src/.""" + return {k: v for k, v in entries(ROOT / "client-plugin").items() if not k.startswith(VENDORED)} + + +def sync_vendored(): + """Rewrite client-plugin/server/src from src/, deleting stale files.""" + target = ROOT / "client-plugin" / VENDORED + if target.exists(): + shutil.rmtree(target) + for name, data in vendored_payload().items(): + path = ROOT / "client-plugin" / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(data.replace(b"\r\n", b"\n")) + return target + def write_zip(path, payload): with zipfile.ZipFile(path, "x", compression=zipfile.ZIP_STORED) as archive: for name, data in sorted(payload.items()): @@ -90,7 +114,7 @@ def build(output, mode="release", native=False): raise ValueError("output must be a new directory outside source") source = entries(ROOT / "src") scripts = entries(ROOT / "scripts") - plugin = entries(ROOT / "client-plugin") + plugin = plugin_entries() tracked = set(git("ls-files").splitlines()) if mode == "release": candidates = {**{"src/"+k:v for k,v in source.items()}, @@ -102,7 +126,7 @@ def build(output, mode="release", native=False): **{"scripts/"+k: digest(v) for k,v in scripts.items()}, **{"client-plugin/"+k: digest(v) for k,v in plugin.items()}} output.mkdir(parents=True) - payload = {**plugin, **{"server/src/"+k:v for k,v in source.items()}} + payload = {**plugin, **{VENDORED+k:v for k,v in source.items()}} payload["LICENSE"] = (ROOT / "LICENSE").read_bytes() payload["SOURCE.json"] = (json.dumps(qualified, indent=2)+"\n").encode() payload["PAYLOAD-SHA256SUMS"] = checksums(payload) @@ -156,7 +180,7 @@ def build(output, mode="release", native=False): "pyinstaller":importlib.metadata.version("pyinstaller"), "command":command} current = {**{"src/"+k:digest(v) for k,v in entries(ROOT/"src").items()}, **{"scripts/"+k:digest(v) for k,v in entries(ROOT/"scripts").items()}, - **{"client-plugin/"+k:digest(v) for k,v in entries(ROOT/"client-plugin").items()}} + **{"client-plugin/"+k:digest(v) for k,v in plugin_entries().items()}} if current != source_hashes: raise ValueError("source changed during build") receipt["artifacts"] = {p.name:digest(p.read_bytes()) for p in paths} @@ -166,9 +190,16 @@ def build(output, mode="release", native=False): if __name__ == "__main__": parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("output") + parser.add_argument("output", nargs="?") + parser.add_argument("--sync-vendored", action="store_true", + help="rewrite client-plugin/server/src from src/ and exit") parser.add_argument("--mode", choices=("dev","release"), default="release") parser.add_argument("--native", action="store_true") args = parser.parse_args() + if args.sync_vendored: + print(sync_vendored()) + raise SystemExit(0) + if args.output is None: + parser.error("output is required") for path in build(args.output, args.mode, args.native): print(path) diff --git a/tests/test_client_vendored.py b/tests/test_client_vendored.py new file mode 100644 index 0000000..f7c06a5 --- /dev/null +++ b/tests/test_client_vendored.py @@ -0,0 +1,76 @@ +"""The plugin folder must run on its own: vendored server code, size limits, isolated launch.""" +import json +import shutil +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +PLUGIN = ROOT / "client-plugin" +sys.path.insert(0, str(ROOT / "scripts")) +package = __import__("build_client_package") +SYNC = "python scripts/build_client_package.py --sync-vendored" + + +def _lf(data): + return data.replace(b"\r\n", b"\n") + + +def test_vendored_server_code_matches_the_builder(): + expected = {k: _lf(v) for k, v in package.vendored_payload().items()} + committed = {"server/src/" + p.relative_to(PLUGIN / "server/src").as_posix(): _lf(p.read_bytes()) + for p in (PLUGIN / "server/src").rglob("*") + if p.is_file() and "__pycache__" not in p.parts and p.suffix not in {".pyc", ".pyo"}} + missing = sorted(set(expected) - set(committed)) + extra = sorted(set(committed) - set(expected)) + changed = sorted(k for k in set(expected) & set(committed) if expected[k] != committed[k]) + assert not (missing or extra or changed), ( + f"client-plugin/server/src is out of date (missing {missing[:5]}, extra {extra[:5]}, " + f"changed {changed[:5]}). Run: {SYNC}") + + +def test_builder_inputs_do_not_count_the_vendored_copy_twice(): + assert not any(k.startswith("server/src/") for k in package.plugin_entries()) + + +def test_plugin_folder_fits_directory_limits(): + files = [p for p in PLUGIN.rglob("*") if p.is_file() and "__pycache__" not in p.parts] + assert len(files) <= 512, len(files) + large = [(p.relative_to(PLUGIN).as_posix(), p.stat().st_size) for p in files if p.stat().st_size >= 256 * 1024] + assert not large, large + assert not list(PLUGIN.rglob(".gitattributes")) + + +def _launch(folder, workspace, requests): + server = json.loads((folder / ".mcp.json").read_text(encoding="utf-8"))["mcpServers"]["index"] + defaults = {"workspace": str(workspace), "state_directory": ""} + args = [a.replace("${CLAUDE_PLUGIN_ROOT}", str(folder)) for a in server["args"]] + for key, value in defaults.items(): + args = [a.replace("${user_config." + key + "}", value) for a in args] + assert not any("${" in a for a in args), args + command = [sys.executable if server["command"] == "python3" else server["command"], *args] + env = None + if server.get("env"): + import os + env = {**os.environ, **server["env"]} + wire = "".join(json.dumps(r) + "\n" for r in requests) + return subprocess.run(command, input=wire, capture_output=True, text=True, timeout=30, env=env, check=False) + + +def test_plugin_folder_alone_starts_and_lists_tools(tmp_path): + folder = tmp_path / "plugin" + shutil.copytree(PLUGIN, folder, ignore=shutil.ignore_patterns("__pycache__")) + workspace = tmp_path / "workspace" + workspace.mkdir() + requests = [{"jsonrpc": "2.0", "id": 1, "method": "initialize", "params": {}}, + {"jsonrpc": "2.0", "id": 2, "method": "tools/list"}] + result = _launch(folder, workspace, requests) + assert result.returncode == 0, result.stderr + rows = [json.loads(line) for line in result.stdout.splitlines()] + assert rows[0]["result"]["serverInfo"]["name"] == "index-local" + names = {t["name"] for t in rows[1]["result"]["tools"]} + assert {"index.map", "index.select", "index.symbol-definition", "index.symbol-references"} <= names + shutil.rmtree(folder / "server/src") + result = _launch(folder, workspace, requests) + assert result.returncode == 1 and not result.stdout + assert result.stderr.strip() == "index: the server code is missing from the plugin folder. Reinstall the plugin."