Files

2136 lines
79 KiB
Python

"""CLI entry point for code-review-graph.
Usage:
code-review-graph install
code-review-graph init
code-review-graph uninstall [--platform NAME] [--dry-run] [--yes] [--repo PATH]
code-review-graph build [--base BASE]
code-review-graph update [--base BASE]
code-review-graph forget PATH [PATH ...] [--dry-run]
code-review-graph watch
code-review-graph status
code-review-graph serve [--auto-watch] [--http] [--host ADDR] [--port PORT]
code-review-graph mcp [--auto-watch]
code-review-graph visualize
code-review-graph wiki
code-review-graph detect-changes [--base BASE] [--brief]
code-review-graph register <path> [--alias name]
code-review-graph unregister <path_or_alias>
code-review-graph repos
code-review-graph daemon start [--foreground]
code-review-graph daemon stop
code-review-graph daemon restart [--foreground]
code-review-graph daemon status
code-review-graph daemon logs [--repo ALIAS] [--follow] [--lines N]
code-review-graph daemon add <path> [--alias NAME]
code-review-graph daemon remove <path_or_alias>
"""
from __future__ import annotations
import sys
# Python version check — must come before any other imports
if sys.version_info < (3, 10):
print("code-review-graph requires Python 3.10 or higher.")
print(f" You are running Python {sys.version}")
print()
print("Install Python 3.10+: https://www.python.org/downloads/")
sys.exit(1)
import argparse
import fnmatch
import json
import logging
import os
from functools import partial
from importlib.metadata import PackageNotFoundError
from importlib.metadata import version as pkg_version
from pathlib import Path
from typing import Iterable, TypedDict
logger = logging.getLogger(__name__)
# Shared platform choices for install and init commands
_PLATFORM_CHOICES = [
"codex", "claude", "claude-code", "cursor", "windsurf", "zed",
"continue", "opencode", "antigravity", "gemini-cli", "qwen", "kiro", "qoder",
"copilot", "copilot-cli", "codebuddy", "hermes", "all",
]
class _EmbeddingRefreshKwargs(TypedDict, total=False):
embedding_provider: str
embedding_model: str
def _get_version() -> str:
"""Get the installed package version.
Tries ``importlib.metadata`` first (canonical source from the installed
dist-info), falling back to the package's ``__version__`` attribute if
metadata is unavailable or corrupt. This matters for editable installs
on filesystems where iCloud / OneDrive can leave orphan dist-info dirs
behind that confuse importlib.metadata's lookup.
"""
try:
v = pkg_version("code-review-graph")
if v:
return v
except PackageNotFoundError as exc:
logger.debug("Package metadata unavailable: %s", exc)
# Fallback: read __version__ directly from the package.
try:
from . import __version__ as fallback_version
if fallback_version:
return fallback_version
except ImportError:
pass
return "dev"
def _supports_color() -> bool:
"""Check if the terminal likely supports ANSI colors."""
if os.environ.get("NO_COLOR"):
return False
if not hasattr(sys.stdout, "isatty"):
return False
return sys.stdout.isatty()
def _configure_utf8_stdio() -> None:
"""Allow Unicode CLI decoration on streams using a legacy encoding."""
for stream in (sys.stdout, sys.stderr):
encoding = getattr(stream, "encoding", None)
if not encoding or encoding.lower().replace("-", "") == "utf8":
continue
reconfigure = getattr(stream, "reconfigure", None)
if reconfigure is None:
continue
try:
reconfigure(encoding="utf-8")
except (OSError, ValueError):
pass
def _print_banner() -> None:
"""Print the startup banner with graph art and available commands."""
color = _supports_color()
version = _get_version()
# ANSI escape codes
c = "\033[36m" if color else "" # cyan — graph art
y = "\033[33m" if color else "" # yellow — center node
b = "\033[1m" if color else "" # bold
d = "\033[2m" if color else "" # dim
g = "\033[32m" if color else "" # green — commands
r = "\033[0m" if color else "" # reset
print(f"""
{c} ●──●──●{r}
{c} │╲ │ ╱│{r} {b}code-review-graph{r} {d}v{version}{r}
{c} ●──{y}{c}──●{r}
{c} │╱ │ ╲│{r} {d}Structural knowledge graph for{r}
{c} ●──●──●{r} {d}smarter code reviews{r}
{b}Commands:{r}
{g}install{r} Set up MCP server for AI coding platforms
{g}init{r} Alias for install
{g}build{r} Full graph build {d}(parse all files){r}
{g}update{r} Incremental update {d}(changed files only){r}
{g}watch{r} Auto-update on file changes
{g}status{r} Show graph statistics
{g}visualize{r} Generate interactive HTML graph
{g}wiki{r} Generate markdown wiki from communities
{g}detect-changes{r} Analyze change impact {d}(risk-scored review){r}
{g}register{r} Register a repository in the multi-repo registry
{g}unregister{r} Remove a repository from the registry
{g}repos{r} List registered repositories
{g}postprocess{r} Run post-processing {d}(flows, communities, FTS){r}
{g}daemon{r} Multi-repo watch daemon management
{g}eval{r} Run evaluation benchmarks
{g}serve{r} Start MCP server {d}(stdio, or {g}--http{r} on localhost:5555){r}
{d}Run{r} {b}code-review-graph <command> --help{r} {d}for details{r}
""")
def _instruction_files_to_modify(
repo_root: Path,
target: str,
) -> list[str]:
"""Return the list of instruction files that ``install`` would write
or modify, given the current state of the repo and the selected
platform target. Used for the dry-run / confirm preview (#173).
A file holding a section from an older release is listed as ``(update)``:
install replaces that block in place rather than leaving it stale (#314).
"""
from .skills import (
_CLAUDE_MD_SECTION,
_CLAUDE_MD_SECTION_MARKER,
_COPILOT_SECTION,
_PLATFORM_INSTRUCTION_CUSTOM_SECTIONS,
_PLATFORM_INSTRUCTION_FILES,
_upgrade_managed_block,
)
targets: list[str] = []
def _describe(filename: str, path: Path, section: str) -> None:
if not path.exists():
targets.append(f"{filename} (new)")
return
content = path.read_text(encoding="utf-8", errors="replace")
if _CLAUDE_MD_SECTION_MARKER not in content:
targets.append(f"{filename} (append)")
elif _upgrade_managed_block(content, section) is not None:
targets.append(f"{filename} (update)")
if target in ("claude", "all"):
_describe("CLAUDE.md", repo_root / "CLAUDE.md", _CLAUDE_MD_SECTION)
for filename, owners in _PLATFORM_INSTRUCTION_FILES.items():
if target != "all" and target not in owners:
continue
section = (
_COPILOT_SECTION
if filename in _PLATFORM_INSTRUCTION_CUSTOM_SECTIONS
else _CLAUDE_MD_SECTION
)
_describe(filename, repo_root / filename, section)
return targets
def _confirm_yes_no(prompt: str, default_yes: bool = True) -> bool:
"""Prompt the user [Y/n] and return True for yes.
Non-interactive environments (no TTY on stdin, e.g. an MCP wrapper
piping the CLI) return ``default_yes`` without blocking — the
stdio transport cannot safely read from stdin without corrupting
the JSON-RPC stream. See: #173, #174
"""
if not sys.stdin.isatty():
return default_yes
suffix = "[Y/n]" if default_yes else "[y/N]"
try:
answer = input(f"{prompt} {suffix} ").strip().lower()
except (EOFError, KeyboardInterrupt):
print()
return False
if not answer:
return default_yes
return answer in ("y", "yes")
def _match_files_to_forget(
stored_files: Iterable[str],
patterns: Iterable[str],
repo_root: Path,
) -> list[str]:
"""Resolve user-supplied paths/globs to stored graph file paths.
The graph keys every parsed file by its absolute path. A user may name a
file with an absolute path, a path relative to the repository root, a
directory whose contents should all be dropped, or a glob pattern. Each
stored file is compared against every pattern in all of those forms and the
sorted set of matching stored paths is returned.
"""
root = repo_root.resolve()
stored = list(stored_files)
matched: set[str] = set()
for raw in patterns:
pattern = str(raw).strip()
if not pattern:
continue
expanded = Path(pattern).expanduser()
absolute = expanded if expanded.is_absolute() else root / expanded
absolute_str = os.path.normpath(str(absolute))
dir_prefix = absolute_str.rstrip(os.sep) + os.sep
for stored_path in stored:
normalised = os.path.normpath(stored_path)
try:
relative = os.path.relpath(normalised, str(root))
except ValueError:
relative = None
# Exact match against the absolute or the repo-relative form.
if normalised == absolute_str:
matched.add(stored_path)
continue
if relative is not None and os.path.normpath(relative) == os.path.normpath(
pattern
):
matched.add(stored_path)
continue
# Every file underneath a named directory.
if normalised.startswith(dir_prefix):
matched.add(stored_path)
continue
# Glob patterns, matched against both the absolute and relative form.
if fnmatch.fnmatch(normalised, absolute_str) or (
relative is not None and fnmatch.fnmatch(relative, pattern)
):
matched.add(stored_path)
return sorted(matched)
def _handle_init(args: argparse.Namespace) -> None:
"""Set up MCP config for detected AI coding platforms."""
from .incremental import ensure_repo_gitignore_excludes_crg, find_repo_root
from .skills import install_platform_configs
repo_root = Path(args.repo) if args.repo else find_repo_root()
if not repo_root:
repo_root = Path.cwd()
dry_run = getattr(args, "dry_run", False)
target = getattr(args, "platform", "all") or "all"
if target == "claude-code":
target = "claude"
auto_yes = getattr(args, "yes", False)
skip_instructions = getattr(args, "no_instructions", False)
print("Installing MCP server config...")
configured = install_platform_configs(repo_root, target=target, dry_run=dry_run)
if not configured:
print("No platforms detected.")
else:
print(f"\nConfigured {len(configured)} platform(s): {', '.join(configured)}")
# Preview the instruction files that would be touched (#173).
instr_targets = _instruction_files_to_modify(repo_root, target)
if instr_targets:
print()
print("Graph instructions will be injected into:")
for t in instr_targets:
print(f" {t}")
if dry_run:
print("\n[dry-run] Would ensure .gitignore ignores .code-review-graph/.")
print("[dry-run] No files were modified.")
return
gitignore_state = ensure_repo_gitignore_excludes_crg(repo_root)
if gitignore_state == "created":
print("Created .gitignore and added .code-review-graph/.")
elif gitignore_state == "updated":
print("Updated .gitignore with .code-review-graph/.")
else:
print(".gitignore already contains .code-review-graph/.")
# Platform-native skills and hooks are installed by default where supported
# so the graph tools are used proactively. Use --no-skills / --no-hooks /
# --no-instructions to opt out.
skip_skills = getattr(args, "no_skills", False)
skip_hooks = getattr(args, "no_hooks", False)
# Legacy: --skills/--hooks/--all still accepted (no-op, everything is default)
from .skills import (
PLATFORMS,
generate_skills,
inject_instruction_files,
install_codebuddy_hooks,
install_codebuddy_skills,
install_codex_hooks,
install_cursor_hooks,
install_gemini_cli_hooks,
install_gemini_cli_skills,
install_git_hook,
install_hermes_skills,
install_hooks,
install_opencode_plugin,
install_qoder_skills,
)
if not skip_skills:
# Claude Code skills are only relevant for Claude (or full install).
if target in ("claude", "all"):
skills_dir = generate_skills(repo_root)
print(f"Generated Claude Code skills in {skills_dir}")
# Gemini CLI skills are workspace-scoped under .gemini/.
if target in ("gemini-cli", "all"):
gemini_skills_dir = install_gemini_cli_skills(repo_root)
print(f"Installed Gemini CLI skills in {gemini_skills_dir}")
# CodeBuddy discovers project skills under .codebuddy/skills/.
if target in ("codebuddy", "all"):
codebuddy_skills_dir = install_codebuddy_skills(repo_root)
print(f"Installed CodeBuddy skills in {codebuddy_skills_dir}")
# Hermes Agent discovers skills under <HERMES_HOME>/skills/.
if target == "hermes" or (target == "all" and PLATFORMS["hermes"]["detect"]()):
hermes_skills_dir = install_hermes_skills(repo_root)
print(f"Installed Hermes Agent skills in {hermes_skills_dir}")
# Confirm before writing instruction files (#173). --yes skips the
# prompt; --no-instructions skips the whole block.
if not skip_instructions and instr_targets:
if auto_yes or _confirm_yes_no(
"Inject graph instructions into the files above?",
default_yes=True,
):
outcomes = inject_instruction_files(repo_root, target=target)
for label, wording in (
("created", "Injected graph instructions into"),
("updated", "Updated graph instructions in"),
):
names = [f for f, o in outcomes.items() if o == label]
if names:
print(f"{wording}: {', '.join(names)}")
# A hand-edited block is never overwritten, so say which file it is
# rather than reporting success the user did not get (#314).
stale = [f for f, o in outcomes.items() if o == "conflict"]
if stale:
print(
"Left edited graph instructions alone in: "
f"{', '.join(stale)}. Delete the section between "
"<!-- code-review-graph MCP tools --> and its closing marker "
"and reinstall to pick up the current text."
)
else:
print("Skipped instruction injection (user declined).")
elif skip_instructions:
print("Skipped instruction injection (--no-instructions).")
# Install Qoder skills (global user-level skills directory)
if not skip_skills and target in ("qoder", "all"):
qoder_skills_dir = install_qoder_skills(repo_root)
if qoder_skills_dir:
print(f"Installed Qoder skills to {qoder_skills_dir}")
if not skip_hooks and target in ("codebuddy", "all"):
try:
codebuddy_settings = install_codebuddy_hooks(repo_root)
print(f"Installed CodeBuddy hooks in {codebuddy_settings}")
except Exception as exc:
logger.warning("Could not install CodeBuddy hooks: %s", exc)
if not skip_hooks and target in ("codex", "all"):
hooks_path = install_codex_hooks(repo_root)
print(f"Installed Codex hooks in {hooks_path}")
git_hook = install_git_hook(repo_root)
if git_hook:
print(f"Installed git pre-commit hook in {git_hook}")
if not skip_hooks and target in ("claude", "qoder", "all"):
platforms_to_install = [target] if target != "all" else ["claude", "qoder"]
for plat in platforms_to_install:
install_hooks(repo_root, platform=plat)
print(f"Installed hooks in {repo_root / f'.{plat}' / 'settings.json'}")
git_hook = install_git_hook(repo_root)
if git_hook:
print(f"Installed git pre-commit hook in {git_hook}")
# Cursor hooks (user-level, only if ~/.cursor exists — matching MCP detect)
if not skip_hooks and target in ("all", "cursor") and PLATFORMS["cursor"]["detect"]():
try:
hooks_path = install_cursor_hooks()
print(f"Installed Cursor hooks in {hooks_path}")
except Exception as exc:
logger.warning("Could not install Cursor hooks: %s", exc)
if not skip_hooks and target in ("gemini-cli", "all"):
try:
gemini_settings = install_gemini_cli_hooks(repo_root)
print(f"Installed Gemini CLI hooks in {gemini_settings}")
except Exception as exc:
logger.warning("Could not install Gemini CLI hooks: %s", exc)
# OpenCode plugin (user-level, gated by same detect() as MCP config)
if not skip_hooks and target in ("all", "opencode") and PLATFORMS["opencode"]["detect"]():
try:
plugin_path = install_opencode_plugin()
print(f"Installed OpenCode plugin in {plugin_path}")
except Exception as exc:
logger.warning("Could not install OpenCode plugin: %s", exc)
print()
print("Next steps:")
print(" 1. code-review-graph build # build the knowledge graph")
print(" 2. Restart your AI coding tool to pick up the new config")
def _handle_data_dir_option(args, repo_root: Path) -> None:
"""Handle --data-dir option by updating registry if specified."""
if hasattr(args, "data_dir") and args.data_dir:
try:
from .registry import Registry
data_dir_path = Path(args.data_dir).expanduser().resolve()
data_dir_path.mkdir(parents=True, exist_ok=True)
Registry().set_data_dir(str(repo_root), str(data_dir_path))
logging.info(f"Graph database will be stored at: {data_dir_path}")
except Exception as exc:
logging.error(f"Failed to set data directory: {exc}")
sys.exit(1)
def _add_embedding_refresh_args(command) -> None:
"""Add explicit, provider-scoped refresh options to a CLI command."""
command.add_argument(
"--embedding-provider",
choices=["local", "openai", "google", "minimax", "voyage"],
default=None,
help=(
"Explicitly refresh an existing embedding index with this provider; "
"requires --embedding-model (default: disabled)"
),
)
command.add_argument(
"--embedding-model",
default=None,
help=(
"Exact model for --embedding-provider. Cloud providers may transmit "
"source-derived text and incur API cost"
),
)
def _embedding_refresh_kwargs(args, parser) -> _EmbeddingRefreshKwargs:
"""Validate the all-or-nothing provider/model opt-in."""
provider = getattr(args, "embedding_provider", None)
model = getattr(args, "embedding_model", None)
if bool(provider) != bool(model):
parser.error(
"--embedding-provider and --embedding-model must be supplied together",
)
if not provider:
return {}
assert isinstance(provider, str)
assert isinstance(model, str)
return {
"embedding_provider": provider,
"embedding_model": model,
}
def _non_negative_int(value: str) -> int:
"""Parse a non-negative integer for bounded CLI output."""
try:
parsed = int(value)
except ValueError as exc:
raise argparse.ArgumentTypeError("must be an integer") from exc
if parsed < 0:
raise argparse.ArgumentTypeError("must be zero or greater")
return parsed
def _positive_int(value: str) -> int:
"""Parse a positive integer for CLI limits."""
parsed = _non_negative_int(value)
if parsed == 0:
raise argparse.ArgumentTypeError("must be greater than zero")
return parsed
_GRAPH_TOOL_COMMANDS = {
"query",
"impact",
"search",
"flows",
"flow",
"communities",
"community",
"architecture",
"large-functions",
"refactor",
}
_PATH_REPO_COMMANDS = frozenset({
"install",
"init",
"uninstall",
"build",
"update",
"postprocess",
"embed",
"watch",
"status",
"forget",
"visualize",
"wiki",
"detect-changes",
"dead-code",
"serve",
"mcp",
*_GRAPH_TOOL_COMMANDS,
})
def _canonicalize_repo_argument(args: argparse.Namespace) -> None:
"""Canonicalize path-valued ``--repo`` arguments in place.
Commands whose ``--repo`` value is a repository *name* rather than a path
(eval configs and daemon log aliases) are deliberately excluded. Every
path consumer receives the same absolute, symlink-resolved spelling before
it opens a database or compares stored paths.
"""
repo = getattr(args, "repo", None)
if args.command in _PATH_REPO_COMMANDS and repo:
args.repo = str(Path(repo).expanduser().resolve())
def _find_explicit_repo_root(start: Path) -> "Path | None":
"""Resolve an explicit --repo for graph-tool commands.
Walks upward from ``start``, stopping at the nearest directory that
contains a ``.code-review-graph``, ``.git``, or ``.svn`` marker. Unlike
``find_repo_root``, a registered subproject (``.code-review-graph``)
counts as a boundary, so a monorepo subdirectory built with
``build --repo mono/module`` resolves to the module — not to the
monorepo's top-level ``.git`` (#697).
"""
current = start.resolve()
if not current.is_dir():
return None
while True:
if any(
(current / marker).exists()
for marker in (".code-review-graph", ".git", ".svn")
):
return current
if current == current.parent:
return None
current = current.parent
def _run_graph_tool_command(args, repo_root: Path) -> None:
"""Run one graph-tool CLI wrapper and emit exactly one JSON value."""
from . import tools
root = str(repo_root)
if args.command == "query":
result = tools.query_graph(
pattern=args.pattern,
target=args.target,
repo_root=root,
)
elif args.command == "impact":
result = tools.get_impact_radius(
changed_files=args.files,
max_depth=args.depth,
max_results=args.max_results,
repo_root=root,
base=args.base,
)
elif args.command == "search":
result = tools.semantic_search_nodes(
query=args.query,
kind=args.kind,
limit=args.limit,
repo_root=root,
)
elif args.command == "flows":
result = tools.list_flows(
repo_root=root,
sort_by=args.sort,
limit=args.limit,
kind=args.kind,
)
elif args.command == "flow":
result = tools.get_flow(
flow_id=args.id,
flow_name=args.name,
include_source=args.source,
repo_root=root,
)
elif args.command == "communities":
result = tools.list_communities_func(
repo_root=root,
sort_by=args.sort,
min_size=args.min_size,
)
elif args.command == "community":
result = tools.get_community_func(
community_name=args.name,
community_id=args.id,
include_members=args.members,
repo_root=root,
)
elif args.command == "architecture":
result = tools.get_architecture_overview_func(
repo_root=root,
detail_level=args.detail_level,
)
elif args.command == "large-functions":
result = tools.find_large_functions(
min_lines=args.min_lines,
kind=args.kind,
file_path_pattern=args.path,
limit=args.limit,
repo_root=root,
)
else:
result = tools.refactor_func(
mode=args.mode,
old_name=args.old_name,
new_name=args.new_name,
kind=args.kind,
file_pattern=args.path,
repo_root=root,
)
print(json.dumps(result, indent=2, default=str))
def main() -> None:
"""Main CLI entry point."""
_configure_utf8_stdio()
ap = argparse.ArgumentParser(
prog="code-review-graph",
description="Persistent incremental knowledge graph for code reviews",
)
ap.add_argument("-v", "--version", action="store_true", help="Show version and exit")
sub = ap.add_subparsers(dest="command")
# install (primary) + init (alias)
install_cmd = sub.add_parser("install", help="Register MCP server with AI coding platforms")
install_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
install_cmd.add_argument(
"--dry-run",
action="store_true",
help="Show what would be done without writing files",
)
install_cmd.add_argument(
"--no-skills",
action="store_true",
help="Skip generating platform-native skill files",
)
install_cmd.add_argument(
"--no-hooks",
action="store_true",
help="Skip installing platform-native hooks",
)
install_cmd.add_argument(
"--no-instructions",
action="store_true",
help="Skip injecting graph instructions into CLAUDE.md / AGENTS.md / etc.",
)
install_cmd.add_argument(
"-y",
"--yes",
action="store_true",
help="Auto-confirm instruction injection without an interactive prompt",
)
# Legacy flags (kept for backwards compat, now no-ops since all is default)
install_cmd.add_argument("--skills", action="store_true", help=argparse.SUPPRESS)
install_cmd.add_argument("--hooks", action="store_true", help=argparse.SUPPRESS)
install_cmd.add_argument(
"--all", action="store_true", dest="install_all", help=argparse.SUPPRESS
)
install_cmd.add_argument(
"--platform",
choices=_PLATFORM_CHOICES,
default="all",
help="Target platform for MCP config (default: all detected)",
)
init_cmd = sub.add_parser("init", help="Alias for install")
init_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
init_cmd.add_argument(
"--dry-run",
action="store_true",
help="Show what would be done without writing files",
)
init_cmd.add_argument(
"--no-skills",
action="store_true",
help="Skip generating platform-native skill files",
)
init_cmd.add_argument(
"--no-hooks",
action="store_true",
help="Skip installing platform-native hooks",
)
init_cmd.add_argument(
"--no-instructions",
action="store_true",
help="Skip injecting graph instructions into CLAUDE.md / AGENTS.md / etc.",
)
init_cmd.add_argument(
"-y",
"--yes",
action="store_true",
help="Auto-confirm instruction injection without an interactive prompt",
)
init_cmd.add_argument("--skills", action="store_true", help=argparse.SUPPRESS)
init_cmd.add_argument("--hooks", action="store_true", help=argparse.SUPPRESS)
init_cmd.add_argument("--all", action="store_true", dest="install_all", help=argparse.SUPPRESS)
init_cmd.add_argument(
"--platform",
choices=_PLATFORM_CHOICES,
default="all",
help="Target platform for MCP config (default: all detected)",
)
uninstall_cmd = sub.add_parser(
"uninstall",
help="Safely remove code-review-graph data, configs, hooks, and generated skills",
)
uninstall_cmd.add_argument(
"--repo",
default=None,
help="Path inside a Git/SVN repository to clean (default: current directory)",
)
uninstall_cmd.add_argument(
"--all-repos",
action="store_true",
help="Also clean every repository listed in the CRG registry",
)
uninstall_cmd.add_argument(
"--keep-data",
action="store_true",
help="Keep graph databases while removing installed integrations",
)
uninstall_cmd.add_argument(
"--keep-user-configs",
action="store_true",
help="Clean repositories only; do not edit files under the user home",
)
uninstall_cmd.add_argument(
"--platform",
choices=_PLATFORM_CHOICES,
default="all",
help="Unbind only this platform's MCP registration and keep the graph "
"data and every other integration. Default: all (full uninstall).",
)
uninstall_cmd.add_argument(
"--dry-run",
action="store_true",
help="Print every planned action without writing or deleting anything",
)
uninstall_cmd.add_argument(
"-y",
"--yes",
action="store_true",
help="Apply without an interactive confirmation",
)
# build
build_cmd = sub.add_parser("build", help="Full graph build (re-parse all files)")
build_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
build_cmd.add_argument("-q", "--quiet", action="store_true", help="Suppress output")
build_cmd.add_argument(
"--skip-flows",
action="store_true",
help="Skip flow/community detection (signatures + FTS only)",
)
build_cmd.add_argument(
"--skip-postprocess",
action="store_true",
help="Skip all post-processing (raw parse only)",
)
build_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
_add_embedding_refresh_args(build_cmd)
# update
update_cmd = sub.add_parser("update", help="Incremental update (only changed files)")
update_cmd.add_argument(
"--base",
default=None,
help="Git diff base (default: the commit the graph was last built at)",
)
update_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
update_cmd.add_argument("-q", "--quiet", action="store_true", help="Suppress output")
update_cmd.add_argument(
"--skip-flows",
action="store_true",
help="Skip flow/community detection (signatures + FTS only)",
)
update_cmd.add_argument(
"--skip-postprocess",
action="store_true",
help="Skip all post-processing (raw parse only)",
)
update_cmd.add_argument(
"--brief",
action="store_true",
help="After re-parsing changed files into the graph, also print the "
"risk summary + Token Savings panel that 'detect-changes --brief' "
"prints. Use this after a rebase or large change set when you "
"want to refresh the graph AND see the impact in one command; "
"use 'detect-changes --brief' alone when the graph is already "
"up to date (analysis only, no re-parse).",
)
update_cmd.add_argument(
"--verify",
action="store_true",
help="Calibrate the estimated savings against tiktoken's "
"cl100k_base tokenizer (the GPT-4 family tokenizer). Adds a "
"second row to the panel with the real token counts. Requires "
"`pip install tiktoken`.",
)
update_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
_add_embedding_refresh_args(update_cmd)
# postprocess
pp_cmd = sub.add_parser(
"postprocess",
help="Run post-processing on existing graph (flows, communities, FTS)",
)
pp_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
pp_cmd.add_argument("--no-flows", action="store_true", help="Skip flow detection")
pp_cmd.add_argument("--no-communities", action="store_true", help="Skip community detection")
pp_cmd.add_argument("--no-fts", action="store_true", help="Skip FTS rebuild")
pp_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
_add_embedding_refresh_args(pp_cmd)
# embed
embed_cmd = sub.add_parser(
"embed",
help="Compute vector embeddings for semantic search",
)
embed_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
embed_cmd.add_argument(
"--provider",
choices=["local", "openai", "google", "minimax", "voyage"],
default=None,
help="Embedding provider (default: local, needs code-review-graph[embeddings])",
)
embed_cmd.add_argument(
"--model",
default=None,
help="Embedding model. For local: HuggingFace ID (default all-MiniLM-L6-v2); "
"for openai/google/minimax/voyage: provider-specific model ID.",
)
embed_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
# watch
watch_cmd = sub.add_parser("watch", help="Watch for changes and auto-update")
watch_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
watch_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
_add_embedding_refresh_args(watch_cmd)
# status
status_cmd = sub.add_parser("status", help="Show graph statistics")
status_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
status_cmd.add_argument("-q", "--quiet", action="store_true", help="Suppress output")
status_cmd.add_argument(
"--json",
action="store_true",
dest="json_output",
help="Output one machine-readable JSON object",
)
status_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
# forget
forget_cmd = sub.add_parser(
"forget",
help="Remove already-parsed files from the graph without a full rebuild",
)
forget_cmd.add_argument(
"paths",
nargs="+",
metavar="PATH",
help="Files, directories, or glob patterns to drop from the graph. "
"Paths may be absolute or relative to the repository root.",
)
forget_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
forget_cmd.add_argument(
"--dry-run",
action="store_true",
help="List the files that would be forgotten without modifying the graph",
)
forget_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
# visualize
vis_cmd = sub.add_parser("visualize", help="Generate interactive HTML graph visualization")
vis_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
vis_cmd.add_argument(
"--mode",
choices=["auto", "full", "community", "file"],
default="auto",
help="Rendering mode: auto (default), full, community, or file",
)
vis_cmd.add_argument(
"--serve",
action="store_true",
help="Start a local HTTP server to view the visualization (localhost:8765)",
)
vis_cmd.add_argument(
"--format",
choices=["html", "json", "graphml", "cypher", "obsidian", "svg"],
default="html",
help="Export format (default: html)",
)
vis_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
# wiki
wiki_cmd = sub.add_parser("wiki", help="Generate markdown wiki from community structure")
wiki_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
wiki_cmd.add_argument(
"--force",
action="store_true",
help="Regenerate all pages even if content unchanged",
)
wiki_cmd.add_argument(
"--data-dir",
default=None,
help="External directory to store graph database (useful for network shares)"
)
# register
register_cmd = sub.add_parser(
"register", help="Register a repository in the multi-repo registry"
)
register_cmd.add_argument("path", help="Path to the repository root")
register_cmd.add_argument("--alias", default=None, help="Short alias for the repository")
# unregister
unregister_cmd = sub.add_parser(
"unregister", help="Remove a repository from the multi-repo registry"
)
unregister_cmd.add_argument("path_or_alias", help="Repository path or alias to remove")
# repos
sub.add_parser("repos", help="List registered repositories")
# eval
eval_cmd = sub.add_parser("eval", help="Run evaluation benchmarks")
eval_cmd.add_argument(
"--benchmark",
default=None,
help="Comma-separated benchmarks to run (token_efficiency, impact_accuracy, "
"agent_baseline, flow_completeness, search_quality, build_performance, "
"multi_hop_retrieval)",
)
eval_cmd.add_argument("--repo", default=None, help="Comma-separated repo config names")
eval_cmd.add_argument("--all", action="store_true", dest="run_all", help="Run all benchmarks")
eval_cmd.add_argument("--report", action="store_true", help="Generate report from results")
eval_cmd.add_argument("--output-dir", default=None, help="Output directory for results")
eval_cmd.add_argument(
"--embed",
action="store_true",
help=(
"Build the vector index after each graph build. Required by the "
"agent_baseline, search_quality and multi_hop_retrieval "
"benchmarks: without it their natural-language questions hit "
"FTS5 only and return zero results (default: disabled)"
),
)
eval_cmd.add_argument(
"--embed-provider",
choices=["local", "openai", "google", "minimax", "voyage"],
default=None,
help="Provider for --embed (default: local, needs "
"code-review-graph[embeddings])",
)
eval_cmd.add_argument(
"--embed-model",
default=None,
help="Model for --embed (default: the provider's own default)",
)
# detect-changes
detect_cmd = sub.add_parser(
"detect-changes",
help="Analyze change impact against the existing graph (read-only). "
"Does NOT re-parse files — for that, use 'update --brief'.",
)
detect_cmd.add_argument("--base", default="HEAD~1", help="Git diff base (default: HEAD~1)")
detect_cmd.add_argument(
"--brief",
action="store_true",
help="Show the risk summary + Token Savings panel instead of the "
"full JSON. Read-only against the existing graph.",
)
detect_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
detect_cmd.add_argument(
"--churn",
action="store_true",
help="Add an opt-in change-frequency term to risk scores. Counts "
"commits per file over 90 days by default; set "
"CRG_CHURN_WINDOW_DAYS to adjust.",
)
detect_cmd.add_argument(
"--verify",
action="store_true",
help="Calibrate the estimated savings against tiktoken's "
"cl100k_base tokenizer (the GPT-4 family tokenizer). Adds a "
"second row to the panel with the real token counts. Requires "
"`pip install tiktoken`.",
)
# enrich (Claude Code PreToolUse hook; reads one JSON object from stdin)
sub.add_parser("enrich", help="Enrich hook input with graph context")
# dead-code
dead_cmd = sub.add_parser(
"dead-code",
help="Find functions/classes with no callers or test references",
)
dead_cmd.add_argument(
"--kind",
choices=["Function", "Class"],
default=None,
help="Filter by node kind",
)
dead_cmd.add_argument(
"--file-pattern",
default=None,
help="Filter by file path substring",
)
dead_cmd.add_argument(
"--limit",
type=_non_negative_int,
default=0,
help="Maximum rows to print (0 = no limit)",
)
dead_cmd.add_argument(
"--json",
action="store_true",
dest="json_output",
help="Output a machine-readable JSON array",
)
dead_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
dead_cmd.add_argument(
"--data-dir",
default=None,
help="External directory containing the graph database",
)
# Graph tool wrappers
query_cmd = sub.add_parser("query", help="Query graph relationships")
query_cmd.add_argument(
"pattern",
choices=[
"callers_of",
"callees_of",
"imports_of",
"importers_of",
"children_of",
"tests_for",
"inheritors_of",
"file_summary",
],
)
query_cmd.add_argument("target", help="Node name, qualified name, or file path")
query_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
impact_cmd = sub.add_parser("impact", help="Analyze the blast radius of changes")
impact_cmd.add_argument(
"--files",
nargs="+",
default=None,
help="Changed files (auto-detected when omitted)",
)
impact_cmd.add_argument("--depth", type=_non_negative_int, default=2)
impact_cmd.add_argument("--max-results", type=_positive_int, default=500)
impact_cmd.add_argument("--base", default="HEAD~1")
impact_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
search_cmd = sub.add_parser("search", help="Search graph entities")
search_cmd.add_argument("query", help="Search string")
search_cmd.add_argument(
"--kind",
choices=["File", "Class", "Function", "Type", "Test"],
default=None,
)
search_cmd.add_argument("--limit", type=_positive_int, default=20)
search_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
flows_cmd = sub.add_parser("flows", help="List stored execution flows")
flows_cmd.add_argument(
"--sort",
choices=["criticality", "depth", "node_count", "file_count", "name"],
default="criticality",
)
flows_cmd.add_argument("--limit", type=_positive_int, default=50)
flows_cmd.add_argument("--kind", default=None, help="Entry-point kind filter")
flows_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
flow_cmd = sub.add_parser("flow", help="Show one stored execution flow")
flow_selector = flow_cmd.add_mutually_exclusive_group(required=True)
flow_selector.add_argument("--id", type=_positive_int, default=None)
flow_selector.add_argument("--name", default=None)
flow_cmd.add_argument("--source", action="store_true", help="Include source snippets")
flow_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
communities_cmd = sub.add_parser("communities", help="List graph communities")
communities_cmd.add_argument(
"--sort",
choices=["size", "cohesion", "name"],
default="size",
)
communities_cmd.add_argument("--min-size", type=_non_negative_int, default=0)
communities_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
community_cmd = sub.add_parser("community", help="Show one graph community")
community_selector = community_cmd.add_mutually_exclusive_group(required=True)
community_selector.add_argument("--id", type=_positive_int, default=None)
community_selector.add_argument("--name", default=None)
community_cmd.add_argument("--members", action="store_true")
community_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
architecture_cmd = sub.add_parser("architecture", help="Show architecture overview")
architecture_cmd.add_argument(
"--detail-level",
choices=["minimal", "standard"],
default="minimal",
)
architecture_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
large_cmd = sub.add_parser("large-functions", help="Find oversized graph nodes")
large_cmd.add_argument("--min-lines", type=_positive_int, default=50)
large_cmd.add_argument(
"--kind",
choices=["Function", "Class", "File", "Test"],
default=None,
)
large_cmd.add_argument("--path", default=None, help="File-path substring filter")
large_cmd.add_argument("--limit", type=_positive_int, default=50)
large_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
refactor_cmd = sub.add_parser("refactor", help="Preview graph-backed refactors")
refactor_cmd.add_argument("mode", choices=["rename", "dead_code", "suggest"])
refactor_cmd.add_argument("--old-name", default=None)
refactor_cmd.add_argument("--new-name", default=None)
refactor_cmd.add_argument(
"--kind",
choices=["Function", "Class"],
default=None,
)
refactor_cmd.add_argument("--path", default=None, help="File-path substring filter")
refactor_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
# serve / mcp
serve_cmd = sub.add_parser(
"serve",
help="Start MCP server (stdio by default, or HTTP on localhost with --http)",
)
serve_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
serve_cmd.add_argument(
"--auto-watch",
action="store_true",
help="Start filesystem watch in a daemon thread while MCP server runs",
)
serve_cmd.add_argument(
"--tools", default=None,
help=(
"Comma-separated list of tool names to expose "
"(e.g. query_graph_tool,semantic_search_nodes_tool). "
"Unlisted tools are removed. Falls back to CRG_TOOLS env var. "
"When unset, all tools are available."
),
)
serve_cmd.add_argument(
"--http",
action="store_true",
help="Listen for MCP over Streamable HTTP on localhost (default port 5555)",
)
serve_cmd.add_argument(
"--host",
default=None,
metavar="ADDR",
help="Bind address for --http (default: 127.0.0.1)",
)
serve_cmd.add_argument(
"--port",
type=int,
default=None,
metavar="PORT",
help="Port for --http (default: 5555)",
)
mcp_cmd = sub.add_parser("mcp", help="Alias for serve")
mcp_cmd.add_argument("--repo", default=None, help="Repository root (auto-detected)")
mcp_cmd.add_argument(
"--auto-watch",
action="store_true",
help="Start filesystem watch in a daemon thread while MCP server runs",
)
# daemon
daemon_cmd = sub.add_parser(
"daemon",
help="Multi-repo watch daemon (start/stop/status/add/remove)",
)
daemon_sub = daemon_cmd.add_subparsers(dest="daemon_command")
daemon_start = daemon_sub.add_parser(
"start",
help="Start the watch daemon",
)
daemon_start.add_argument(
"--foreground",
action="store_true",
help="Run in foreground instead of daemonizing",
)
daemon_sub.add_parser(
"stop",
help="Stop the watch daemon",
)
daemon_restart = daemon_sub.add_parser(
"restart",
help="Restart the watch daemon",
)
daemon_restart.add_argument(
"--foreground",
action="store_true",
help="Run in foreground instead of daemonizing",
)
daemon_sub.add_parser("status", help="Show daemon and watcher status")
daemon_logs = daemon_sub.add_parser(
"logs",
help="View daemon or watcher logs",
)
daemon_logs.add_argument(
"--repo",
default=None,
help="Show logs for a specific repo alias",
)
daemon_logs.add_argument(
"--follow",
action="store_true",
help="Follow log output (tail -f)",
)
daemon_logs.add_argument(
"--lines",
type=int,
default=50,
help="Number of lines to show (default: 50)",
)
daemon_add = daemon_sub.add_parser(
"add",
help="Add a repo to the watch config",
)
daemon_add.add_argument("path", help="Path to the repository")
daemon_add.add_argument(
"--alias",
default=None,
help="Short alias for the repo",
)
daemon_remove = daemon_sub.add_parser(
"remove",
help="Remove a repo from the watch config",
)
daemon_remove.add_argument(
"path_or_alias",
help="Repository path or alias to remove",
)
args = ap.parse_args()
if args.version:
print(f"code-review-graph {_get_version()}")
return
if not args.command:
_print_banner()
return
_canonicalize_repo_argument(args)
if (
args.command == "refactor"
and args.mode == "rename"
and (not args.old_name or not args.new_name)
):
refactor_cmd.error("rename requires --old-name and --new-name")
if args.command == "enrich":
from .enrich import run_hook
run_hook()
return
if args.command in _GRAPH_TOOL_COMMANDS:
from .incremental import find_project_root, get_db_path
if args.repo:
# For an explicit --repo the walk must treat .code-review-graph
# as a project boundary too: the plain .git/.svn walk resolves a
# registered monorepo subdirectory to the monorepo root and the
# graph built at the --repo path is never found (#697). Nearest
# marker wins, so pointing inside a repo still works.
repo_root = _find_explicit_repo_root(Path(args.repo).expanduser())
if repo_root is None:
print(
f"--repo does not look like a project root (no .git, .svn, "
f"or .code-review-graph found at or above): {args.repo}",
file=sys.stderr,
)
raise SystemExit(1)
else:
repo_root = find_project_root()
db_path = get_db_path(repo_root)
if not db_path.exists():
print(
f"No graph found at {db_path}. Run `code-review-graph build` first.",
file=sys.stderr,
)
raise SystemExit(1)
_run_graph_tool_command(args, repo_root)
return
embedding_refresh_kwargs = _embedding_refresh_kwargs(args, ap)
if args.command in ("serve", "mcp"):
from .main import main as serve_main
auto_watch = getattr(args, "auto_watch", False)
if args.command == "serve":
if args.port is not None and not args.http:
serve_cmd.error("--port requires --http")
if args.host is not None and not args.http:
serve_cmd.error("--host requires --http")
if args.http:
host = args.host if args.host is not None else "127.0.0.1"
port = args.port if args.port is not None else 5555
serve_main(
repo_root=args.repo,
auto_watch=auto_watch,
transport="streamable-http",
host=host,
port=port,
tools=args.tools,
)
else:
serve_main(repo_root=args.repo, auto_watch=auto_watch, tools=args.tools)
else:
serve_main(repo_root=args.repo, auto_watch=auto_watch)
return
if args.command == "daemon":
if not args.daemon_command:
daemon_cmd.print_help()
return
from .daemon_cli import (
_handle_add,
_handle_logs,
_handle_remove,
_handle_restart,
_handle_start,
_handle_status,
_handle_stop,
)
handlers = {
"start": _handle_start,
"stop": _handle_stop,
"restart": _handle_restart,
"status": _handle_status,
"logs": _handle_logs,
"add": _handle_add,
"remove": _handle_remove,
}
handler = handlers.get(args.daemon_command)
if handler:
handler(args)
return
if args.command == "eval":
from .eval.reporter import generate_full_report, generate_readme_tables
from .eval.runner import run_eval
if getattr(args, "report", False):
output_dir = Path(getattr(args, "output_dir", None) or "evaluate/results")
report = generate_full_report(output_dir)
report_path = Path("evaluate/reports/summary.md")
report_path.parent.mkdir(parents=True, exist_ok=True)
report_path.write_text(report, encoding="utf-8")
print(f"Report written to {report_path}")
tables = generate_readme_tables(output_dir)
print("\n--- README Tables (copy-paste) ---\n")
print(tables)
else:
repos = (
[r.strip() for r in args.repo.split(",")] if getattr(args, "repo", None) else None
)
benchmarks = (
[b.strip() for b in args.benchmark.split(",")]
if getattr(args, "benchmark", None)
else None
)
if not repos and not benchmarks and not getattr(args, "run_all", False):
print("Specify --all, --repo, or --benchmark. See --help.")
return
results = run_eval(
repos=repos,
benchmarks=benchmarks,
output_dir=getattr(args, "output_dir", None),
embed=getattr(args, "embed", False),
embedding_provider=getattr(args, "embed_provider", None),
embedding_model=getattr(args, "embed_model", None),
)
print(f"\nCompleted {len(results)} benchmark(s).")
print("Run 'code-review-graph eval --report' to generate tables.")
return
if args.command == "uninstall":
from .uninstall import UninstallReport
from .uninstall import run as run_uninstall
target_repo = Path(args.repo).expanduser() if args.repo else None
platform_target = getattr(args, "platform", "all") or "all"
scoped_platforms = None if platform_target == "all" else [platform_target]
options = {
"repo": target_repo,
"all_repos": args.all_repos,
"keep_data": args.keep_data,
"keep_user_configs": args.keep_user_configs,
"platforms": scoped_platforms,
}
def _print_report(report: UninstallReport) -> None:
for action in report.removed_paths:
print(f" delete {action}")
for action in report.edited_paths:
print(f" edit {action}")
for action in report.skipped_paths:
print(f" skip {action}")
for error in report.errors:
print(f" error {error}")
preview = run_uninstall(**options, dry_run=True)
if scoped_platforms:
print(f"code-review-graph unbind ({platform_target}) — planned actions:")
else:
print("code-review-graph uninstall — planned actions:")
_print_report(preview)
if preview.total_actions == 0:
if preview.errors:
raise SystemExit(1)
if scoped_platforms:
print(
f" (nothing to do — {platform_target} has no "
"code-review-graph MCP registration)"
)
else:
print(" (nothing to do — no code-review-graph artifacts found)")
return
if args.dry_run:
print("\n[dry-run] No changes made.")
if preview.errors:
raise SystemExit(1)
return
action_word = "unbind" if scoped_platforms else "uninstall"
if not args.yes and not _confirm_yes_no(
f"\nProceed with {action_word}?", default_yes=False
):
print("Aborted.")
return
uninstall_result = run_uninstall(**options, dry_run=False)
print("\nApplied actions:")
_print_report(uninstall_result)
print(
f"Done. Removed {len(uninstall_result.removed_paths)} path(s); "
f"edited {len(uninstall_result.edited_paths)} shared file(s)."
)
if uninstall_result.errors:
raise SystemExit(1)
return
if args.command in ("init", "install"):
_handle_init(args)
return
if args.command in ("register", "unregister", "repos"):
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
from .registry import Registry
registry = Registry()
if args.command == "register":
try:
entry = registry.register(args.path, alias=args.alias)
alias_info = f" (alias: {entry['alias']})" if entry.get("alias") else ""
print(f"Registered: {entry['path']}{alias_info}")
except ValueError as exc:
logging.error(str(exc))
sys.exit(1)
elif args.command == "unregister":
if registry.unregister(args.path_or_alias):
print(f"Unregistered: {args.path_or_alias}")
else:
print(f"Not found: {args.path_or_alias}")
sys.exit(1)
elif args.command == "repos":
repos = registry.list_repos()
if not repos:
print("No repositories registered.")
print("Use: code-review-graph register <path> [--alias name]")
else:
for entry in repos:
alias = entry.get("alias", "")
alias_str = f" ({alias})" if alias else ""
print(f" {entry['path']}{alias_str}")
return
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
from .graph import GraphStore
from .incremental import (
find_project_root,
find_repo_root,
get_db_path,
watch,
)
if args.command == "postprocess":
repo_root = Path(args.repo) if args.repo else find_project_root()
_handle_data_dir_option(args, repo_root)
db_path = get_db_path(repo_root)
store = GraphStore(db_path)
try:
from .tools.build import run_postprocess
result = run_postprocess(
flows=not getattr(args, "no_flows", False),
communities=not getattr(args, "no_communities", False),
fts=not getattr(args, "no_fts", False),
repo_root=str(repo_root),
**embedding_refresh_kwargs,
)
parts = []
if result.get("flows_detected"):
parts.append(f"{result['flows_detected']} flows")
if result.get("communities_detected"):
parts.append(f"{result['communities_detected']} communities")
if result.get("fts_indexed"):
parts.append(f"{result['fts_indexed']} FTS entries")
print(f"Post-processing: {', '.join(parts) or 'done'}")
finally:
store.close()
return
if args.command == "embed":
repo_root = Path(args.repo) if args.repo else find_project_root()
_handle_data_dir_option(args, repo_root)
from .tools.docs import embed_graph
result = embed_graph(
repo_root=str(repo_root),
model=args.model,
provider=args.provider,
)
if result.get("status") == "error":
logging.error(result.get("error", "embed_graph failed"))
sys.exit(1)
print(result.get("summary", "Embedding done."))
return
if args.command in ("update", "detect-changes"):
# update and detect-changes require git for diffing
repo_root = Path(args.repo) if args.repo else find_repo_root()
if not repo_root:
logging.error(
"Not in a git repository. '%s' requires git for diffing.",
args.command,
)
logging.error("Use 'build' for a full parse, or run 'git init' first.")
sys.exit(1)
elif args.command == "dead-code":
requested_root = Path(args.repo).expanduser() if args.repo else None
repo_root = find_project_root(requested_root)
else:
repo_root = Path(args.repo) if args.repo else find_project_root()
# Handle --data-dir for commands that support it
_data_dir_cmds = (
"build",
"update",
"detect-changes",
"status",
"forget",
"watch",
"visualize",
"wiki",
"dead-code",
)
# Read-only consumers must not create graph.db / data dirs / registry
# entries when the graph is missing (follow-up to #777 / #782; see #803).
_read_only_db_cmds = frozenset({
"status",
"detect-changes",
"visualize",
"wiki",
"watch",
})
explicit_data_dir = bool(getattr(args, "data_dir", None))
read_only_explicit_data_dir = (
args.command in _read_only_db_cmds and explicit_data_dir
)
if args.command in _data_dir_cmds and not read_only_explicit_data_dir:
_handle_data_dir_option(args, repo_root)
if args.command in _read_only_db_cmds:
if read_only_explicit_data_dir:
db_path = Path(args.data_dir).expanduser().resolve() / "graph.db"
else:
db_path = get_db_path(repo_root, read_only=True)
legacy_db = repo_root / ".code-review-graph.db"
default_db = repo_root / ".code-review-graph" / "graph.db"
if (
not read_only_explicit_data_dir
and not db_path.exists()
and db_path.resolve() == default_db.resolve()
and legacy_db.exists()
):
# Preserve the established one-time legacy migration, but do not
# materialize graph state when neither database exists.
db_path = get_db_path(repo_root)
else:
db_path = get_db_path(repo_root)
if (
args.command in ("dead-code", "forget", *_read_only_db_cmds)
and not db_path.exists()
):
print(
f"No graph found at {db_path}. Run `code-review-graph build` first.",
file=sys.stderr,
)
raise SystemExit(1)
store = GraphStore(db_path)
try:
if args.command == "dead-code":
from .refactor import find_dead_code
items = find_dead_code(
store,
kind=args.kind,
file_pattern=args.file_pattern,
root=repo_root,
)
total = len(items)
shown = items[: args.limit] if args.limit else items
if args.json_output:
print(json.dumps(shown, indent=2))
else:
print(f"Dead code: {total} item(s); showing {len(shown)}")
for item in shown:
kind = item.get("kind", "?")
name = item.get("name", "?")
file_path = item.get("relative_path") or item.get("file", "?")
line = item.get("line", "?")
print(f" [{kind}] {name} ({file_path}:{line})")
elif args.command == "build":
pp = (
"none"
if getattr(args, "skip_postprocess", False)
else ("minimal" if getattr(args, "skip_flows", False) else "full")
)
from .tools.build import build_or_update_graph
previous_disable = logging.root.manager.disable
if args.quiet:
logging.disable(logging.INFO)
try:
result = build_or_update_graph(
full_rebuild=True,
repo_root=str(repo_root),
postprocess=pp,
**embedding_refresh_kwargs,
)
finally:
logging.disable(previous_disable)
parsed = result.get("files_parsed", 0)
nodes = result.get("total_nodes", 0)
edges = result.get("total_edges", 0)
if not args.quiet:
print(
f"Full build: {parsed} files, {nodes} nodes, {edges} edges "
f"(postprocess={pp})"
)
if result.get("errors"):
print(f"Errors: {len(result['errors'])}")
elif args.command == "update":
pp = (
"none"
if getattr(args, "skip_postprocess", False)
else ("minimal" if getattr(args, "skip_flows", False) else "full")
)
from .tools.build import build_or_update_graph
previous_disable = logging.root.manager.disable
if args.quiet:
logging.disable(logging.INFO)
try:
result = build_or_update_graph(
full_rebuild=False,
repo_root=str(repo_root),
base=args.base,
postprocess=pp,
**embedding_refresh_kwargs,
)
except RuntimeError as exc:
print(f"Error: {exc}", file=sys.stderr)
sys.exit(1)
finally:
logging.disable(previous_disable)
nodes = result.get("total_nodes", 0)
edges = result.get("total_edges", 0)
if not args.quiet:
if result.get("build_type") == "full":
# No usable incremental base (fresh/legacy graph, or the
# last-synced commit was lost to a rewrite/shallow clone),
# so the update fell back to a full rebuild.
parsed = result.get("files_parsed", 0)
print(
f"Full rebuild (no usable incremental base): "
f"{parsed} files, {nodes} nodes, {edges} edges"
f" (postprocess={pp})"
)
else:
updated = result.get("files_updated", 0)
print(
f"Incremental: {updated} files updated, "
f"{nodes} nodes, {edges} edges"
f" (postprocess={pp})"
)
# --brief: append a one-line change-impact summary with the same
# estimated context-savings approximation that detect-changes uses.
# Same baseline (changed files vs analysis response), so the two
# commands are directly comparable.
if getattr(args, "brief", False) and not args.quiet:
from .changes import analyze_changes
from .context_savings import (
attach_context_savings,
estimate_file_tokens,
format_context_savings_panel,
)
from .incremental import (
get_changed_files,
get_staged_and_unstaged,
)
# Reuse the base the update actually resolved to (args.base is
# None by default now, which get_changed_files cannot accept).
brief_base = result.get("base_resolved") or "HEAD~1"
changed = get_changed_files(repo_root, brief_base)
if not changed:
changed = get_staged_and_unstaged(repo_root)
if changed:
impact = analyze_changes(
store,
changed,
repo_root=str(repo_root),
base=brief_base,
)
original_tokens = estimate_file_tokens(repo_root, changed)
attach_context_savings(
impact,
original_tokens=original_tokens,
)
summary = impact.get("summary", "")
if summary:
print(summary)
verified = None
if getattr(args, "verify", False):
from .context_savings import verify_with_tiktoken
verified = verify_with_tiktoken(
repo_root, changed, impact,
)
if verified is None:
print(
"Note: --verify requires tiktoken. "
"Install with `pip install tiktoken`.",
)
panel = format_context_savings_panel(
impact.get("context_savings"),
original_tokens=original_tokens,
response=impact,
verified=verified,
)
if panel:
print(panel)
elif args.command == "status":
stats = store.get_stats()
stored_branch = store.get_metadata("git_branch")
stored_sha = store.get_metadata("git_head_sha")
from .incremental import _git_branch_info, detect_vcs
vcs = detect_vcs(repo_root)
current_branch = None
current_sha = None
if vcs == "git":
current_branch, current_sha = _git_branch_info(repo_root)
stored_svn_branch = store.get_metadata("svn_branch")
stored_rev = store.get_metadata("svn_revision")
if args.json_output:
print(json.dumps({
"nodes": stats.total_nodes,
"edges": stats.total_edges,
"files": stats.files_count,
"languages": list(stats.languages),
"last_updated": stats.last_updated,
"vcs": vcs,
"built_on_branch": stored_branch,
"built_at_commit": stored_sha,
"current_branch": current_branch,
"current_sha": current_sha,
"svn_branch": stored_svn_branch,
"svn_revision": stored_rev,
}))
elif not args.quiet:
print(f"Nodes: {stats.total_nodes}")
print(f"Edges: {stats.total_edges}")
print(f"Files: {stats.files_count}")
print(f"Languages: {', '.join(stats.languages)}")
print(f"Last updated: {stats.last_updated or 'never'}")
if stored_branch:
print(f"Built on branch: {stored_branch}")
if stored_sha:
print(f"Built at commit: {stored_sha[:12]}")
if stored_branch and current_branch and stored_branch != current_branch:
print(
f"WARNING: Graph was built on '{stored_branch}' "
f"but you are now on '{current_branch}'. "
f"Run 'code-review-graph build' to rebuild."
)
if vcs == "svn":
if stored_svn_branch:
print(f"SVN branch: {stored_svn_branch}")
if stored_rev:
print(f"SVN revision at build: {stored_rev}")
elif args.command == "forget":
stored_files = store.get_all_files()
targets = _match_files_to_forget(stored_files, args.paths, repo_root)
if not targets:
print("No parsed files matched the given path(s).")
print(f"The graph currently tracks {len(stored_files)} file(s).")
else:
header = (
"[dry-run] Would forget these files:"
if args.dry_run
else "Forgetting these files:"
)
print(header)
for file_path in targets:
try:
display = os.path.relpath(file_path, str(repo_root))
except ValueError:
display = file_path
print(f" {display}")
if args.dry_run:
print(
f"\n[dry-run] {len(targets)} file(s) would be removed "
"from the graph. No changes made."
)
else:
from .forget import forget_files
summary = forget_files(store, repo_root, targets)
reparsed = summary.get("reparsed", [])
if reparsed:
print(
f" re-resolved {len(reparsed)} referring file(s) "
"so no edges dangle"
)
remaining = len(stored_files) - len(targets)
print(
f"\nForgot {len(targets)} file(s); "
f"{remaining} file(s) remain in the graph."
)
elif args.command == "watch":
from .postprocessing import run_post_processing
try:
callback = (
partial(run_post_processing, **embedding_refresh_kwargs)
if embedding_refresh_kwargs
else run_post_processing
)
watch(repo_root, store, on_files_updated=callback)
except RuntimeError as exc:
print(f"Error: {exc}", file=sys.stderr)
sys.exit(1)
elif args.command == "visualize":
from .incremental import get_data_dir
# Prefer an explicit --data-dir so read-only resolution still
# writes exports next to the graph without registry side-effects.
if getattr(args, "data_dir", None):
data_dir = Path(args.data_dir).expanduser().resolve()
data_dir.mkdir(parents=True, exist_ok=True)
else:
data_dir = get_data_dir(repo_root)
fmt = getattr(args, "format", "html") or "html"
if fmt == "json":
from .exports import export_json
out = data_dir / "graph.json"
export_json(store, out)
print(f"JSON exported: {out}")
elif fmt == "graphml":
from .exports import export_graphml
out = data_dir / "graph.graphml"
export_graphml(store, out)
print(f"GraphML exported: {out}")
elif fmt == "cypher":
from .exports import export_neo4j_cypher
out = data_dir / "graph.cypher"
export_neo4j_cypher(store, out)
print(f"Neo4j Cypher exported: {out}")
elif fmt == "obsidian":
from .exports import export_obsidian_vault
out = data_dir / "obsidian"
export_obsidian_vault(store, out)
print(f"Obsidian vault exported: {out}")
elif fmt == "svg":
from .exports import export_svg
out = data_dir / "graph.svg"
export_svg(store, out)
print(f"SVG exported: {out}")
else:
from .visualization import generate_html
html_path = data_dir / "graph.html"
vis_mode = getattr(args, "mode", "auto") or "auto"
generate_html(store, html_path, mode=vis_mode)
print(f"Visualization ({vis_mode}): {html_path}")
if getattr(args, "serve", False):
import functools
import http.server
serve_dir = html_path.parent
port = 8765
http_handler = functools.partial(
http.server.SimpleHTTPRequestHandler,
directory=str(serve_dir),
)
print(f"Serving at http://localhost:{port}/graph.html")
print("Press Ctrl+C to stop.")
with http.server.HTTPServer(("localhost", port), http_handler) as httpd:
try:
httpd.serve_forever()
except KeyboardInterrupt:
print("\nServer stopped.")
else:
print("Open in browser to explore.")
elif args.command == "wiki":
from .incremental import get_data_dir
from .wiki import generate_wiki
if getattr(args, "data_dir", None):
data_dir = Path(args.data_dir).expanduser().resolve()
data_dir.mkdir(parents=True, exist_ok=True)
else:
data_dir = get_data_dir(repo_root)
wiki_dir = data_dir / "wiki"
result = generate_wiki(store, wiki_dir, force=args.force)
total = result["pages_generated"] + result["pages_updated"] + result["pages_unchanged"]
print(
f"Wiki: {result['pages_generated']} new, "
f"{result['pages_updated']} updated, "
f"{result['pages_unchanged']} unchanged "
f"({total} total pages)"
)
print(f"Output: {wiki_dir}")
elif args.command == "detect-changes":
from .changes import analyze_changes
from .context_savings import (
attach_context_savings,
estimate_file_tokens,
)
from .incremental import get_changed_files, get_staged_and_unstaged
base = args.base
changed = get_changed_files(repo_root, base)
if not changed:
changed = get_staged_and_unstaged(repo_root)
if not changed:
print("No changes detected.")
else:
result = analyze_changes(
store,
changed,
repo_root=str(repo_root),
base=base,
include_churn=getattr(args, "churn", False),
)
original_tokens = estimate_file_tokens(repo_root, changed)
attach_context_savings(
result,
original_tokens=original_tokens,
)
if args.brief:
from .context_savings import (
format_context_savings_panel,
verify_with_tiktoken,
)
print(result.get("summary", "No summary available."))
verified = None
if getattr(args, "verify", False):
verified = verify_with_tiktoken(repo_root, changed, result)
if verified is None:
print(
"Note: --verify requires tiktoken. "
"Install with `pip install tiktoken`.",
)
panel = format_context_savings_panel(
result.get("context_savings"),
original_tokens=original_tokens,
response=result,
verified=verified,
)
if panel:
print(panel)
else:
print(json.dumps(result, indent=2, default=str))
finally:
store.close()