#!/usr/bin/env python3 """Collect Git changes or one commit into deterministic, hunk-addressable JSON. Only read-only Git commands are used. All patch strings in the output are byte-for-byte decodings of Git diff stdout; the script never reconstructs source patches. """ from __future__ import annotations import argparse import hashlib import json import os import re import subprocess import sys import tempfile from dataclasses import dataclass from pathlib import Path from typing import Iterable, Sequence SCHEMA_VERSION = 1 REVIEW_DIR = ".semantic-review" EXCLUDE_PATHSPEC = ":(exclude).semantic-review/**" DIFF_OPTIONS = ( "--no-ext-diff", "--no-textconv", "--no-color", "--binary", "--full-index", "--find-renames=50%", "--diff-algorithm=histogram", "--unified=3", "--src-prefix=a/", "--dst-prefix=b/", "--submodule=short", ) HUNK_HEADER = re.compile(r"^(@{2,}) .*? \1(?:.*)(?:\r?\n)?$") NORMALIZE_HEADER = re.compile(r"^(@{2,}) .*? \1(.*?)(\r?\n)?$") class CollectionError(RuntimeError): """Raised when Git output cannot be collected safely.""" @dataclass(frozen=True) class ChangedPath: status: str old_path: str new_path: str @dataclass class PendingHunk: scope: str status: str old_path: str new_path: str kind: str header: str patch: str additions: int deletions: int sequence: int identity_material: str = "" hunk_id: str = "" def git_env() -> dict[str, str]: env = os.environ.copy() env.update( { "LC_ALL": "C", "LANG": "C", "GIT_OPTIONAL_LOCKS": "0", "GIT_PAGER": "cat", "GIT_EXTERNAL_DIFF": "", } ) return env def git_executable() -> str: """Return Git executable, with a narrowly named override for hermetic tests.""" return os.environ.get("SEMANTIC_REVIEW_GIT", "git") def run_git( repo: Path, args: Sequence[str], *, allow_diff_exit: bool = False, ) -> bytes: command = [git_executable(), "-C", os.fspath(repo), *args] completed = subprocess.run( command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, env=git_env(), check=False, ) accepted = {0, 1} if allow_diff_exit else {0} if completed.returncode not in accepted: detail = completed.stderr.decode("utf-8", "replace").strip() raise CollectionError( f"Git command failed ({completed.returncode}): {' '.join(command)}" + (f"\n{detail}" if detail else "") ) return completed.stdout def repository_root(repo_arg: str) -> Path: candidate = Path(repo_arg).expanduser().resolve() output = run_git(candidate, ("rev-parse", "--show-toplevel")) return Path(output.decode("utf-8", "surrogateescape").rstrip("\n")).resolve() def head_oid(root: Path) -> str | None: completed = subprocess.run( [git_executable(), "-C", os.fspath(root), "rev-parse", "--verify", "HEAD"], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, env=git_env(), check=False, ) if completed.returncode != 0: return None return completed.stdout.decode("ascii", "strict").strip() def decode_path(raw: bytes) -> str: return raw.decode("utf-8", "surrogateescape") def parse_name_status(raw: bytes) -> list[ChangedPath]: fields = raw.split(b"\0") if fields and fields[-1] == b"": fields.pop() changes: list[ChangedPath] = [] index = 0 while index < len(fields): status = fields[index].decode("ascii", "replace") index += 1 if not status: raise CollectionError("Git emitted an empty name-status record") if status[0] in {"R", "C"}: if index + 1 >= len(fields): raise CollectionError("Git emitted a truncated rename/copy record") old_path = decode_path(fields[index]) new_path = decode_path(fields[index + 1]) index += 2 else: if index >= len(fields): raise CollectionError("Git emitted a truncated name-status record") path = decode_path(fields[index]) index += 1 old_path = path new_path = path changes.append(ChangedPath(status, old_path, new_path)) return changes def literal_pathspec(path: str) -> str: return f":(literal){path}" def tracked_changes(root: Path, scope: str) -> list[ChangedPath]: return compared_changes(root, scope, ()) def compared_changes( root: Path, scope: str, comparison: Sequence[str], ) -> list[ChangedPath]: cached = ("--cached",) if scope == "staged" else () output = run_git( root, ( "diff", *cached, *DIFF_OPTIONS, "--name-status", "-z", *comparison, "--", ".", EXCLUDE_PATHSPEC, ), ) return parse_name_status(output) def tracked_patch( root: Path, scope: str, change: ChangedPath, comparison: Sequence[str] = (), ) -> str: cached = ("--cached",) if scope == "staged" else () paths = [literal_pathspec(change.old_path)] if change.new_path != change.old_path: paths.append(literal_pathspec(change.new_path)) output = run_git( root, ("diff", *cached, *DIFF_OPTIONS, *comparison, "--", *paths), ) return output.decode("utf-8", "surrogateescape") def untracked_paths(root: Path) -> list[str]: output = run_git( root, ( "ls-files", "--others", "--exclude-standard", "-z", "--", ".", EXCLUDE_PATHSPEC, ), ) paths = [decode_path(item) for item in output.split(b"\0") if item] return sorted(paths, key=lambda item: item.encode("utf-8", "surrogateescape")) def untracked_patch(root: Path, path: str) -> str: output = run_git( root, ("diff", "--no-index", *DIFF_OPTIONS, "--", "/dev/null", path), allow_diff_exit=True, ) return output.decode("utf-8", "surrogateescape") def is_hunk_header(line: str) -> bool: return bool(HUNK_HEADER.match(line)) def normalize_hunk_header(header: str) -> str: match = NORMALIZE_HEADER.match(header) if not match: return header.rstrip("\r\n") marker, context, _newline = match.groups() return f"{marker} {marker}{context}" def line_stats(lines: Iterable[str]) -> tuple[int, int]: additions = 0 deletions = 0 for line in lines: if line.startswith("+") and not line.startswith("+++"): additions += 1 elif line.startswith("-") and not line.startswith("---"): deletions += 1 return additions, deletions def split_patch( scope: str, change: ChangedPath, patch: str, ) -> list[PendingHunk]: lines = patch.splitlines(keepends=True) starts = [index for index, line in enumerate(lines) if is_hunk_header(line)] if not starts: kind = "empty" if not patch else "binary-or-metadata" additions, deletions = line_stats(lines) return [ PendingHunk( scope=scope, status=change.status, old_path=change.old_path, new_path=change.new_path, kind=kind, header="", patch=patch, additions=additions, deletions=deletions, sequence=1, ) ] prelude = "".join(lines[: starts[0]]) hunks: list[PendingHunk] = [] for sequence, start in enumerate(starts, start=1): end = starts[sequence] if sequence < len(starts) else len(lines) hunk_lines = lines[start:end] additions, deletions = line_stats(hunk_lines[1:]) hunks.append( PendingHunk( scope=scope, status=change.status, old_path=change.old_path, new_path=change.new_path, kind="text", header=hunk_lines[0].rstrip("\r\n"), patch=prelude + "".join(hunk_lines), additions=additions, deletions=deletions, sequence=sequence, ) ) return hunks def identity_material(hunk: PendingHunk) -> str: lines = hunk.patch.splitlines(keepends=True) if hunk.kind == "text": first_hunk = next( (index for index, line in enumerate(lines) if is_hunk_header(line)), len(lines), ) body = "".join(lines[first_hunk + 1 :]) content = normalize_hunk_header(hunk.header) + "\n" + body else: content = hunk.patch return "\0".join( ( hunk.scope, hunk.status, hunk.old_path, hunk.new_path, hunk.kind, content, ) ) def assign_ids(hunks: list[PendingHunk]) -> None: buckets: dict[str, list[PendingHunk]] = {} for hunk in hunks: hunk.identity_material = identity_material(hunk) digest = hashlib.sha256( hunk.identity_material.encode("utf-8", "surrogateescape") ).hexdigest().upper() buckets.setdefault(digest, []).append(hunk) used: set[str] = set() for digest in sorted(buckets): bucket = buckets[digest] if len(bucket) == 1: candidates = [(bucket[0], f"H-{digest[:16]}")] else: candidates = [] for hunk in bucket: discriminator = hashlib.sha256( (hunk.header + "\0" + hunk.patch).encode( "utf-8", "surrogateescape" ) ).hexdigest().upper() candidates.append((hunk, f"H-{digest[:12]}-{discriminator[:8]}")) candidates.sort(key=lambda pair: (pair[1], pair[0].sequence)) for duplicate_index, (hunk, candidate) in enumerate(candidates, start=1): hunk_id = candidate if hunk_id in used: hunk_id = f"{candidate}-{duplicate_index}" if hunk_id in used: raise CollectionError("Unable to assign unique stable hunk IDs") hunk.hunk_id = hunk_id used.add(hunk_id) def collect_worktree(root: Path) -> list[PendingHunk]: hunks: list[PendingHunk] = [] for scope in ("staged", "unstaged"): for change in tracked_changes(root, scope): hunks.extend(split_patch(scope, change, tracked_patch(root, scope, change))) for path in untracked_paths(root): change = ChangedPath("A", "/dev/null", path) hunks.extend(split_patch("untracked", change, untracked_patch(root, path))) assign_ids(hunks) return hunks def resolve_commit(root: Path, revision: str) -> str: if not revision.strip(): raise CollectionError("Commit revision must not be empty") output = run_git( root, ("rev-parse", "--verify", "--end-of-options", f"{revision}^{{commit}}"), ) return output.decode("ascii", "strict").strip() def commit_base(root: Path, commit_oid: str) -> str: output = run_git(root, ("rev-list", "--parents", "-n", "1", commit_oid)) parts = output.decode("ascii", "strict").strip().split() if not parts or parts[0] != commit_oid: raise CollectionError(f"Unable to resolve parents for commit {commit_oid}") if len(parts) > 1: return parts[1] empty_tree = run_git(root, ("hash-object", "-t", "tree", "/dev/null")) return empty_tree.decode("ascii", "strict").strip() def collect_commit( root: Path, revision: str, ) -> tuple[list[PendingHunk], str, str]: commit_oid = resolve_commit(root, revision) base_oid = commit_base(root, commit_oid) comparison = (base_oid, commit_oid) hunks: list[PendingHunk] = [] for change in compared_changes(root, "commit", comparison): hunks.extend( split_patch( "commit", change, tracked_patch(root, "commit", change, comparison), ) ) assign_ids(hunks) return hunks, commit_oid, base_oid def patch_sha256(patch: str) -> str: return hashlib.sha256(patch.encode("utf-8", "surrogateescape")).hexdigest() def build_document( root: Path, hunks: list[PendingHunk], target: dict[str, str], ) -> dict[str, object]: records = [ { "id": hunk.hunk_id, "scope": hunk.scope, "status": hunk.status, "old_path": hunk.old_path, "new_path": hunk.new_path, "kind": hunk.kind, "header": hunk.header, "additions": hunk.additions, "deletions": hunk.deletions, "patch_sha256": patch_sha256(hunk.patch), "patch": hunk.patch, } for hunk in hunks ] evidence = json.dumps(records, ensure_ascii=True, sort_keys=True, separators=(",", ":")) return { "schema_version": SCHEMA_VERSION, "generator": "semantic-diff-review/collect_changes.py", "repository": { "root": os.fspath(root), "head": head_oid(root), "target": target, }, "evidence_sha256": hashlib.sha256(evidence.encode("ascii")).hexdigest(), "hunks": records, } def atomic_write_json(path: Path, document: dict[str, object]) -> None: path.parent.mkdir(parents=True, exist_ok=True) rendered = json.dumps(document, ensure_ascii=True, indent=2, sort_keys=False) + "\n" with tempfile.NamedTemporaryFile( mode="w", encoding="utf-8", dir=path.parent, prefix=f".{path.name}.", suffix=".tmp", delete=False, ) as handle: temp_path = Path(handle.name) handle.write(rendered) handle.flush() os.fsync(handle.fileno()) os.replace(temp_path, path) def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--repo", default=".", help="Path inside the Git repository") parser.add_argument( "--output", help="Output path (default: /.semantic-review/changes.json)", ) parser.add_argument( "--commit", nargs="?", const="HEAD", metavar="REV", help=( "Collect one commit against its first parent instead of working-tree " "changes; omit REV to review HEAD" ), ) return parser.parse_args() def main() -> int: args = parse_args() try: root = repository_root(args.repo) output = ( Path(args.output).expanduser().resolve() if args.output else root / REVIEW_DIR / "changes.json" ) if args.commit is None: hunks = collect_worktree(root) target = {"kind": "working-tree"} else: hunks, commit_oid, base_oid = collect_commit(root, args.commit) target = { "kind": "commit", "revision": args.commit, "commit": commit_oid, "base": base_oid, } atomic_write_json(output, build_document(root, hunks, target)) except (CollectionError, OSError) as exc: print(f"error: {exc}", file=sys.stderr) return 1 if args.commit is None: counts = { scope: sum(1 for hunk in hunks if hunk.scope == scope) for scope in ("staged", "unstaged", "untracked") } detail = ( f"{counts['staged']} staged, {counts['unstaged']} unstaged, " f"{counts['untracked']} untracked" ) else: detail = f"commit {commit_oid} against {base_oid}" print(f"Collected {len(hunks)} hunks ({detail}) -> {output}") return 0 if __name__ == "__main__": raise SystemExit(main())