From d480fe38d2bbd90d20bb8d8beee17d3680e8509c Mon Sep 17 00:00:00 2001 From: Krish Garg <181051779+krishhgg@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:32:10 +0000 Subject: [PATCH] fix(affected): accept qualified seeds, exit nonzero on a miss or tie `graphify affected` could not name one of two same-named methods, and a failed lookup printed "No unique node match" on stdout with exit 0. On httpx, `Client.send` and `httpx/_client.py::Client.send` missed, and `send` resolved to a sourceless `Send` stub left by an annotation. Seed resolution now runs in tiers (exact id, qualified, exact label, bare name, source path, substring) and resolves only when a tier has exactly one candidate. `Class.method`, `path::Class.method`, `path::function` and `path::Class` are accepted. The class must own the member through a `method` or `contains` edge, and the path restricts the node's own source_file, exact spelling first. Source-backed nodes outrank sourceless stubs. A miss or a tie prints to stderr and exits 1, and a tie lists its candidates. affected_nodes, explain and path are unchanged. Related: #3485, #3913, #1669, #2706, #2707. Co-Authored-By: Claude Opus 5.5 --- graphify/__main__.py | 1 + graphify/affected.py | 452 ++++++++++++++++++++++-- graphify/cli.py | 18 +- tests/test_affected_cli.py | 682 ++++++++++++++++++++++++++++++++++++- 4 files changed, 1113 insertions(+), 40 deletions(-) diff --git a/graphify/__main__.py b/graphify/__main__.py index 5f5750b885..cde43bcc59 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -794,6 +794,7 @@ def _run_cli() -> None: print(" --budget N cap output at N tokens (default 2000)") print(" --graph path to graph.json (default graphify-out/graph.json)") print(" affected \"X\" reverse traversal to find nodes impacted by X") + print(" X: label, node id, file path, Class.method, or path::symbol") print(" --relation R edge relation to traverse in reverse (repeatable)") print(" --depth N reverse traversal depth (default 2)") print(" --graph path to graph.json (default graphify-out/graph.json)") diff --git a/graphify/affected.py b/graphify/affected.py index 0184a8f88f..1f29a15b37 100644 --- a/graphify/affected.py +++ b/graphify/affected.py @@ -2,8 +2,9 @@ from collections import deque from dataclasses import dataclass +from itertools import chain from pathlib import Path -from typing import Iterable +from typing import Iterable, Iterator import unicodedata import networkx as nx @@ -44,11 +45,76 @@ class AffectedHit: via_location: "str | None" = None +class SeedResolutionError(LookupError): + """The `affected` query named no node, or several; the message says which.""" + + +# Cap on the candidates an ambiguity message lists; a short substring query +# can tie hundreds of nodes. +_MAX_LISTED_CANDIDATES = 20 + + def _node_label(graph: nx.Graph, node_id: str) -> str: data = graph.nodes[node_id] return str(data.get("label") or node_id) +def _owned_pairs(edges: Iterable[tuple]) -> Iterator[tuple[str, str]]: + """(owner, member) for each `method`/`contains` edge among `edges`. + + The owner is the edge's stored source: its `_src`/`_tgt` markers where + present, else arc order. An undirected graph (`build_from_json`'s default) + keeps direction only in the markers, so reading arc order there reversed + ownership whenever the member node was inserted before its owner. Markers + that do not name the edge's own endpoints are ignored, as in serve.py. + """ + for u, v, data in edges: + if str(data.get("relation", "")) not in ("method", "contains"): + continue + src, tgt = data.get("_src", u), data.get("_tgt", v) + if {src, tgt} != {u, v}: + src, tgt = u, v + if src != tgt: + yield str(src), str(tgt) + + +class _Ownership: + """Owner and member lookups over `method`/`contains` edges, for seed + resolution. (`affected_nodes` keeps its own member seeding unchanged.) + + A DiGraph is read per node, from that node's own edges. An undirected + graph's arc order is not its direction, and its fallback order comes from + a whole-graph edge pass, so it is indexed from one such pass on first use. + Scanning every edge per lookup made a qualified seed cost (matching + members x edges): 17 s for 4,000 edges. + """ + + def __init__(self, graph: nx.Graph) -> None: + self._graph = graph + self._index: dict[str, list[tuple[str, str]]] | None = None + + def _pairs(self, node_id: str) -> list[tuple[str, str]]: + graph = self._graph + if isinstance(graph, nx.DiGraph): + return list(_owned_pairs( + chain(graph.out_edges(node_id, data=True), graph.in_edges(node_id, data=True)) + )) + if self._index is None: + self._index = {} + for pair in _owned_pairs(graph.edges(data=True)): + for end in pair: + self._index.setdefault(end, []).append(pair) + return self._index.get(node_id, []) + + def members(self, node_id: str) -> list[str]: + """Nodes `node_id` owns through one `method`/`contains` edge.""" + return [member for owner, member in self._pairs(node_id) if owner == node_id] + + def owners(self, node_id: str) -> list[str]: + """Nodes that own `node_id` through one `method`/`contains` edge.""" + return [owner for owner, member in self._pairs(node_id) if member == node_id] + + def _format_location(data: dict) -> str: source_file = data.get("source_file") or "-" source_location = data.get("source_location") @@ -63,6 +129,20 @@ def _bare_name(label: str) -> str: return label[:-2] if label.endswith("()") else label +def _label_names(label: str) -> tuple[str, ...]: + """Bare names a node label answers to. + + A method label (".send()") also answers to its name without the leading + ".", so "send" names it as it names the function "send()". Only the + method shape (leading "." and trailing "()") loses the dot: a dotfile such + as ".config" keeps it, so "Config()" still means the class `Config`. + """ + bare = _bare_name(label) + if bare.startswith(".") and _normalize_label(label).endswith("()"): + return (bare, bare[1:]) + return (bare,) + + def _normalize_label(label: str) -> str: return unicodedata.normalize("NFC", label).casefold() @@ -135,56 +215,338 @@ def _prefer_file_node( return None -def resolve_seed(graph: nx.Graph, query: str, root: Path | None = None) -> str | None: - # A trailing path separator must not change a source-file match — serve's - # _find_node tokenizes the path (which drops it), so strip it here for parity - # (otherwise `affected "src/x.ts/"` returned None while `explain` resolved it). - query = query.rstrip("/\\") or query - if query in graph: - return query - query_lower = _normalize_label(query) - exact_label_matches = [ +def _label_matches(graph: nx.Graph, name: str, node_ids: Iterable[str]) -> list[str]: + """Nodes among `node_ids` labeled `name`: exact label first, else bare name. + + Source-backed nodes come first in both, so a sourceless stub that matches + exactly (`Handle`, left by a `handler: Handle` annotation) cannot hide the + real definition that matches by bare name (`handle()`). + """ + node_ids = list(node_ids) + name_lower = _normalize_label(name) + exact = [ + node_id + for node_id in node_ids + if _normalize_label(str(graph.nodes[node_id].get("label", ""))) == name_lower + ] + bare = _named_ids(graph, name, node_ids) + return _source_backed(graph, exact) or _source_backed(graph, bare) or exact or bare + + +def _named_ids(graph: nx.Graph, name: str, node_ids: Iterable[str]) -> list[str]: + """Nodes among `node_ids` whose label names `name`, exactly or by bare name.""" + name_bare = _bare_name(name) + return [ + node_id + for node_id in node_ids + if name_bare in _label_names(str(graph.nodes[node_id].get("label", ""))) + ] + + +def _bare_name_matches(graph: nx.Graph, query: str) -> list[str]: + query_bare = _bare_name(query) + return [ str(node_id) for node_id, data in graph.nodes(data=True) - if _normalize_label(str(data.get("label", ""))) == query_lower + if query_bare in _label_names(str(data.get("label", ""))) ] - if len(exact_label_matches) == 1: - return exact_label_matches[0] - # Callable labels are decorated ("name()"), so a bare "name" query falls - # through exact matching and then ties with any "name*" sibling in the - # contains pass. Match on the undecorated name before giving up. - query_bare = _bare_name(query_lower) - bare_name_matches = [ + + +def _source_backed(graph: nx.Graph, node_ids: list[str]) -> list[str]: + return [ + node_id + for node_id in node_ids + if str(graph.nodes[node_id].get("source_file") or "").strip() + ] + + +def _prefer_source_backed(graph: nx.Graph, hits: list[str], query: str) -> list[str]: + """`hits` without sourceless stubs when any source-backed node is among them. + + A tier that matched only stubs yields to the real definitions with the + query's bare name: the stub `Send` (id `send`, left by a `send: Send` + annotation) must not answer for the real `send` methods. + """ + if hits and not _source_backed(graph, hits): + hits = _source_backed(graph, _bare_name_matches(graph, query)) or hits + return _source_backed(graph, hits) or hits + + +def _is_path(graph: nx.Graph, text: str, root: Path | None) -> bool: + """Is the left half of a `::` query a path? + + Yes when it has a path separator or names a file in the graph, exactly or + after normalization, even when that matches several files (see + `_path_files`), spaces or not: `my pkg/client.py`. Otherwise only a file + suffix without whitespace counts, so a docstring label such as + `Registers a function. .. versionadded:: 0.11` stays prose. + """ + if "/" in text or "\\" in text or _path_files(graph, text, root): + return True + return not any(char.isspace() for char in text) and bool(Path(text).suffix) + + +def _path_files(graph: nx.Graph, path: str, root: Path | None) -> dict[str, list[str]]: + """The files `path` names (compared in repo-relative form), each with its + nodes. + + The exact spelling wins: on Linux `a.py` and `A.py` are two files, and so + are `caf\u00e9.py` and `cafe\u0301.py` (composed and decomposed). When any + node has the exact path, only that file is returned. Otherwise every file + that matches after case and Unicode normalization is returned. One such + file is what a wrong-case path on Windows or macOS means. Several are a + tie that the caller must report, not settle. + """ + query_path = _as_repo_relative(path, root) + query_key = _normalize_label(query_path) + by_file: dict[str, list[str]] = {} + for node_id, data in graph.nodes(data=True): + source_file = str(data.get("source_file") or "") + if source_file and _normalize_label(source_file) == query_key: + by_file.setdefault(source_file, []).append(str(node_id)) + if query_path in by_file: + return {query_path: by_file[query_path]} + return by_file + + +def _split_qualified(query: str) -> tuple[str, bool, str, str]: + """`query` as (path, has `::`, owner, member); owner is "" for a plain symbol. + + The path is kept exactly as typed: on Linux ` a.py` and `a.py` are two + files, so trimming it could name the other one. Only the symbol half is + trimmed, since labels do not start or end with spaces. + """ + path_part, sep, symbol = query.partition("::") + if not sep: + path_part, symbol = "", query + symbol = symbol.strip() + owner, _, member = symbol.rpartition(".") + return path_part, bool(sep), owner, member if owner else symbol + + +def _owns_members( + graph: nx.Graph, ownership: _Ownership, name: str, node_ids: list[str] +) -> bool: + return any(ownership.members(node_id) for node_id in _named_ids(graph, name, node_ids)) + + +def _nested_qualifier(graph: nx.Graph, query: str, ownership: _Ownership) -> str | None: + """`Inner.run` for an `Outer.Inner.run` query that walks two ownership + levels. Only one owner level is supported, so the caller refuses it + instead of guessing. An owner whose own label is dotted (`Outer.Inner`, + owning `.run()`) is one level and is not this case, but only when that + node really owns a member with the requested name: a heading + `# Outer.Inner` that owns nothing (or only other sections) does not lift + the refusal. None when the owner part is not dotted, a segment is not an + identifier (a Markdown heading such as `Changelog.0.28.1 (6th December, + 2024)`), the owner part as a whole owns the member, or its first segment + names nothing that owns members.""" + _, _, owner, member = _split_qualified(query) + if "." not in owner: + return None + segments = [*owner.split("."), member.removesuffix("()")] + if not all(segment.isidentifier() for segment in segments): + return None + node_ids = [str(node_id) for node_id in graph.nodes] + member_bare = _bare_name(member) + if any( + member_bare in _label_names(str(graph.nodes[member_id].get("label", ""))) + for owner_id in _named_ids(graph, owner, node_ids) + for member_id in ownership.members(owner_id) + ): + return None + if not _owns_members(graph, ownership, owner.split(".", 1)[0], node_ids): + return None + return f"{owner.rsplit('.', 1)[-1]}.{member}" + + +def _qualified_scope( + graph: nx.Graph, query: str, root: Path | None, ownership: _Ownership +) -> str | None: + """How strictly `query` is scoped by its own qualifier. + + "path" for a `::` whose left half is a path (see `_is_path`): only that + file can answer it. "chain" for an `Outer.Inner.run` chain (see + `_nested_qualifier`), which is refused. "owner" for a dotted query whose + owner part names a node that owns members (a dotted owner label such as + `Outer.Inner` included): the owner's members answer it first, and no + substring match can. None for everything else, such as a namespace (the + Ruby label `Billing::Invoice::Line`, queried as `Invoice::Line`) or a + file name such as `index.ts` next to an unrelated `index()` function. + """ + path_part, sep, owner, member = _split_qualified(query) + if sep: + if path_part and (owner or member) and _is_path(graph, path_part, root): + return "path" + return None + if not owner: + return None + if _nested_qualifier(graph, query, ownership): + return "chain" + node_ids = [str(node_id) for node_id in graph.nodes] + if _owns_members(graph, ownership, owner, node_ids): + return "owner" + return None + + +def _qualified_matches( + graph: nx.Graph, query: str, root: Path | None, ownership: _Ownership +) -> list[str]: + """Nodes named by `Class.member`, `path::symbol` or `path::Class.member`. + + `path` restricts the named node itself (see `_path_files`). `Class` + must own the member through a `method`/`contains` edge (the same + ownership `affected_nodes` seeds from, #1669), so two same-named methods + are told apart by their real owner and never by label text alone. Only + owners of a member that has the requested name, in the requested file, + are eligible: an unrelated `class Handle` elsewhere must not outrank the + `handle()` that owns `send()` in `pkg.py`. Inherited members are not + followed: name the defining class. + + When several owners carry the name, each keeps its own best-named member + and all of them are returned, so the caller reports a tie. The member's + label decoration never ranks one owner over another: a Markdown `# Client` + heading with a `## send` section must not outrank the class `Client` + whose method is labeled `.send()`, nor the other way round. Nothing in + graphify's seed lookups (`explain`, `path`, `query`, `affected`) prefers + code nodes over document nodes, so neither does this. + + A path that matches several files only after normalization (`my client.py` + and `My Client.py` for `MY CLIENT.PY`) is a tie between files. The hits + are returned when they span at least two of those files, so they are + listed as a tie. Otherwise the result is a miss: the symbol may exist in + one file only, but which file was meant is still not known. + """ + path_part, sep, owner, member = _split_qualified(query) + if sep and not (path_part and (owner or member)): + return [] + if not sep and not owner: + return [] + files = _path_files(graph, path_part, root) if sep else {} + candidates = ( + [node_id for nodes in files.values() for node_id in nodes] if sep + else [str(node_id) for node_id in graph.nodes] + ) + named = _named_ids(graph, member, candidates) + if owner: + # Group the named members by owner once: one owner lookup per member, + # however many owners share the name. + by_owner: dict[str, list[str]] = {} + for node_id in named: + for owner_id in ownership.owners(node_id): + by_owner.setdefault(owner_id, []).append(node_id) + picked: set[str] = set() + for owner_id in _label_matches(graph, owner, sorted(by_owner)): + picked.update(_label_matches(graph, member, by_owner[owner_id])) + hits = [node_id for node_id in named if node_id in picked] + else: + hits = _label_matches(graph, member, named) + if len(files) > 1: + spanned = {str(graph.nodes[node_id].get("source_file") or "") for node_id in hits} + if len(spanned) < 2: + return [] + return hits + + +def _name_tiers(graph: nx.Graph, query: str, root: Path | None) -> Iterator[list[str]]: + """Candidate lists for `query` by exact label, bare name and source path, + most specific first (computed lazily).""" + query_lower = _normalize_label(query) + yield [ str(node_id) for node_id, data in graph.nodes(data=True) - if _bare_name(str(data.get("label", ""))) == query_bare + if _normalize_label(str(data.get("label", ""))) == query_lower ] - if len(bare_name_matches) == 1: - return bare_name_matches[0] - # Compare paths in repo-relative form. Only this branch is path-shaped; the - # label branches above keep the query verbatim. + # Callable labels are decorated ("name()", ".method()"), so a bare "name" + # query falls through exact matching and then ties with any "name*" sibling + # in the contains pass. Match on the undecorated name before giving up. + yield _bare_name_matches(graph, query) + # Compare paths in repo-relative form. Only this tier is path-shaped; the + # label tiers above keep the query verbatim. query_path = _normalize_label(_as_repo_relative(query, root)) exact_source_matches = [ str(node_id) for node_id, data in graph.nodes(data=True) if _normalize_label(str(data.get("source_file", ""))) in (query_lower, query_path) ] - if len(exact_source_matches) == 1: - return exact_source_matches[0] - if exact_source_matches: + if len(exact_source_matches) > 1: preferred_file_node = _prefer_file_node( graph, exact_source_matches, _as_repo_relative(query, root) ) if preferred_file_node is not None: - return preferred_file_node + exact_source_matches = [preferred_file_node] + yield exact_source_matches + + +def resolve_seed_candidates( + graph: nx.Graph, query: str, root: Path | None = None +) -> list[str]: + """Nodes `query` names: one id when it resolves, several when it is + ambiguous, none when nothing matches. Never picks among ties. + + Tiers run most specific first, and the first with exactly one candidate + wins. In every tier a source-backed definition outranks a sourceless + stub. An exact node id comes first. A qualified query (see + `_qualified_scope`) is then answered from the scope it names, before any + unrestricted label match, so a Markdown heading `# Client.send` cannot + stand in for the method. A `path::` query and a refused `Outer.Inner.run` + chain stop there: a heading `# missing.py::Client.send` or + `# Outer.Inner.run` must not answer them. A `Class.member` query whose + class has no such member may still be answered by the label, bare-name + and source-path tiers (a file named `CHANGELOG.md` next to a `Changelog` + heading), but never by a substring. Only an unqualified query that no + tier matched at all reaches the substring tier: a substring is weaker + evidence than names that tied. + """ + # A trailing path separator must not change a source-file match — serve's + # _find_node tokenizes the path (which drops it), so strip it here for parity + # (otherwise `affected "src/x.ts/"` returned None while `explain` resolved it). + query = query.rstrip("/\\") or query + ownership = _Ownership(graph) + ambiguous: list[str] = [] + if query in graph: + hits = _prefer_source_backed(graph, [query], query) + if len(hits) == 1: + return hits + ambiguous = hits + scope = _qualified_scope(graph, query, root, ownership) + if scope: + hits = _prefer_source_backed(graph, _qualified_matches(graph, query, root, ownership), query) + if hits or scope != "owner": + return hits if len(hits) == 1 else ambiguous or hits + for hits in _name_tiers(graph, query, root): + hits = _prefer_source_backed(graph, hits, query) + if len(hits) == 1: + return hits + ambiguous = ambiguous or hits + if ambiguous or scope: + return ambiguous + query_lower = _normalize_label(query) contains_matches = [ str(node_id) for node_id, data in graph.nodes(data=True) if query_lower in _normalize_label(str(data.get("label", ""))) ] - if len(contains_matches) == 1: - return contains_matches[0] - return None + return _source_backed(graph, contains_matches) or contains_matches + + +def resolve_seed(graph: nx.Graph, query: str, root: Path | None = None) -> str | None: + """The one node `query` names, or None when it names none or several.""" + candidates = resolve_seed_candidates(graph, query, root) + return candidates[0] if len(candidates) == 1 else None + + +def _ambiguous_seed_message(graph: nx.Graph, query: str, candidates: list[str]) -> str: + lines = [f"Ambiguous seed {query}: {len(candidates)} nodes match."] + for node_id in candidates[:_MAX_LISTED_CANDIDATES]: + lines.append( + f"- {_node_label(graph, node_id)} {_format_location(graph.nodes[node_id])}" + f" (id: {node_id})" + ) + if len(candidates) > _MAX_LISTED_CANDIDATES: + lines.append(f"... and {len(candidates) - _MAX_LISTED_CANDIDATES} more") + lines.append("Pass one of the ids above, or narrow the seed as Class.method or path::symbol.") + return "\n".join(lines) def affected_nodes( @@ -263,10 +625,34 @@ def format_affected( depth: int = 2, root: Path | None = None, ) -> str: + """Report what a change to `query` affects. + + Raises SeedResolutionError when `query` names no node or several: printing + that as a normal report let a typo or a tie read as "nothing depends on + this". + """ relation_list = tuple(relations) - seed = resolve_seed(graph, query, root) - if seed is None: - return f"No unique node match for {query}" + candidates = resolve_seed_candidates(graph, query, root) + if not candidates: + message = f"No unique node match for {query}" + nested = _nested_qualifier(graph, query, _Ownership(graph)) + if nested: + message += ( + f"\nOnly one owner level is supported (Class.method). " + f"Try {nested}, path::{nested}, or a node id." + ) + path_part, sep, _, _ = _split_qualified(query) + files = sorted(_path_files(graph, path_part, root)) if sep and path_part else [] + if len(files) > 1: + message += ( + f"\nThe path matches {len(files)} files: " + + ", ".join(repr(name) for name in files) + + ". Spell one exactly." + ) + raise SeedResolutionError(message) + if len(candidates) > 1: + raise SeedResolutionError(_ambiguous_seed_message(graph, query, candidates)) + seed = candidates[0] hits = affected_nodes(graph, seed, relations=relation_list, depth=depth) lines = [ diff --git a/graphify/cli.py b/graphify/cli.py index 3b516cea60..df7487ab01 100644 --- a/graphify/cli.py +++ b/graphify/cli.py @@ -1398,7 +1398,12 @@ def dispatch_command(cmd: str) -> None: if len(sys.argv) < 3: print("Usage: graphify affected \"\" [--relation R] [--depth N] [--graph path]", file=sys.stderr) sys.exit(1) - from graphify.affected import DEFAULT_AFFECTED_RELATIONS, format_affected, load_graph + from graphify.affected import ( + DEFAULT_AFFECTED_RELATIONS, + SeedResolutionError, + format_affected, + load_graph, + ) query = sys.argv[2] graph_path = _default_graph_path() depth = 2 @@ -1453,15 +1458,20 @@ def dispatch_command(cmd: str) -> None: # --graph falls back to its own directory. from graphify.paths import GRAPHIFY_OUT_NAME graph_root = gp.parent.parent if gp.parent.name == GRAPHIFY_OUT_NAME else gp.parent - print( - format_affected( + try: + report = format_affected( graph, query, relations=relations or DEFAULT_AFFECTED_RELATIONS, depth=depth, root=graph_root, ) - ) + except SeedResolutionError as exc: + # A miss or a tie exits nonzero on stderr: on stdout with exit 0 a + # script read it as "nothing depends on this". + print(exc, file=sys.stderr) + sys.exit(1) + print(report) elif cmd in ("god-nodes", "god_nodes"): # god_nodes has long been an analyzer (analyze.py), an MCP tool, and a # README-advertised capability, but never a CLI subcommand — `graphify diff --git a/tests/test_affected_cli.py b/tests/test_affected_cli.py index 7c452907da..2f7699c076 100644 --- a/tests/test_affected_cli.py +++ b/tests/test_affected_cli.py @@ -392,10 +392,15 @@ def test_affected_absolute_seed_outside_root_misses_cleanly(tmp_path, monkeypatc monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) monkeypatch.setattr(mainmod.sys, "argv", ["graphify", "affected", outside_seed, "--graph", str(gp)]) - mainmod.main() + # A miss exits nonzero on stderr, so a script cannot read it as "no dependents". + import pytest + with pytest.raises(SystemExit) as exc: + mainmod.main() - out = capsys.readouterr().out - assert "Affected nodes for Foo" not in out # must NOT resolve to the in-root Foo + captured = capsys.readouterr() + assert exc.value.code == 1 + assert "Affected nodes for Foo" not in captured.out # must NOT resolve to the in-root Foo + assert f"No unique node match for {outside_seed}" in captured.err def test_affected_absolute_seed_with_graph_not_under_out_dir(tmp_path, monkeypatch, capsys): @@ -423,3 +428,674 @@ def test_affected_absolute_seed_with_graph_not_under_out_dir(tmp_path, monkeypat assert "Affected nodes for Foo" in out assert "X()" in out + +# ── Qualified seeds, stub preference, and nonzero exit on a miss/ambiguity ── + +def _two_send_graph(): + """httpx-shaped graph: `Client.send` and `AsyncClient.send` share the label + `.send()` in one file, a nested `send()` lives in another, and a parameter + annotation left a sourceless `Send` stub whose id is the bare name `send`.""" + g = nx.DiGraph() + g.add_node("client_py", label="_client.py", source_file="pkg/_client.py", source_location="L1") + g.add_node("client", label="Client", source_file="pkg/_client.py", source_location="L10") + g.add_node("client_send", label=".send()", source_file="pkg/_client.py", source_location="L20") + g.add_node("async_client", label="AsyncClient", source_file="pkg/_client.py", source_location="L50") + g.add_node("async_send", label=".send()", source_file="pkg/_client.py", source_location="L60") + g.add_node("handle", label="handle()", source_file="pkg/asgi.py", source_location="L5") + g.add_node("handle_send", label="send()", source_file="pkg/asgi.py", source_location="L8") + g.add_node("helper", label="helper()", source_file="pkg/util.py", source_location="L3") + g.add_node("helper_other", label="helper()", source_file="pkg/other.py", source_location="L3") + g.add_node("send", label="Send", source_file="", source_location="") + g.add_node("caller", label="caller()", source_file="app.py", source_location="L4") + g.add_node("async_caller", label="async_caller()", source_file="app.py", source_location="L9") + g.add_edge("client_py", "client", relation="contains") + g.add_edge("client_py", "async_client", relation="contains") + g.add_edge("client", "client_send", relation="method") + g.add_edge("async_client", "async_send", relation="method") + g.add_edge("handle", "handle_send", relation="contains") + g.add_edge("caller", "client_send", relation="calls", source_file="app.py", source_location="L5") + g.add_edge("async_caller", "async_send", relation="calls", source_file="app.py", source_location="L10") + g.add_edge("caller", "send", relation="references") + return g + + +def test_resolve_seed_class_dot_method_uses_owning_class(): + from graphify.affected import resolve_seed + + g = _two_send_graph() + assert resolve_seed(g, "Client.send") == "client_send" + assert resolve_seed(g, "AsyncClient.send") == "async_send" + assert resolve_seed(g, "Client.send()") == "client_send" + # A nested function is reachable through its `contains` owner too. + assert resolve_seed(g, "handle.send") == "handle_send" + # The owner must really own the member: no label-text guess. + assert resolve_seed(g, "Client.helper") is None + + +def test_resolve_seed_path_scoped_forms(tmp_path): + from graphify.affected import resolve_seed + + g = _two_send_graph() + assert resolve_seed(g, "pkg/_client.py::Client.send") == "client_send" + assert resolve_seed(g, "pkg/_client.py::AsyncClient.send") == "async_send" + assert resolve_seed(g, "./pkg/_client.py::Client.send") == "client_send" + assert resolve_seed(g, str(tmp_path / "pkg" / "_client.py") + "::Client.send", tmp_path) == "client_send" + assert resolve_seed(g, "pkg/_client.py::Client") == "client" + assert resolve_seed(g, "pkg/util.py::helper") == "helper" + assert resolve_seed(g, "pkg/other.py::helper()") == "helper_other" + # Both `.send()` methods live in this file: the path alone cannot pick one. + assert resolve_seed(g, "pkg/_client.py::send") is None + # The path must be the member's own file, not a guess from the label. + assert resolve_seed(g, "pkg/asgi.py::Client.send") is None + + +def test_resolve_seed_bare_name_does_not_pick_sourceless_stub(): + from graphify.affected import resolve_seed + + g = _two_send_graph() + # `send` is the stub's id and `Send` its label, but three real definitions + # carry that name: ambiguous, not the stub. + assert resolve_seed(g, "send") is None + assert resolve_seed(g, "Send") is None + + g.remove_nodes_from(["client_send", "async_send"]) + # One real definition left: it wins over the stub. + assert resolve_seed(g, "send") == "handle_send" + assert resolve_seed(g, "Send") == "handle_send" + + g.remove_node("handle_send") + # No real namesake: the stub is the only answer and stays reachable. + assert resolve_seed(g, "send") == "send" + + +def test_resolve_seed_substring_does_not_break_a_name_tie(): + """Two `Config` classes tie on the name; the substring pass must not then + hand back the one `load_config()` that happens to contain `config()`.""" + from graphify.affected import resolve_seed + + g = nx.DiGraph() + g.add_node("a", label="Config", source_file="pkg/a.py", source_location="L1") + g.add_node("b", label="Config", source_file="pkg/b.py", source_location="L1") + g.add_node("c", label="load_config()", source_file="pkg/c.py", source_location="L1") + + assert resolve_seed(g, "Config()") is None + + +def test_resolve_seed_qualified_miss_does_not_fall_back_to_substring(): + """A qualified query that matches nothing is a miss, even when some label + (here a docs heading) happens to contain the whole query text.""" + from graphify.affected import resolve_seed + + g = _two_send_graph() + g.add_node("doc", label="Using pkg/missing.py::Client.send safely", source_file="guide.md") + g.add_node("doc2", label="Why Client.nope was removed", source_file="guide.md") + assert resolve_seed(g, "pkg/missing.py::Client.send") is None + assert resolve_seed(g, "Client.nope") is None + + # A left half that names a file in the graph is a path even without an + # extension (`Makefile::all`), so a root file named `Invoice` scopes + # `Invoice::Line` to itself and the miss is final, even though a Ruby + # label `Billing::Invoice::Line` contains the query. + g.add_node("line", label="Billing::Invoice::Line", source_file="billing.rb", source_location="L3") + g.add_node("invoice_file", label="Invoice", source_file="Invoice", source_location="L1") + assert resolve_seed(g, "Invoice::Line") is None + + +def test_resolve_seed_unqualified_dotted_and_namespace_queries_keep_substring_match(): + """Boundary of the rule above, unchanged from before: a `::` whose left half + is a namespace rather than a path (Ruby's `Billing::Invoice::Line`), and a + dotted name whose owner part names nothing that owns members, still reach + the substring tier, even next to an unrelated `index()` function.""" + from graphify.affected import resolve_seed + + g = nx.DiGraph() + g.add_node("line", label="Billing::Invoice::Line", source_file="billing.rb", source_location="L3") + g.add_node("index", label="web/index.ts", source_file="web/index.ts", source_location="L1") + assert resolve_seed(g, "Invoice::Line") == "line" + assert resolve_seed(g, "index.ts") == "index" + + g.add_node("index_fn", label="index()", source_file="views.py", source_location="L4") + assert resolve_seed(g, "index.ts") == "index" + + # A file named after a heading that owns sections is still found by its + # exact label, and a docstring label containing `::` is prose, not a path. + g.add_node("changelog_file", label="CHANGELOG.md", source_file="CHANGELOG.md", source_location="L1") + g.add_node("changelog", label="Changelog", source_file="CHANGELOG.md", source_location="L1") + g.add_node("unreleased", label="Unreleased", source_file="CHANGELOG.md", source_location="L3") + g.add_edge("changelog", "unreleased", relation="contains") + prose = "Registers a function. .. versionadded:: 0.11" + g.add_node("rationale", label=prose, source_file="app.py", source_location="L9") + assert resolve_seed(g, "CHANGELOG.md") == "changelog_file" + assert resolve_seed(g, prose) == "rationale" + + +def test_resolve_seed_reads_ownership_from_direction_markers_on_undirected_graph(): + """`build_from_json` returns an undirected graph by default, which keeps + edge direction only in `_src`/`_tgt`. Ownership must follow those markers + whichever endpoint was inserted first, never the reverse.""" + from graphify.affected import resolve_seed + from graphify.build import build_from_json + + outer = {"id": "outer", "label": "Outer", "source_file": "pkg.py", "source_location": "L1"} + inner = {"id": "inner", "label": "Inner", "source_file": "pkg.py", "source_location": "L2"} + edge = {"source": "outer", "target": "inner", "relation": "contains", + "confidence": "EXTRACTED", "source_file": "pkg.py", "source_location": "L2"} + for nodes in ([outer, inner], [inner, outer]): + g = build_from_json({"nodes": nodes, "edges": [edge]}) + assert not g.is_directed() + assert resolve_seed(g, "Outer.Inner") == "inner", [n["id"] for n in nodes] + assert resolve_seed(g, "Inner.Outer") is None, [n["id"] for n in nodes] + + +def test_resolve_seed_owner_prefers_real_definition_over_stub(tmp_path): + """Extracted fixture: `handler: Handle` leaves a sourceless stub labeled + `Handle`, which matches the owner `handle` exactly. The real `handle()` + (matched by bare name) must still be the owner of its nested `send()`.""" + import pytest + + from graphify.affected import resolve_seed + from graphify.build import build_from_json + from graphify.extract import extract + + (tmp_path / "pkg.py").write_text( + "def handle():\n def send():\n pass\n send()\n", encoding="utf-8" + ) + (tmp_path / "server.py").write_text( + "def unrelated(handler: Handle):\n pass\n", encoding="utf-8" + ) + extraction = extract( + [tmp_path / "pkg.py", tmp_path / "server.py"], root=tmp_path, cache_root=tmp_path + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + stubs = [n for n, d in g.nodes(data=True) if d.get("label") == "Handle"] + if not stubs or g.nodes[stubs[0]].get("source_file"): + pytest.fail(f"fixture no longer yields a sourceless Handle stub: {stubs}") + nested = [ + n for n, d in g.nodes(data=True) + if d.get("label") == "send()" and d.get("source_file") == "pkg.py" + ] + assert len(nested) == 1, nested + assert resolve_seed(g, "handle.send") == nested[0], directed + assert resolve_seed(g, "pkg.py::handle.send") == nested[0], directed + + # A source-backed `class Handle` elsewhere matches `handle` exactly too. + # It owns no `send`, so it is not an eligible owner for this query. + (tmp_path / "other.py").write_text( + "class Handle:\n def unrelated(self):\n pass\n", encoding="utf-8" + ) + extraction = extract( + [tmp_path / "pkg.py", tmp_path / "other.py"], root=tmp_path, cache_root=tmp_path + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + classes = [ + n for n, d in g.nodes(data=True) + if d.get("label") == "Handle" and d.get("source_file") == "other.py" + ] + assert classes, "fixture no longer yields the source-backed class Handle" + nested = [ + n for n, d in g.nodes(data=True) + if d.get("label") == "send()" and d.get("source_file") == "pkg.py" + ] + assert len(nested) == 1, nested + assert resolve_seed(g, "pkg.py::handle.send") == nested[0], directed + assert resolve_seed(g, "handle.send") == nested[0], directed + assert resolve_seed(g, "pkg.py::Handle.unrelated") is None, directed + assert resolve_seed(g, "Handle.unrelated") is not None, directed + + +def test_resolve_seed_path_scope_keeps_case_distinct_files_apart(): + """On a case-sensitive filesystem `a.py` and `A.py` are two files, so a path + only scopes to its exact-case file. A wrong-case path still resolves when + a single file matches it case-insensitively (Windows, macOS).""" + from graphify.affected import resolve_seed + + g = nx.DiGraph() + g.add_node("lower_foo", label="Foo", source_file="a.py", source_location="L1") + g.add_node("upper_bar", label="Bar", source_file="A.py", source_location="L1") + assert resolve_seed(g, "a.py::Bar") is None + assert resolve_seed(g, "A.py::Foo") is None + assert resolve_seed(g, "a.py::Foo") == "lower_foo" + assert resolve_seed(g, "A.py::Bar") == "upper_bar" + # Neither file is spelled `A.PY`, and two match it case-insensitively. + assert resolve_seed(g, "A.PY::Bar") is None + + g.remove_node("lower_foo") + assert resolve_seed(g, "a.py::Bar") == "upper_bar" + + # Composed and decomposed spellings are two files on Linux too. + composed, decomposed = "caf\u00e9.py", "cafe\u0301.py" + g = nx.DiGraph() + g.add_node("composed_foo", label="Foo", source_file=composed, source_location="L1") + g.add_node("decomposed_bar", label="Bar", source_file=decomposed, source_location="L1") + assert resolve_seed(g, f"{composed}::Bar") is None + assert resolve_seed(g, f"{decomposed}::Foo") is None + assert resolve_seed(g, f"{composed}::Foo") == "composed_foo" + assert resolve_seed(g, f"{decomposed}::Bar") == "decomposed_bar" + g.remove_node("composed_foo") + assert resolve_seed(g, f"{composed}::Bar") == "decomposed_bar" + + # The path is not trimmed: ` a.py` and `a.py` are two files too. + g = nx.DiGraph() + g.add_node("spaced_run", label="run()", source_file=" a.py", source_location="L1") + g.add_node("plain_run", label="run()", source_file="a.py", source_location="L1") + assert resolve_seed(g, " a.py::run") == "spaced_run" + assert resolve_seed(g, "a.py::run") == "plain_run" + assert resolve_seed(g, "./ a.py::run") == "spaced_run" + + +def test_resolve_seed_path_matching_several_files_is_a_reported_tie(): + """A path that matches several files only after case folding keeps its + path scope (spaces or no extension notwithstanding): a heading whose text + is the query must not answer it. The tie is listed when the symbol is in + both files, and a miss names the files when it is in only one.""" + import pytest + + from graphify.affected import resolve_seed + + for lower, upper, owner, member, query in ( + ("my client.py", "My Client.py", "Client", ".send()", "MY CLIENT.PY::Client.send"), + ("makefile", "Makefile", "", "all", "MAKEFILE::all"), + ): + g = nx.DiGraph() + for key, source in (("lower", lower), ("upper", upper)): + g.add_node(f"{key}_file", label=source, source_file=source, source_location="L1") + if owner: + g.add_node(f"{key}_owner", label=owner, source_file=source, source_location="L1") + g.add_edge(f"{key}_owner", f"{key}_member", relation="method") + g.add_node(f"{key}_member", label=member, source_file=source, source_location="L2") + g.add_node("heading", label=query, source_file="guide.md", source_location="L1") + + assert resolve_seed(g, query) is None, query + assert resolve_seed(g, f"{lower}::{query.split('::')[1]}") == "lower_member", query + assert resolve_seed(g, f"{upper}::{query.split('::')[1]}") == "upper_member", query + from graphify.affected import SeedResolutionError, format_affected, resolve_seed_candidates + + assert resolve_seed_candidates(g, query) == ["lower_member", "upper_member"], query + with pytest.raises(SeedResolutionError) as exc: + format_affected(g, query) + assert "2 nodes match" in str(exc.value) and "heading" not in str(exc.value), query + + # Symbol in one file only: still no guess at which file was meant. + g.remove_node("upper_member") + assert resolve_seed(g, query) is None, query + with pytest.raises(SeedResolutionError) as exc: + format_affected(g, query) + message = str(exc.value) + assert f"No unique node match for {query}" in message, message + assert f"The path matches 2 files: {upper!r}, {lower!r}" in message, message + + +def test_affected_refuses_nested_owner_chain_instead_of_guessing(): + """`Outer.Inner.run` names an owner of an owner. Only one level is + supported, so it must miss with a hint, not fall through to the docs node + whose text contains it.""" + import pytest + + from graphify.affected import resolve_seed + + g = nx.DiGraph() + g.add_node("outer", label="Outer", source_file="pkg.py", source_location="L1") + g.add_node("inner", label="Inner", source_file="pkg.py", source_location="L2") + g.add_node("run", label=".run()", source_file="pkg.py", source_location="L3") + g.add_edge("outer", "inner", relation="contains") + g.add_edge("inner", "run", relation="method") + g.add_node("doc", label="Using Outer.Inner.run safely", source_file="guide.md") + + assert resolve_seed(g, "Outer.Inner.run") is None + from graphify.affected import SeedResolutionError, format_affected + + with pytest.raises(SeedResolutionError) as exc: + format_affected(g, "Outer.Inner.run") + message = str(exc.value) + assert "No unique node match for Outer.Inner.run" in message + assert "Only one owner level is supported" in message + assert "Inner.run" in message.splitlines()[1] + assert resolve_seed(g, "Inner.run") == "run" + assert resolve_seed(g, "pkg.py::Inner.run") == "run" + + # An owner whose own label is dotted is one level, not a chain. + g = nx.DiGraph() + g.add_node("owner", label="Outer.Inner", source_file="pkg.py", source_location="L1") + g.add_node("member", label=".run()", source_file="pkg.py", source_location="L2") + g.add_edge("owner", "member", relation="method") + assert resolve_seed(g, "Outer.Inner.run") == "member" + + +def test_affected_refuses_nested_owner_chain_over_exact_and_bare_headings(tmp_path): + """Extracted fixture: a heading whose text is exactly the chain, or the + chain with `()`, must not answer it. The refusal comes before the label + tiers. A heading `# Outer.Inner` does not lift it either, alone or owning + an unrelated section: only an `Outer.Inner` node that owns `run` would.""" + import pytest + + from graphify.affected import resolve_seed + from graphify.build import build_from_json + from graphify.extract import extract + + (tmp_path / "nested.py").write_text( + "class Outer:\n class Inner:\n def run(self):\n pass\n", + encoding="utf-8", + ) + guides = ( + ("Outer.Inner.run", "# Outer.Inner.run\n"), + ("Outer.Inner.run()", "# Outer.Inner.run()\n"), + ("Outer.Inner.run", "# Outer.Inner\n\n# Outer.Inner.run\n"), + ("Outer.Inner.run", "# Outer.Inner\n\n## Details\n\n# Outer.Inner.run\n"), + ) + for heading, guide in guides: + (tmp_path / "guide.md").write_text(guide, encoding="utf-8") + extraction = extract( + [tmp_path / "nested.py", tmp_path / "guide.md"], root=tmp_path, cache_root=tmp_path + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + labels = {d.get("label") for _, d in g.nodes(data=True)} + assert heading in labels, labels + assert resolve_seed(g, "Outer.Inner.run") is None, (guide, directed) + from graphify.affected import SeedResolutionError, format_affected + + with pytest.raises(SeedResolutionError) as exc: + format_affected(g, "Outer.Inner.run") + assert "Only one owner level is supported" in str(exc.value), (guide, directed) + run = [n for n, d in g.nodes(data=True) if d.get("label") == ".run()"] + assert resolve_seed(g, "Inner.run") == run[0], (guide, directed) + + +def test_resolve_seed_path_with_spaces_keeps_its_scope(tmp_path): + """Extracted fixture: paths with spaces, relative and absolute, existing + and missing, each next to a heading whose text is exactly the query. A + `::` left half with a separator, or one that names a file in the graph, + is a path even with spaces in it.""" + from graphify.affected import resolve_seed + from graphify.build import build_from_json + from graphify.extract import extract + + (tmp_path / "my pkg").mkdir() + (tmp_path / "my pkg" / "client.py").write_text( + "class Client:\n def send(self):\n pass\n", encoding="utf-8" + ) + (tmp_path / "my client.py").write_text( + "class Other:\n def ping(self):\n pass\n", encoding="utf-8" + ) + absolute = (tmp_path / "my pkg").as_posix() + queries = [ + "my pkg/client.py::Client.send", + "my pkg/missing.py::Client.send", + "my client.py::Other.ping", + f"{absolute}/client.py::Client.send", + f"{absolute}/missing.py::Client.send", + ] + (tmp_path / "guide.md").write_text( + "".join(f"# {query}\n\n" for query in queries), encoding="utf-8" + ) + extraction = extract( + [tmp_path / "my pkg" / "client.py", tmp_path / "my client.py", tmp_path / "guide.md"], + root=tmp_path, cache_root=tmp_path, + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + headings = {d.get("label") for _, d in g.nodes(data=True) if d.get("source_file") == "guide.md"} + assert set(queries) <= headings, headings + send = [n for n, d in g.nodes(data=True) if d.get("label") == ".send()"] + ping = [n for n, d in g.nodes(data=True) if d.get("label") == ".ping()"] + assert resolve_seed(g, queries[0], tmp_path) == send[0], directed + assert resolve_seed(g, queries[1], tmp_path) is None, directed + assert resolve_seed(g, queries[2], tmp_path) == ping[0], directed + assert resolve_seed(g, queries[3], tmp_path) == send[0], directed + assert resolve_seed(g, queries[4], tmp_path) is None, directed + + +def test_affected_nodes_member_seeding_unchanged_from_v8(): + """Qualified resolution reads ownership direction from `_src`/`_tgt`, but + `affected_nodes` keeps v8's own member seeding. On an undirected graph + v8 reads arc order, and so does its reverse walk; reading direction in + only one of the two reported `callee` as affected by `Svc`, which it + calls, not the other way round.""" + from graphify.affected import affected_nodes + from graphify.build import build_from_json + + def edge(source, target, relation, loc): + return {"source": source, "target": target, "relation": relation, + "confidence": "EXTRACTED", "source_file": "pkg.py", "source_location": loc} + + nodes = [ + {"id": "callee", "label": "callee()", "source_file": "pkg.py", "source_location": "L1"}, + {"id": "method", "label": ".run()", "source_file": "pkg.py", "source_location": "L5"}, + {"id": "owner", "label": "Svc", "source_file": "pkg.py", "source_location": "L4"}, + ] + edges = [edge("owner", "method", "method", "L5"), edge("method", "callee", "calls", "L6")] + for directed in (False, True): + g = build_from_json({"nodes": nodes, "edges": edges}, directed=directed) + assert [h.node_id for h in affected_nodes(g, "owner")] == [], directed + + # Directed: a caller of the method is still reached through member seeding. + nodes.append({"id": "caller", "label": "caller()", "source_file": "app.py", "source_location": "L2"}) + edges.append(edge("caller", "method", "calls", "L3")) + g = build_from_json({"nodes": nodes, "edges": edges}, directed=True) + assert [h.node_id for h in affected_nodes(g, "owner")] == ["caller"] + + +def test_resolve_seed_qualified_scope_beats_exact_heading_labels(tmp_path): + """Extracted fixture: Markdown headings whose text is exactly the query. + A qualified query is answered from the scope it names first, so the + `# Client.send` heading cannot stand in for the method, and a heading + cannot answer a `path::` query for a file that does not exist.""" + import pytest + + from graphify.affected import resolve_seed + from graphify.build import build_from_json + from graphify.extract import extract + + (tmp_path / "pkg.py").write_text( + "class Client:\n def send(self):\n pass\n", encoding="utf-8" + ) + (tmp_path / "guide.md").write_text( + "# Client.send\n\n# pkg.py::Client.send\n\n# missing.py::Client.send\n", + encoding="utf-8", + ) + extraction = extract( + [tmp_path / "pkg.py", tmp_path / "guide.md"], root=tmp_path, cache_root=tmp_path + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + headings = { + d.get("label") for _, d in g.nodes(data=True) if d.get("source_file") == "guide.md" + } + assert {"Client.send", "pkg.py::Client.send", "missing.py::Client.send"} <= headings + method = [ + n for n, d in g.nodes(data=True) + if d.get("label") == ".send()" and d.get("source_file") == "pkg.py" + ] + assert len(method) == 1, method + assert resolve_seed(g, "Client.send") == method[0], directed + assert resolve_seed(g, "pkg.py::Client.send") == method[0], directed + assert resolve_seed(g, "missing.py::Client.send") is None, directed + from graphify.affected import SeedResolutionError, format_affected + + with pytest.raises(SeedResolutionError): + format_affected(g, "missing.py::Client.send") + + +def test_affected_ties_a_heading_tree_with_the_method_it_names(tmp_path): + """Extracted fixture: `# Client` with a `## send` section is a second + `Client` owning a `send` member. Neither owner outranks the other by + member-label decoration (`send` vs `.send()`), and graphify's seed lookups + have no code-over-document rule, so `Client.send` is a tie listed on + stderr. Before, the heading won and hid the method's real caller. The + path form picks the method, and its caller is reported.""" + import pytest + + from graphify.affected import resolve_seed + from graphify.build import build_from_json + from graphify.extract import extract + + (tmp_path / "pkg.py").write_text( + "class Client:\n def send(self):\n pass\n" + " def request(self):\n self.send()\n", + encoding="utf-8", + ) + (tmp_path / "guide.md").write_text("# Client\n\n## send\n", encoding="utf-8") + extraction = extract( + [tmp_path / "pkg.py", tmp_path / "guide.md"], root=tmp_path, cache_root=tmp_path + ) + for directed in (True, False): + g = build_from_json(extraction, directed=directed) + method = [n for n, d in g.nodes(data=True) if d.get("label") == ".send()"] + heading = [ + n for n, d in g.nodes(data=True) + if d.get("label") == "send" and d.get("source_file") == "guide.md" + ] + assert len(method) == 1 and len(heading) == 1, (method, heading) + assert resolve_seed(g, "pkg.py::Client.send") == method[0], directed + assert resolve_seed(g, "guide.md::Client.send") == heading[0], directed + assert resolve_seed(g, "Client.send") is None, directed + from graphify.affected import resolve_seed_candidates + + assert set(resolve_seed_candidates(g, "Client.send", tmp_path)) == {method[0], heading[0]} + + g = build_from_json(extraction, directed=True) + from graphify.affected import SeedResolutionError, format_affected + + with pytest.raises(SeedResolutionError) as exc: + format_affected(g, "Client.send") + assert "pkg.py:L2" in str(exc.value) and "guide.md:L3" in str(exc.value) + report = format_affected(g, "pkg.py::Client.send") + assert "- .request() [calls] pkg.py:L5" in report + + +def test_qualified_resolution_looks_up_owners_once_per_member(monkeypatch): + """With N classes all labeled `Client`, each owning `.send()`, owner + lookups grow with N, not N squared: matching members are grouped by owner + once. Checking every owner against every member took 19 s at 2,000.""" + from graphify import affected + + calls = [] + original_owners, original_members = affected._Ownership.owners, affected._Ownership.members + + def owners(self, node_id): + calls.append(node_id) + return original_owners(self, node_id) + + def members(self, node_id): + calls.append(node_id) + return original_members(self, node_id) + + monkeypatch.setattr(affected._Ownership, "owners", owners) + monkeypatch.setattr(affected._Ownership, "members", members) + n = 500 + for graph_class in (nx.DiGraph, nx.Graph): + g = graph_class() + for i in range(n): + g.add_node(f"c{i}", label="Client", source_file=f"m{i}.py", source_location="L1") + g.add_node(f"s{i}", label=".send()", source_file=f"m{i}.py", source_location="L2") + g.add_edge(f"c{i}", f"s{i}", relation="method", _src=f"c{i}", _tgt=f"s{i}") + calls.clear() + assert len(affected.resolve_seed_candidates(g, "Client.send")) == n + assert len(calls) <= 3 * n, f"{len(calls)} ownership lookups for {n} owners" + + +def test_qualified_resolution_reads_undirected_edges_once(): + """On an undirected graph ownership is indexed from one whole-graph edge + pass per resolution. Rescanning every edge per lookup made `C0.run` on + 4,000 classes take 17 s.""" + from graphify.affected import resolve_seed + + g = nx.Graph() + for i in range(300): + g.add_node(f"c{i}", label=f"C{i}", source_file="pkg.py", source_location=f"L{i}") + g.add_node(f"m{i}", label=".run()", source_file="pkg.py", source_location=f"L{i}") + g.add_edge(f"c{i}", f"m{i}", relation="method", _src=f"c{i}", _tgt=f"m{i}") + view = g.edges + scans = [] + + class _CountingEdges: + def __call__(self, nbunch=None, data=False, default=None): + if nbunch is None: + scans.append(1) + return view(nbunch, data=data, default=default) + + def __iter__(self): + scans.append(1) + return iter(view) + + def __getattr__(self, name): + return getattr(view, name) + + # `Graph.edges` is a cached_property, so the instance dict holds the view. + g.__dict__["edges"] = _CountingEdges() + assert resolve_seed(g, "C0.run") == "m0" + assert resolve_seed(g, "pkg.py::C7.run") == "m7" + assert len(scans) <= 2, f"{len(scans)} whole-graph edge scans for two lookups" + + +def test_resolve_seed_keeps_dotfile_names_intact(): + """Only a method label (".name()") answers to its name without the dot. + `Config()` resolved to the class before; a `.config` dotfile must not tie it.""" + from graphify.affected import resolve_seed + + g = nx.DiGraph() + g.add_node("config_class", label="Config", source_file="config.py", source_location="L1") + g.add_node("dotconfig", label=".config", source_file=".config", source_location="L1") + assert resolve_seed(g, "Config()") == "config_class" + assert resolve_seed(g, "Config") == "config_class" + assert resolve_seed(g, ".config") == "dotconfig" + + +def _run_affected(monkeypatch, tmp_path, seed): + import pytest + + gp = tmp_path / "graph.json" + gp.write_text( + json.dumps(json_graph.node_link_data(_two_send_graph(), edges="links")), encoding="utf-8" + ) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + monkeypatch.setattr(mainmod.sys, "argv", ["graphify", "affected", seed, "--graph", str(gp)]) + with pytest.raises(SystemExit) as exc: + mainmod.main() + return exc.value.code + + +def test_affected_cli_qualified_seed_reports_its_own_callers(monkeypatch, tmp_path, capsys): + gp = tmp_path / "graph.json" + gp.write_text( + json.dumps(json_graph.node_link_data(_two_send_graph(), edges="links")), encoding="utf-8" + ) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + for seed, caller, other in ( + ("Client.send", "caller()", "async_caller()"), + ("pkg/_client.py::AsyncClient.send", "async_caller()", "caller()"), + ): + monkeypatch.setattr(mainmod.sys, "argv", ["graphify", "affected", seed, "--graph", str(gp)]) + mainmod.main() + out = capsys.readouterr().out + assert "Affected nodes for .send()" in out, seed + assert f"- {caller} [calls] app.py:" in out, seed + assert f"- {other} " not in out, seed + + +def test_affected_cli_ambiguous_seed_lists_candidates_and_fails(monkeypatch, tmp_path, capsys): + for seed in ("send", ".send()", "pkg/_client.py::send"): + code = _run_affected(monkeypatch, tmp_path, seed) + captured = capsys.readouterr() + assert code not in (0, None), seed + assert captured.out == "", seed + assert "pkg/_client.py:L20" in captured.err, seed + assert "pkg/_client.py:L60" in captured.err, seed + assert "client_send" in captured.err and "async_send" in captured.err, seed + assert "Affected nodes for" not in captured.err, seed + # The bare name also lists the nested function, and never the stub. + _run_affected(monkeypatch, tmp_path, "send") + err = capsys.readouterr().err + assert "pkg/asgi.py:L8" in err + assert "(id: send)" not in err + + +def test_affected_cli_missing_seed_fails(monkeypatch, tmp_path, capsys): + for seed in ("no_such_symbol", "Client.nope", "pkg/missing.py::Client.send"): + code = _run_affected(monkeypatch, tmp_path, seed) + captured = capsys.readouterr() + assert code not in (0, None), seed + assert captured.out == "", seed + assert f"No unique node match for {seed}" in captured.err, seed