From 90e4e42141d15b410b8188423463e2c719f3ac31 Mon Sep 17 00:00:00 2001 From: azizur100389 Date: Thu, 1 Oct 2026 19:39:56 +0100 Subject: [PATCH 1/2] fix(report): surface JavaScript files without structural symbols Co-Authored-By: OpenAI Codex --- README.md | 7 +++ graphify/extract.py | 15 +++++ graphify/extractors/engine.py | 29 ++++++++++ graphify/report.py | 20 +++++++ tests/test_data_only_code.py | 100 ++++++++++++++++++++++++++++++++++ 5 files changed, 171 insertions(+) create mode 100644 tests/test_data_only_code.py diff --git a/README.md b/README.md index 46d7291cb2..e2d1ae273d 100644 --- a/README.md +++ b/README.md @@ -338,6 +338,13 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg - **Suggested questions** — 4–5 questions the graph is uniquely positioned to answer. - **Confidence tags** — every inferred relationship is marked `EXTRACTED`, `INFERRED`, or `AMBIGUOUS`. You always know what was found vs guessed. +JavaScript files that yield only file nodes or simple bindings are listed under +**Files without structural symbols** in `GRAPH_REPORT.md`, with byte sizes and +a bounded list of the largest files. This flags possible data banks that AST +structure cannot describe; it does not automatically run semantic extraction. +Files with functions, classes, imports, calls, or parser errors are excluded +from this coverage notice. + --- ## What files it handles diff --git a/graphify/extract.py b/graphify/extract.py index b54889c8bb..14de73d395 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -7468,6 +7468,21 @@ def extract( file=sys.stderr, flush=True, ) + # #3946: JavaScript data banks can yield only a file node and literal + # bindings. Keep this limitation visible in persisted graphs/reports; this + # is a structural-coverage signal, not a data classifier or an LLM fallback. + _symbol_free_count = sum( + 1 for result in per_file for node in (result or {}).get("nodes", []) + if node.get("_no_structural_symbols") + ) + if _symbol_free_count: + print( + f" note: {_symbol_free_count} JavaScript file(s) without functions, classes, imports or calls; " + "their contents may contain data not extracted by AST. See the " + "Files without structural symbols section in GRAPH_REPORT.md.", + file=sys.stderr, flush=True, + ) + # #2543: collect sources that must NOT be stamped as up-to-date in the # incremental manifest. Two cases: # - extractor returned an error (missing optional extra, parse failure, …) diff --git a/graphify/extractors/engine.py b/graphify/extractors/engine.py index d5f48054ac..8a123ea5af 100644 --- a/graphify/extractors/engine.py +++ b/graphify/extractors/engine.py @@ -3698,6 +3698,7 @@ def _extract_generic( try: parser = Parser(language) source = path.read_bytes() if source_override is None else source_override + source_bytes = len(source) # In C and C++, if the .h file does not end with a newline '\n' an error # is throwed even if the file is valid. In order to avoid this, a new line # char is added only if the original file does not end with it. @@ -7174,6 +7175,34 @@ def _scan_js_module_dispatch(n) -> None: # fold them in so the cross-file resolver sees them (#1668). if _ruby_mixin_calls: raw_calls.extend(_ruby_mixin_calls) + # #3946: inspect the already parsed JavaScript tree, including call + # initializers that may not have produced graph edges. A file/binding node + # alone does not describe the contents of a data bank. Persist the signal + # on the file node so cluster-only reports need no original source access. + if (config.ts_module == "tree_sitter_javascript" and not root.has_error + and path.suffix.lower() in {".js", ".jsx", ".mjs", ".cjs"} + and not raw_calls + and all(e.get("relation") == "contains" for e in clean_edges)): + structural_types = { + "function_declaration", "generator_function_declaration", + "function_expression", "generator_function", "arrow_function", + "method_definition", "class", "class_declaration", + "import_statement", "call_expression", "new_expression", + } + pending = [root] + has_structure = False + while pending: + current = pending.pop() + if current.type in structural_types: + has_structure = True + break + pending.extend(current.named_children) + if not has_structure: + file_node = next((n for n in nodes if n["id"] == file_nid), None) + if file_node is not None: + file_node["_no_structural_symbols"] = True + file_node["_source_bytes"] = source_bytes + result = {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls} # Export the per-file field->type tables for the corpus member-call # resolvers (#3151): a field declared on a superclass in ANOTHER file can diff --git a/graphify/report.py b/graphify/report.py index 168197d54e..93c7109aae 100644 --- a/graphify/report.py +++ b/graphify/report.py @@ -323,6 +323,26 @@ def _real_count(nodes) -> int: f" {d.get('source_file', '')} · relation: {d.get('relation', 'unknown')}", ] + # File nodes are intentionally omitted from Knowledge Gaps. Report the + # AST coverage limitation separately, using persisted metadata so report + # regeneration never needs to reopen the original corpus (#3946). + symbol_free = [d for _, d in G.nodes(data=True) if d.get("_no_structural_symbols")] + if symbol_free: + symbol_free.sort(key=lambda d: (-d.get("_source_bytes", 0), d.get("source_file", ""))) + lines += [ + "", "## Files without structural symbols", + f"{len(symbol_free)} JavaScript file(s) yielded no functions, classes, imports or calls. " + "They may contain data whose contents are not extracted by AST; " + "file nodes and simple bindings do not imply complete coverage.", + ] + for data in symbol_free[:5]: + path = str(data.get("source_file", data.get("label", ""))) + path = path.replace("\n", "\\n").replace("\r", "\\r").replace("`", "\\`") + lines.append(f"- `{path}` — {data.get('_source_bytes', 0):,} bytes") + if len(symbol_free) > 5: + lines.append(f"- {len(symbol_free) - 5} more file(s).") + lines.append("Review these files and choose whether their data belongs in the graph before requesting semantic extraction.") + # --- Gaps section --- isolated = [n for n in G.nodes() if G.degree(n) <= 1 and _real_node(n)] # Same threshold the Summary and Communities headers used (#3148): this diff --git a/tests/test_data_only_code.py b/tests/test_data_only_code.py new file mode 100644 index 0000000000..70aa9f7a63 --- /dev/null +++ b/tests/test_data_only_code.py @@ -0,0 +1,100 @@ +"""Data-shaped JavaScript must not look like complete structural coverage.""" +import json + +import pytest + +from graphify.build import build_from_json +from graphify.extract import extract +from graphify.report import generate + + +DATA_FORMS = [ + "const BANK = [{question: 'why?', answer: 'because'}];", + "var BANK = [1, 2];", + "window.BANK = [1, 2];", + "module.exports = [1, 2];", + "export default [1, 2];", + "export const BANK = [1, 2];", + "const namespace = {BANK: [1, 2]};", +] + + +def report_for(result): + graph = build_from_json(json.loads(json.dumps(result))) + return generate(graph, {}, {}, {}, [], [], {"total_files": 1, "total_words": 0}, {}, ".") + + +@pytest.mark.parametrize("source", DATA_FORMS) +def test_data_forms_are_visible_in_extraction_and_persisted_report(tmp_path, capsys, source): + p = tmp_path / "bank.js" + p.write_text(source, encoding="utf-8") + result = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + file_node = next(n for n in result["nodes"] if n["label"] == "bank.js") + assert file_node["_no_structural_symbols"] is True + assert file_node["_source_bytes"] == len(source.encode("utf-8")) + assert result["failed_sources"] == [] + assert "without functions, classes, imports or calls" in capsys.readouterr().err + report = report_for(result) + assert "## Files without structural symbols" in report + assert "bank.js" in report and f"{len(source)} bytes" in report + assert "may contain data" in report + assert "not extracted by AST" in report + + +@pytest.mark.parametrize("source", [ + "function foo() {}", "class Foo {}", "const foo = () => 1;", + "import {foo} from './foo';", "const BANK = loadData();", + "export {foo} from './foo';", "const obj = {foo() {return 1;}};", +]) +def test_structural_js_is_not_reported_as_data_only(tmp_path, source): + p = tmp_path / "code.js" + p.write_text(source) + result = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + assert not any(n.get("_no_structural_symbols") for n in result["nodes"]) + assert "## Files without structural symbols" not in report_for(result) + + +def test_report_bounds_list_and_orders_by_size(tmp_path): + paths = [] + for i in range(8): + p = tmp_path / f"bank{i}.js" + p.write_text("window.BANK = [" + ",".join(["1"] * (i + 1)) + "];" ) + paths.append(p) + r = extract(paths, root=tmp_path, cache_root=tmp_path, parallel=False) + section = report_for(r).split("## Files without structural symbols", 1)[1].split("## ", 1)[0] + assert "8 JavaScript" in section + assert section.count(" bytes") == 5 + assert section.index("bank7.js") < section.index("bank6.js") + assert "3 more" in section + + +def test_syntax_errors_are_not_classified_as_data_only(tmp_path): + p = tmp_path / "broken.js" + p.write_text("function foo( {") + r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + assert not any(n.get("_no_structural_symbols") for n in r["nodes"]) + + +@pytest.mark.parametrize("extension", [".js", ".jsx", ".mjs", ".cjs"]) +def test_byte_sizes_are_utf8_and_report_survives_source_removal(tmp_path, extension): + p = tmp_path / ("bank" + extension) + source = 'window.BANK = ["caf' + chr(233) + '"];' + p.write_text(source, encoding="utf-8") + r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + expected = len(source.encode("utf-8")) + assert next(n for n in r["nodes"] if n.get("_no_structural_symbols"))["_source_bytes"] == expected + p.unlink() + assert f"{expected} bytes" in report_for(r) + + +def test_incremental_replacement_removes_old_coverage_notice(tmp_path): + from graphify.build import merge_raw_extraction + p = tmp_path / "bank.js" + p.write_text("window.BANK = [1];") + old = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + p.write_text("function answer() {return 42;}") + fresh = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + graph_path = tmp_path / "graph.json" + graph_path.write_text(json.dumps(old), encoding="utf-8") + combined = merge_raw_extraction(fresh, graph_path, root=str(tmp_path)) + assert "## Files without structural symbols" not in report_for(combined) From be288478d8146c41de669593cacc24662a2a541a Mon Sep 17 00:00:00 2001 From: azizur100389 Date: Thu, 1 Oct 2026 20:22:05 +0100 Subject: [PATCH 2/2] fix(report): preserve backticks in source names Co-Authored-By: OpenAI Codex --- graphify/report.py | 7 +++++-- tests/test_data_only_code.py | 8 ++++++++ 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/graphify/report.py b/graphify/report.py index 93c7109aae..6d3ca599fe 100644 --- a/graphify/report.py +++ b/graphify/report.py @@ -337,8 +337,11 @@ def _real_count(nodes) -> int: ] for data in symbol_free[:5]: path = str(data.get("source_file", data.get("label", ""))) - path = path.replace("\n", "\\n").replace("\r", "\\r").replace("`", "\\`") - lines.append(f"- `{path}` — {data.get('_source_bytes', 0):,} bytes") + path = path.replace("\n", "\\n").replace("\r", "\\r") + # Backslashes do not escape backticks inside Markdown code spans. + # A longer delimiter and padding preserve arbitrary source names. + delimiter = "`" * (max((len(run) for run in re.findall(r"`+", path)), default=0) + 1) + lines.append(f"- {delimiter} {path} {delimiter} — {data.get('_source_bytes', 0):,} bytes") if len(symbol_free) > 5: lines.append(f"- {len(symbol_free) - 5} more file(s).") lines.append("Review these files and choose whether their data belongs in the graph before requesting semantic extraction.") diff --git a/tests/test_data_only_code.py b/tests/test_data_only_code.py index 70aa9f7a63..1feb38b26a 100644 --- a/tests/test_data_only_code.py +++ b/tests/test_data_only_code.py @@ -98,3 +98,11 @@ def test_incremental_replacement_removes_old_coverage_notice(tmp_path): graph_path.write_text(json.dumps(old), encoding="utf-8") combined = merge_raw_extraction(fresh, graph_path, root=str(tmp_path)) assert "## Files without structural symbols" not in report_for(combined) + + +def test_report_preserves_backticks_in_source_filename(tmp_path): + p = tmp_path / "bank`draft.js" + source = "window.BANK = [1];" + p.write_text(source, encoding="utf-8") + r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False) + assert "- `` bank`draft.js ``" in report_for(r)