Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -338,6 +338,13 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg
- **Suggested questions** — 4–5 questions the graph is uniquely positioned to answer.
- **Confidence tags** — every inferred relationship is marked `EXTRACTED`, `INFERRED`, or `AMBIGUOUS`. You always know what was found vs guessed.

JavaScript files that yield only file nodes or simple bindings are listed under
**Files without structural symbols** in `GRAPH_REPORT.md`, with byte sizes and
a bounded list of the largest files. This flags possible data banks that AST
structure cannot describe; it does not automatically run semantic extraction.
Files with functions, classes, imports, calls, or parser errors are excluded
from this coverage notice.

---

## What files it handles
Expand Down
15 changes: 15 additions & 0 deletions graphify/extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -7468,6 +7468,21 @@ def extract(
file=sys.stderr, flush=True,
)

# #3946: JavaScript data banks can yield only a file node and literal
# bindings. Keep this limitation visible in persisted graphs/reports; this
# is a structural-coverage signal, not a data classifier or an LLM fallback.
_symbol_free_count = sum(
1 for result in per_file for node in (result or {}).get("nodes", [])
if node.get("_no_structural_symbols")
)
if _symbol_free_count:
print(
f" note: {_symbol_free_count} JavaScript file(s) without functions, classes, imports or calls; "
"their contents may contain data not extracted by AST. See the "
"Files without structural symbols section in GRAPH_REPORT.md.",
file=sys.stderr, flush=True,
)

# #2543: collect sources that must NOT be stamped as up-to-date in the
# incremental manifest. Two cases:
# - extractor returned an error (missing optional extra, parse failure, …)
Expand Down
29 changes: 29 additions & 0 deletions graphify/extractors/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -3698,6 +3698,7 @@ def _extract_generic(
try:
parser = Parser(language)
source = path.read_bytes() if source_override is None else source_override
source_bytes = len(source)
# In C and C++, if the .h file does not end with a newline '\n' an error
# is throwed even if the file is valid. In order to avoid this, a new line
# char is added only if the original file does not end with it.
Expand Down Expand Up @@ -7174,6 +7175,34 @@ def _scan_js_module_dispatch(n) -> None:
# fold them in so the cross-file resolver sees them (#1668).
if _ruby_mixin_calls:
raw_calls.extend(_ruby_mixin_calls)
# #3946: inspect the already parsed JavaScript tree, including call
# initializers that may not have produced graph edges. A file/binding node
# alone does not describe the contents of a data bank. Persist the signal
# on the file node so cluster-only reports need no original source access.
if (config.ts_module == "tree_sitter_javascript" and not root.has_error
and path.suffix.lower() in {".js", ".jsx", ".mjs", ".cjs"}
and not raw_calls
and all(e.get("relation") == "contains" for e in clean_edges)):
structural_types = {
"function_declaration", "generator_function_declaration",
"function_expression", "generator_function", "arrow_function",
"method_definition", "class", "class_declaration",
"import_statement", "call_expression", "new_expression",
}
pending = [root]
has_structure = False
while pending:
current = pending.pop()
if current.type in structural_types:
has_structure = True
break
pending.extend(current.named_children)
if not has_structure:
file_node = next((n for n in nodes if n["id"] == file_nid), None)
if file_node is not None:
file_node["_no_structural_symbols"] = True
file_node["_source_bytes"] = source_bytes

result = {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
# Export the per-file field->type tables for the corpus member-call
# resolvers (#3151): a field declared on a superclass in ANOTHER file can
Expand Down
23 changes: 23 additions & 0 deletions graphify/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -323,6 +323,29 @@ def _real_count(nodes) -> int:
f" {d.get('source_file', '')} · relation: {d.get('relation', 'unknown')}",
]

# File nodes are intentionally omitted from Knowledge Gaps. Report the
# AST coverage limitation separately, using persisted metadata so report
# regeneration never needs to reopen the original corpus (#3946).
symbol_free = [d for _, d in G.nodes(data=True) if d.get("_no_structural_symbols")]
if symbol_free:
symbol_free.sort(key=lambda d: (-d.get("_source_bytes", 0), d.get("source_file", "")))
lines += [
"", "## Files without structural symbols",
f"{len(symbol_free)} JavaScript file(s) yielded no functions, classes, imports or calls. "
"They may contain data whose contents are not extracted by AST; "
"file nodes and simple bindings do not imply complete coverage.",
]
for data in symbol_free[:5]:
path = str(data.get("source_file", data.get("label", "")))
path = path.replace("\n", "\\n").replace("\r", "\\r")
# Backslashes do not escape backticks inside Markdown code spans.
# A longer delimiter and padding preserve arbitrary source names.
delimiter = "`" * (max((len(run) for run in re.findall(r"`+", path)), default=0) + 1)
lines.append(f"- {delimiter} {path} {delimiter} — {data.get('_source_bytes', 0):,} bytes")
if len(symbol_free) > 5:
lines.append(f"- {len(symbol_free) - 5} more file(s).")
lines.append("Review these files and choose whether their data belongs in the graph before requesting semantic extraction.")

# --- Gaps section ---
isolated = [n for n in G.nodes() if G.degree(n) <= 1 and _real_node(n)]
# Same threshold the Summary and Communities headers used (#3148): this
Expand Down
108 changes: 108 additions & 0 deletions tests/test_data_only_code.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
"""Data-shaped JavaScript must not look like complete structural coverage."""
import json

import pytest

from graphify.build import build_from_json
from graphify.extract import extract
from graphify.report import generate


DATA_FORMS = [
"const BANK = [{question: 'why?', answer: 'because'}];",
"var BANK = [1, 2];",
"window.BANK = [1, 2];",
"module.exports = [1, 2];",
"export default [1, 2];",
"export const BANK = [1, 2];",
"const namespace = {BANK: [1, 2]};",
]


def report_for(result):
graph = build_from_json(json.loads(json.dumps(result)))
return generate(graph, {}, {}, {}, [], [], {"total_files": 1, "total_words": 0}, {}, ".")


@pytest.mark.parametrize("source", DATA_FORMS)
def test_data_forms_are_visible_in_extraction_and_persisted_report(tmp_path, capsys, source):
p = tmp_path / "bank.js"
p.write_text(source, encoding="utf-8")
result = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
file_node = next(n for n in result["nodes"] if n["label"] == "bank.js")
assert file_node["_no_structural_symbols"] is True
assert file_node["_source_bytes"] == len(source.encode("utf-8"))
assert result["failed_sources"] == []
assert "without functions, classes, imports or calls" in capsys.readouterr().err
report = report_for(result)
assert "## Files without structural symbols" in report
assert "bank.js" in report and f"{len(source)} bytes" in report
assert "may contain data" in report
assert "not extracted by AST" in report


@pytest.mark.parametrize("source", [
"function foo() {}", "class Foo {}", "const foo = () => 1;",
"import {foo} from './foo';", "const BANK = loadData();",
"export {foo} from './foo';", "const obj = {foo() {return 1;}};",
])
def test_structural_js_is_not_reported_as_data_only(tmp_path, source):
p = tmp_path / "code.js"
p.write_text(source)
result = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
assert not any(n.get("_no_structural_symbols") for n in result["nodes"])
assert "## Files without structural symbols" not in report_for(result)


def test_report_bounds_list_and_orders_by_size(tmp_path):
paths = []
for i in range(8):
p = tmp_path / f"bank{i}.js"
p.write_text("window.BANK = [" + ",".join(["1"] * (i + 1)) + "];" )
paths.append(p)
r = extract(paths, root=tmp_path, cache_root=tmp_path, parallel=False)
section = report_for(r).split("## Files without structural symbols", 1)[1].split("## ", 1)[0]
assert "8 JavaScript" in section
assert section.count(" bytes") == 5
assert section.index("bank7.js") < section.index("bank6.js")
assert "3 more" in section


def test_syntax_errors_are_not_classified_as_data_only(tmp_path):
p = tmp_path / "broken.js"
p.write_text("function foo( {")
r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
assert not any(n.get("_no_structural_symbols") for n in r["nodes"])


@pytest.mark.parametrize("extension", [".js", ".jsx", ".mjs", ".cjs"])
def test_byte_sizes_are_utf8_and_report_survives_source_removal(tmp_path, extension):
p = tmp_path / ("bank" + extension)
source = 'window.BANK = ["caf' + chr(233) + '"];'
p.write_text(source, encoding="utf-8")
r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
expected = len(source.encode("utf-8"))
assert next(n for n in r["nodes"] if n.get("_no_structural_symbols"))["_source_bytes"] == expected
p.unlink()
assert f"{expected} bytes" in report_for(r)


def test_incremental_replacement_removes_old_coverage_notice(tmp_path):
from graphify.build import merge_raw_extraction
p = tmp_path / "bank.js"
p.write_text("window.BANK = [1];")
old = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
p.write_text("function answer() {return 42;}")
fresh = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
graph_path = tmp_path / "graph.json"
graph_path.write_text(json.dumps(old), encoding="utf-8")
combined = merge_raw_extraction(fresh, graph_path, root=str(tmp_path))
assert "## Files without structural symbols" not in report_for(combined)


def test_report_preserves_backticks_in_source_filename(tmp_path):
p = tmp_path / "bank`draft.js"
source = "window.BANK = [1];"
p.write_text(source, encoding="utf-8")
r = extract([p], root=tmp_path, cache_root=tmp_path, parallel=False)
assert "- `` bank`draft.js ``" in report_for(r)
Loading