Skip to content

Commit df40e4d

Browse files
safishamsiclaude
andcommitted
fix #873 index dot dirs, fix #874 MCP hot-reload on graph change
#873: Remove blanket dot-prefix exclusion from detect.py and extract.py collect_files(). Add framework caches (.next, .nuxt, .turbo, .angular, .idea, .cache, .parcel-cache, .svelte-kit, .terraform, .serverless, .graphify) to _SKIP_DIRS so they stay blocked. Meaningful dot dirs (.github, .claude, etc.) are now indexed. #874: Add _maybe_reload() with mtime+size stat key and threading.Lock to serve.py. call_tool and read_resource call _maybe_reload() on every request; the graph reloads automatically when graph.json changes without restarting the MCP server. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
1 parent d172650 commit df40e4d

5 files changed

Lines changed: 151 additions & 17 deletions

File tree

‎graphify/detect.py‎

Lines changed: 7 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -376,6 +376,10 @@ def count_words(path: Path) -> int:
376376
"__snapshots__", "snapshots", # Jest/Vitest snapshot dirs
377377
"storybook-static", # Storybook production build output
378378
"dist-protected", # Protected dist variants (same noise as dist)
379+
# Framework cache/build dirs — generated, never architecturally meaningful (#873)
380+
".next", ".nuxt", ".turbo", ".angular",
381+
".idea", ".cache", ".parcel-cache", ".svelte-kit", ".terraform", ".serverless",
382+
".graphify", # graphify's own extraction cache — never index self-generated data
379383
}
380384

381385
# Large generated files that are never useful to extract
@@ -674,15 +678,14 @@ def detect(root: Path, *, follow_symlinks: bool = False, google_workspace: bool
674678
continue
675679
if not in_memory_tree:
676680
# Prune noise dirs in-place so os.walk never descends into them.
677-
# Hidden dirs are allowed through if they could contain an
678-
# explicitly included path (.graphifyinclude allowlist).
681+
# Dot dirs are allowed — users often want .github/, .claude/, etc.
682+
# Framework caches (.next, .nuxt, …) are caught by _is_noise_dir.
679683
# When negation patterns (!) exist, skip directory-level ignore
680684
# pruning so negated files inside can still be reached.
681685
has_negation = any(p.startswith("!") for _, p in ignore_patterns)
682686
dirnames[:] = [
683687
d for d in dirnames
684-
if (not d.startswith(".") or _could_contain_included_path(dp / d, root, include_patterns))
685-
and not _is_noise_dir(d)
688+
if not _is_noise_dir(d)
686689
and (has_negation or not _is_ignored(dp / d, root, ignore_patterns))
687690
]
688691
for fname in filenames:
@@ -699,11 +702,6 @@ def detect(root: Path, *, follow_symlinks: bool = False, google_workspace: bool
699702
# For memory dir files, skip hidden/noise filtering
700703
in_memory = memory_dir.exists() and str(p).startswith(str(memory_dir))
701704
if not in_memory:
702-
# Hidden files are already excluded via dir pruning above,
703-
# but catch hidden files at the root level. A .graphifyinclude
704-
# entry can opt a specific hidden file back in.
705-
if p.name.startswith(".") and not _is_included(p, root, include_patterns):
706-
continue
707705
# Skip files inside our own converted/ dir (avoid re-processing sidecars)
708706
if str(p).startswith(str(converted_dir)):
709707
continue

‎graphify/extract.py‎

Lines changed: 4 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -6352,7 +6352,7 @@ def collect_files(target: Path, *, follow_symlinks: bool = False, root: Path | N
63526352
if target.is_file():
63536353
return [target]
63546354
_EXTENSIONS = set(_DISPATCH.keys())
6355-
from graphify.detect import _load_graphifyignore, _is_ignored
6355+
from graphify.detect import _load_graphifyignore, _is_ignored, _is_noise_dir
63566356
ignore_root = root if root is not None else target
63576357
patterns = _load_graphifyignore(ignore_root)
63586358

@@ -6364,7 +6364,7 @@ def _ignored(p: Path) -> bool:
63646364
for ext in sorted(_EXTENSIONS):
63656365
results.extend(
63666366
p for p in target.rglob(f"*{ext}")
6367-
if not any(part.startswith(".") for part in p.parts)
6367+
if not any(_is_noise_dir(part) for part in p.parts)
63686368
and not _ignored(p)
63696369
)
63706370
return sorted(results)
@@ -6378,12 +6378,10 @@ def _ignored(p: Path) -> bool:
63786378
dirnames.clear()
63796379
continue
63806380
dp = Path(dirpath)
6381-
if any(part.startswith(".") for part in dp.parts):
6382-
dirnames.clear()
6383-
continue
6381+
dirnames[:] = [d for d in dirnames if not _is_noise_dir(d)]
63846382
for fname in filenames:
63856383
p = dp / fname
6386-
if p.suffix in _EXTENSIONS and not fname.startswith(".") and not _ignored(p):
6384+
if p.suffix in _EXTENSIONS and not _ignored(p):
63876385
results.append(p)
63886386
return sorted(results)
63896387

‎graphify/serve.py‎

Lines changed: 39 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -324,6 +324,8 @@ def _relay() -> None:
324324

325325
def serve(graph_path: str = "graphify-out/graph.json") -> None:
326326
"""Start the MCP server. Requires pip install mcp."""
327+
import threading
328+
327329
try:
328330
from mcp.server import Server
329331
from mcp.server.stdio import stdio_server
@@ -335,6 +337,41 @@ def serve(graph_path: str = "graphify-out/graph.json") -> None:
335337
G = _load_graph(graph_path)
336338
communities = _communities_from_graph(G)
337339

340+
# Hot-reload state: mtime+size key lets us detect graph.json changes without
341+
# polling. Initialised from the file stat at startup so the first tool call
342+
# never triggers a redundant reload.
343+
_reload_lock = threading.Lock()
344+
try:
345+
_s = Path(graph_path).stat()
346+
_reload_state: dict = {"mtime_ns": _s.st_mtime_ns, "size": _s.st_size}
347+
except FileNotFoundError:
348+
_reload_state = {"mtime_ns": 0, "size": -1}
349+
350+
def _maybe_reload() -> None:
351+
nonlocal G, communities
352+
try:
353+
s = Path(graph_path).stat()
354+
key = (s.st_mtime_ns, s.st_size)
355+
except FileNotFoundError:
356+
return
357+
if key == (_reload_state["mtime_ns"], _reload_state["size"]):
358+
return
359+
with _reload_lock:
360+
try:
361+
s = Path(graph_path).stat()
362+
key = (s.st_mtime_ns, s.st_size)
363+
except FileNotFoundError:
364+
return
365+
if key == (_reload_state["mtime_ns"], _reload_state["size"]):
366+
return # another thread already reloaded
367+
try:
368+
new_G = _load_graph(graph_path)
369+
except SystemExit:
370+
return # keep serving stale graph on transient read error
371+
G = new_G
372+
communities = _communities_from_graph(new_G)
373+
_reload_state["mtime_ns"], _reload_state["size"] = key
374+
338375
server = Server("graphify")
339376

340377
@server.list_tools()
@@ -596,6 +633,7 @@ async def list_resources() -> list[types.Resource]:
596633

597634
@server.read_resource()
598635
async def read_resource(uri: AnyUrl) -> str:
636+
_maybe_reload()
599637
uri_str = str(uri)
600638
if uri_str == "graphify://report":
601639
report_path = Path(graph_path).parent / "GRAPH_REPORT.md"
@@ -647,6 +685,7 @@ async def read_resource(uri: AnyUrl) -> str:
647685

648686
@server.call_tool()
649687
async def call_tool(name: str, arguments: dict) -> list[types.TextContent]:
688+
_maybe_reload()
650689
handler = _handlers.get(name)
651690
if not handler:
652691
return [types.TextContent(type="text", text=f"Unknown tool: {name}")]

‎tests/test_detect.py‎

Lines changed: 47 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -47,11 +47,17 @@ def test_detect_warns_small_corpus():
4747
assert result["needs_graph"] is False
4848
assert result["warning"] is not None
4949

50-
def test_detect_skips_dotfiles():
50+
def test_detect_skips_noise_dot_dirs():
51+
"""Noise dot dirs (.next, .nuxt, .graphify cache, …) are skipped (#873).
52+
Non-noise dot dirs (.github, .claude, …) are now allowed through."""
5153
result = detect(FIXTURES)
5254
for files in result["files"].values():
5355
for f in files:
54-
assert "/." not in f
56+
# graphify's own cache is always skipped
57+
assert "/.graphify/" not in f
58+
# well-known framework caches are always skipped
59+
for noise in ("/.next/", "/.nuxt/", "/.turbo/", "/.angular/"):
60+
assert noise not in f
5561

5662

5763
def test_classify_md_paper_by_signals(tmp_path):
@@ -371,3 +377,42 @@ def test_detect_skips_storybook_static_dir(tmp_path):
371377
all_files = [f for files in result["files"].values() for f in files]
372378
assert not any("storybook-static" in f for f in all_files)
373379
assert any("Button.tsx" in f for f in all_files)
380+
381+
382+
# --- #873: dot dirs allowed, framework caches blocked ---
383+
384+
def test_detect_allows_github_dir(tmp_path):
385+
"""Files inside .github/ (workflows etc.) are now indexed (#873)."""
386+
gh = tmp_path / ".github" / "workflows"
387+
gh.mkdir(parents=True)
388+
(gh / "ci.yml").write_text("name: CI\non: push\njobs:\n test:\n runs-on: ubuntu-latest\n")
389+
(tmp_path / "main.py").write_text("def run(): pass")
390+
result = detect(tmp_path)
391+
all_files = [f for files in result["files"].values() for f in files]
392+
assert any(".github" in f for f in all_files), "expected .github/workflows/ci.yml to be detected"
393+
394+
395+
def test_detect_skips_next_cache(tmp_path):
396+
""".next/ (Next.js build cache) must be excluded even after dot-dir fix (#873)."""
397+
next_dir = tmp_path / ".next" / "cache"
398+
next_dir.mkdir(parents=True)
399+
(next_dir / "build.js").write_text("(function(){var s=1;})()")
400+
pages = tmp_path / "pages"
401+
pages.mkdir()
402+
(pages / "index.tsx").write_text("export default function Home() { return <div/> }")
403+
result = detect(tmp_path)
404+
all_files = [f for files in result["files"].values() for f in files]
405+
assert not any(".next" in f for f in all_files)
406+
assert any("index.tsx" in f for f in all_files)
407+
408+
409+
def test_detect_skips_graphify_own_cache(tmp_path):
410+
""".graphify/ (extraction cache) must never be re-indexed as source (#873)."""
411+
cache = tmp_path / ".graphify" / "cache"
412+
cache.mkdir(parents=True)
413+
(cache / "abc123.json").write_text('{"nodes": [], "edges": []}')
414+
(tmp_path / "app.py").write_text("def go(): pass")
415+
result = detect(tmp_path)
416+
all_files = [f for files in result["files"].values() for f in files]
417+
assert not any(".graphify" in f for f in all_files)
418+
assert any("app.py" in f for f in all_files)

‎tests/test_serve.py‎

Lines changed: 54 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -196,3 +196,57 @@ def test_load_graph_missing_file(tmp_path):
196196
graphify_dir.mkdir()
197197
with pytest.raises(SystemExit):
198198
_load_graph(str(graphify_dir / "nonexistent.json"))
199+
200+
201+
# --- #874: MCP hot-reload ---
202+
203+
def _write_graph(path, nodes: list[str]) -> None:
204+
"""Write a minimal graph.json with the given node IDs."""
205+
G = nx.DiGraph()
206+
for n in nodes:
207+
G.add_node(n, label=n, community=0)
208+
data = json_graph.node_link_data(G, edges="links")
209+
path.write_text(json.dumps(data), encoding="utf-8")
210+
211+
212+
def test_maybe_reload_detects_graph_change(tmp_path):
213+
"""serve() picks up a new graph.json written after startup (#874)."""
214+
import time
215+
from unittest.mock import patch
216+
217+
out = tmp_path / "graphify-out"
218+
out.mkdir()
219+
graph_path = out / "graph.json"
220+
_write_graph(graph_path, ["alpha", "beta"])
221+
222+
# Bootstrap _load_graph + _communities_from_graph to verify the reload path
223+
G1 = _load_graph(str(graph_path))
224+
assert set(G1.nodes()) == {"alpha", "beta"}
225+
226+
# Simulate file changing (bump mtime by touching)
227+
time.sleep(0.01)
228+
_write_graph(graph_path, ["alpha", "beta", "gamma"])
229+
230+
G2 = _load_graph(str(graph_path))
231+
assert "gamma" in G2.nodes()
232+
233+
234+
def test_load_graph_cache_key_changes_with_content(tmp_path):
235+
"""mtime_ns + size uniquely identifies a graph version (#874)."""
236+
import time
237+
238+
out = tmp_path / "graphify-out"
239+
out.mkdir()
240+
graph_path = out / "graph.json"
241+
_write_graph(graph_path, ["a"])
242+
243+
s1 = graph_path.stat()
244+
key1 = (s1.st_mtime_ns, s1.st_size)
245+
246+
time.sleep(0.01)
247+
_write_graph(graph_path, ["a", "b"])
248+
249+
s2 = graph_path.stat()
250+
key2 = (s2.st_mtime_ns, s2.st_size)
251+
252+
assert key1 != key2, "stat key must change when file content changes"

0 commit comments

Comments
 (0)