From 1a467ae4269a3bd0c9db76a15b58163eaabc7a8f Mon Sep 17 00:00:00 2001 From: Sapientropic Date: Wed, 8 Jul 2026 15:51:20 +0800 Subject: [PATCH 1/4] Fix open foreground issue closeouts --- .../foreground_compact_language.py | 2 + .../mcp/agent_recall_projection.py | 52 +++ .../mcp/current_source_route_policy.py | 39 +++ .../ops/doctors/provider_doctor_projection.py | 18 +- .../ops/storage_governance_projection.py | 18 +- .../test_agent_recall_compact_projection.py | 49 +++ ...st_aippocampus_mcp_server_search_memory.py | 24 ++ .../aippocampus/test_compact_surface_scan.py | 63 ++++ tests/aippocampus/test_provider_doctor.py | 11 +- tests/aippocampus/test_run_tests_tiers.py | 18 +- tests/aippocampus/test_search_clean_source.py | 2 +- ...arch_clean_source_foreground_projection.py | 2 +- tests/aippocampus/test_storage_governance.py | 62 ++-- tests/aippocampus/test_update_sync.py | 12 +- tools/aippocampus/compact_surface_scan.py | 317 +++++++++++++++--- tools/aippocampus/run_tests.py | 64 +++- 16 files changed, 650 insertions(+), 103 deletions(-) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/foreground_compact_language.py b/skills/aippocampus/scripts/aippocampus_runtime/foreground_compact_language.py index 913696f4..ec655512 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/foreground_compact_language.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/foreground_compact_language.py @@ -34,8 +34,10 @@ "output_boundary", "policy_boundary", "privacy_boundary", + "risk_boundary", "source_boundary", "source_reopen_boundary", + "suppression_boundary", "write_boundary", } ) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py b/skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py index f7d930df..4077145f 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py @@ -351,6 +351,58 @@ def _select_initial_foreground_action( foreground_action, context.recall_selector, ) + if not current_source_route_policy.primary_deepen_followthrough_reopenable( + context.memory_packets, + ): + registry_fallback = recall_choices.registry_source_search_fallback_action( + context.recovery_cue + ) + if registry_fallback: + registry_fallback["route_choice_posture"] = "route_note_requires_source_search" + registry_fallback["why"] = ( + "The top recall route is route-note navigation without a " + "reopenable clean-source handle; search registered sources for " + "the original cue anchors before deepening or claiming." + ) + foreground_action = registry_fallback + safe_next_actions = [ + { + "id": "refine_low_specificity_recall_cue", + "label": "Refine low-specificity recall cue", + "tool_name": "agent_recall", + "arguments_template": {"query": "{tighter_cue}", "max": 3}, + "requires": ["tighter_cue"], + "template_only": True, + "command_template": 'aippocampus agent recall "{tighter_cue}" --json', + "route_choice_posture": "route_note_requires_source_search", + "mutation_risk": "read_only", + "claim_boundary": "no_claim_before_reopen", + "why": ( + "If source search does not find the right anchor, " + "tighten the cue with a more specific phrase, object, " + "person, or time clue." + ), + } + ] + else: + weak_route_recovery_card = _weak_route_recovery_card() + foreground_action = { + "action_id": "recover_route_note_without_source_ref", + "label": "Recover route-note recall", + "tool_name": "search_memory", + "why": ( + "The top route-note candidate has no reopenable source " + "handle; provide a more specific cue before relying on it." + ), + "mutation_risk": "read_only", + "claim_boundary": "no_claim_before_reopen", + } | context.search_fields + return ForegroundSelection( + foreground_action=foreground_action, + miss_recovery_card=miss_recovery_card, + weak_route_recovery_card=weak_route_recovery_card, + safe_next_actions=safe_next_actions, + ) foreground_action = recall_choices.with_low_specificity_foreground_action( foreground_action, metrics=context.metrics, diff --git a/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py b/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py index 679270ca..6aa1c3b3 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py @@ -23,6 +23,45 @@ def primary_current_source_reopenable(memory_packets: list[dict[str, Any]]) -> b ) +def primary_deepen_followthrough_reopenable(memory_packets: list[dict[str, Any]]) -> bool: + """Return whether the primary route is safe to expose as agent_deepen. + + Low-specificity recall sometimes surfaces route-note navigation scents. A + route note is useful orientation, but it should not become the first + foreground `agent_deepen` action unless it carries joined clean-source + evidence that `agent_deepen` can reopen. + """ + + if not memory_packets: + return False + primary = memory_packets[0] + if primary.get("output_mode") != "reopenable_route": + return False + if not _route_note_like(primary): + return True + return _route_note_has_joined_source_ref(primary) + + +def _route_note_like(packet: Mapping[str, Any]) -> bool: + markers = { + str(packet.get("route_kind") or ""), + str(packet.get("matched_cue_family") or ""), + str(packet.get("origin") or ""), + str(packet.get("route_origin") or ""), + str(packet.get("source") or ""), + } + return any("route_note" in marker.casefold() for marker in markers) + + +def _route_note_has_joined_source_ref(packet: Mapping[str, Any]) -> bool: + for key in ("joined_evidence_refs", "source_refs"): + value = packet.get(key) + if isinstance(value, list) and any(isinstance(item, Mapping) for item in value): + return True + source_ref = packet.get("source_ref") + return isinstance(source_ref, Mapping) and bool(source_ref.get("message_id")) + + def apw_card_allows_primary( *, should_replace: bool, diff --git a/skills/aippocampus/scripts/aippocampus_runtime/ops/doctors/provider_doctor_projection.py b/skills/aippocampus/scripts/aippocampus_runtime/ops/doctors/provider_doctor_projection.py index 98db9202..d292e1c3 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/ops/doctors/provider_doctor_projection.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/ops/doctors/provider_doctor_projection.py @@ -10,6 +10,7 @@ canonical_foreground_action_fields, foreground_shell_action, ) +from aippocampus_runtime.foreground_compact_language import compact_frontstage_projection from aippocampus_runtime.model.routing import DEFAULT_DEEPSEEK_API_KEY_ENV from aippocampus_runtime.ops.doctors.common import as_dict @@ -146,11 +147,18 @@ def compact_provider_doctor_card(report: dict[str, Any]) -> dict[str, Any]: "frontstage_rule": "readiness and next check first; diagnostics stay in full detail", }, } - return { - key: value - for key, value in card.items() - if value not in (None, "", [], {}) - } + return compact_frontstage_projection( + { + key: value + for key, value in card.items() + if value not in (None, "", [], {}) + }, + extra_denied_keys={ + "audit_json_available", + "boundary_summary", + "full_audit_command", + }, + ) def render_text(report: dict[str, Any]) -> str: diff --git a/skills/aippocampus/scripts/aippocampus_runtime/ops/storage_governance_projection.py b/skills/aippocampus/scripts/aippocampus_runtime/ops/storage_governance_projection.py index 799cf117..65ea20f6 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/ops/storage_governance_projection.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/ops/storage_governance_projection.py @@ -7,6 +7,7 @@ from aippocampus_runtime import core from aippocampus_runtime.contracts import canonical_foreground_action_fields +from aippocampus_runtime.foreground_compact_language import compact_frontstage_projection from aippocampus_runtime.ops.storage_governance_actions import ( candidate_can_offer_compact_apply, storage_gc_summary_actions, @@ -295,6 +296,8 @@ def bounded_cli_projection( action_fields = canonical_foreground_action_fields( summary_actions[0], safe_next_actions=summary_actions, + max_safe_next_actions=1, + safe_next_read_only_only=True, ) next_steps = [ "Continue normal work when cleanup was not the user goal.", @@ -392,6 +395,19 @@ def bounded_cli_projection( canonical_foreground_action_fields( primary, safe_next_actions=[primary, *existing_actions], + max_safe_next_actions=1, + safe_next_read_only_only=True, ) ) - return projection + projection["details_available"] = True + return compact_frontstage_projection( + projection, + extra_denied_keys={ + "comparable_metrics_command", + "full_audit_available", + "full_audit_flag", + "operator_audit_command", + "policy_model", + "report_sources", + }, + ) diff --git a/tests/aippocampus/test_agent_recall_compact_projection.py b/tests/aippocampus/test_agent_recall_compact_projection.py index 31c297f5..e607b64b 100644 --- a/tests/aippocampus/test_agent_recall_compact_projection.py +++ b/tests/aippocampus/test_agent_recall_compact_projection.py @@ -289,6 +289,55 @@ def test_low_specificity_thread_candidate_choices_get_public_safe_differentiator self.assertNotIn("C:\\", encoded) assert_compact_frontstage_payload(self, public, max_top_level_diagnostics=1) + def test_route_note_only_low_specificity_route_uses_source_search_not_blind_deepen(self) -> None: + public = agent_continuity_cli_support.public_recall_projection( + { + "kind": "aippocampus_agent_continuity_path", + "schema_version": "agent-continuity-path-v1", + "mode": "recall", + "status": "ok", + "last_recall_cache_available": True, + "recall_selector_id": "sel_route_note_only", + "foreground_action_card": { + "decision": "use_route_first", + "canonical_action": { + "action_id": "agent_deepen_selected_route", + "tool_name": "agent_deepen", + "arguments": {"request_index": 1, "last_recall": True}, + "claim_boundary": "no_claim_before_reopen", + }, + }, + "memory_packets": [ + { + "route_id": "route_note_only", + "route_label": "Low-specificity route", + "route_kind": "route_note", + "matched_cue_family": "route_note_title", + "output_mode": "reopenable_route", + "claim_permission": "no_claim_before_reopen", + } + ], + "metrics": { + "memory_packet_count": 1, + "deepen_request_count": 1, + "route_label_specificity_floor": 0.0, + "topic_label_present_count": 0, + }, + }, + query="之前 key persistence 那个模糊线索", + ) + + action = public["foreground_action"] + self.assertEqual(action["id"], "search_registry_sources_for_original_cue_anchors") + self.assertEqual(action["tool_name"], "search_memory") + self.assertEqual(action["arguments"]["scope"], "all_registered_sources") + self.assertIn("route-note navigation", action["why"]) + self.assertEqual( + [item["id"] for item in public["safe_next_actions"]], + ["refine_low_specificity_recall_cue"], + ) + self.assertNotIn("source_ref_not_found", json.dumps(public, ensure_ascii=False)) + def test_full_action_menu_is_foreground_compatible_without_detail_switch_claim(self) -> None: payload = { "kind": "aippocampus_agent_continuity_path", diff --git a/tests/aippocampus/test_aippocampus_mcp_server_search_memory.py b/tests/aippocampus/test_aippocampus_mcp_server_search_memory.py index c0d04f7d..e7e870db 100644 --- a/tests/aippocampus/test_aippocampus_mcp_server_search_memory.py +++ b/tests/aippocampus/test_aippocampus_mcp_server_search_memory.py @@ -97,6 +97,30 @@ def test_search_memory_open_source_compact_returns_bounded_source_snippet(self) self.assertNotIn("source_window", open_payload) self.assertNotIn(str(self.cwd), json.dumps(open_payload, ensure_ascii=False)) + def test_search_memory_accepts_common_schema_args_without_handler_mismatch(self) -> None: + anchor = "MCP schema handler detail compatibility anchor" + self._write_registry_thread(anchor=anchor) + + payload = call_mcp_tool_payload( + "search_memory", + { + "query": anchor, + "scope": "all_registered_sources", + "registry_dir": str(self.cwd), + "cwd": str(self.cwd), + "detail": "compact", + "search_budget": "deep", + "max_elapsed_ms": 15000, + "max": 5, + }, + ) + encoded = json.dumps(payload, ensure_ascii=False) + + self.assertEqual(payload["kind"], "aippocampus_registry_source_search") + self.assertEqual(payload["mcp_search_scope"], "all_registered_sources") + self.assertNotIn("unexpected keyword argument", encoded) + self.assertNotIn("TypeError", encoded) + def test_search_memory_near_hit_candidate_opens_source_as_navigation(self) -> None: query = "隐私保护 到底 要到什么程度" source_text = "隐私保护 到底 要帮用户继续,而不是变成隐私阻断器。" diff --git a/tests/aippocampus/test_compact_surface_scan.py b/tests/aippocampus/test_compact_surface_scan.py index 829be70d..c6a06a41 100644 --- a/tests/aippocampus/test_compact_surface_scan.py +++ b/tests/aippocampus/test_compact_surface_scan.py @@ -1,5 +1,6 @@ from __future__ import annotations +import sys import unittest from tools.aippocampus import compact_surface_scan @@ -42,6 +43,68 @@ def test_cli_scan_covers_background_success_and_recovery_paths(self) -> None: compact_surface_scan.CLI_COMMANDS, ) + def test_cli_scan_covers_search_doctor_and_storage_surfaces(self) -> None: + self.assertIn( + 'aippocampus search --all "compact foreground audit" --json --max 5', + compact_surface_scan.CLI_COMMANDS, + ) + self.assertIn("aippocampus doctor provider --json", compact_surface_scan.CLI_COMMANDS) + self.assertIn( + "aippocampus storage gc --dry-run --summary-json --cwd .", + compact_surface_scan.CLI_COMMANDS, + ) + self.assertIn( + "aippocampus storage gc --dry-run --json --top 1 --cwd .", + compact_surface_scan.CLI_COMMANDS, + ) + deep_search = next( + probe + for probe in compact_surface_scan.CLI_PROBES + if "--search-budget deep" in probe.command + ) + self.assertEqual(deep_search.profile, "detail_or_full") + + def test_cli_scan_records_successful_elapsed_time(self) -> None: + check = compact_surface_scan.scan_cli_command( + [ + sys.executable, + "-c", + "import json; print(json.dumps({'ok': True, 'safe_next_actions': []}))", + ], + cwd=compact_surface_scan.PATHS.repo_root, + timeout_seconds=2, + ) + + self.assertTrue(check["ok"], check) + self.assertIn("elapsed_ms", check) + self.assertGreaterEqual(check["elapsed_ms"], 0) + self.assertEqual(check["timeout_seconds"], 2) + + def test_cli_scan_timeout_is_structured_failure(self) -> None: + check = compact_surface_scan.scan_cli_command( + [sys.executable, "-c", "import time; time.sleep(2)"], + cwd=compact_surface_scan.PATHS.repo_root, + timeout_seconds=0.1, + ) + + self.assertFalse(check["ok"], check) + self.assertEqual(check["error"], "timeout") + self.assertIn("time.sleep", check["surface"]) + self.assertIn("elapsed_ms", check) + self.assertLess(check["elapsed_ms"], 1500) + + def test_run_scan_reports_first_slow_surface(self) -> None: + report = compact_surface_scan.run_scan( + cwd=compact_surface_scan.PATHS.repo_root, + include_cli=False, + slow_probe_ms=0, + ) + + self.assertTrue(report["ok"], report) + self.assertIn("elapsed_ms", report) + self.assertGreater(report["checked_count"], 0) + self.assertIsNotNone(report["first_slow_surface"]) + def test_mcp_key_tool_compact_scan_covers_aippo_and_background(self) -> None: checks = compact_surface_scan.scan_mcp_key_tool_compact_cards( cwd=compact_surface_scan.PATHS.repo_root, diff --git a/tests/aippocampus/test_provider_doctor.py b/tests/aippocampus/test_provider_doctor.py index 613bbebc..c18e02c9 100644 --- a/tests/aippocampus/test_provider_doctor.py +++ b/tests/aippocampus/test_provider_doctor.py @@ -304,7 +304,9 @@ def test_cli_doctor_provider_runs_via_public_facade(self) -> None: self.assertNotIn("agent_next_action", payload) self.assertNotIn(payload["foreground_action"], payload["safe_next_actions"]) self.assertEqual(payload["safe_next_actions"][0]["id"], "inspect_provider_doctor_detail") - self.assertEqual(payload["full_audit_command"], "aippocampus doctor provider --detail full --json") + self.assertTrue(payload["details_available"]) + self.assertNotIn("full_audit_command", payload) + self.assertNotIn("operator_detail_command", payload) self.assertNotIn("recommended_actions", payload) self.assertIn("recommended_action_count", payload) self.assertNotIn("provider_env", payload) @@ -312,7 +314,9 @@ def test_cli_doctor_provider_runs_via_public_facade(self) -> None: self.assertNotIn("credential_validation", payload) self.assertNotIn("boundary_detail", payload) self.assertNotIn("cannot_claim", payload) - self.assertIn("boundary_summary", payload) + self.assertNotIn("boundary_summary", payload) + self.assertNotIn("claim_boundary", payload["foreground_action"]) + self.assertNotIn("claim_boundary", payload["safe_next_actions"][0]) self.assertNotIn(fixture_value, proc.stdout) def test_cli_compact_json_missing_key_returns_guidance_card_without_failure(self) -> None: @@ -400,6 +404,9 @@ def test_compact_provider_doctor_card_normalizes_guidance_actions(self) -> None: self.assertEqual(card["recommended_action_ids"], ["set_provider_env_in_hook_environment"]) self.assertNotIn("provider_env", card) self.assertNotIn("legacy_aliases", card) + self.assertTrue(card["details_available"]) + self.assertNotIn("operator_detail_command", card) + self.assertNotIn("claim_boundary", card["foreground_action"]) def test_explicit_dotenv_discovery_reports_candidate_without_secret_or_path_by_default(self) -> None: with tempfile.TemporaryDirectory() as tmp, provider_env(): diff --git a/tests/aippocampus/test_run_tests_tiers.py b/tests/aippocampus/test_run_tests_tiers.py index df956718..98f0cfe1 100644 --- a/tests/aippocampus/test_run_tests_tiers.py +++ b/tests/aippocampus/test_run_tests_tiers.py @@ -259,13 +259,15 @@ def test_tier_report_exposes_module_counts_test_counts_and_top_contributors(self ], ) self.assertEqual(report["tier_aliases"], {}) - self.assertEqual(report["tiers"]["quick"]["budget"]["module_count_target"], 46) + self.assertEqual(report["tiers"]["quick"]["budget"]["module_count_target"], 50) self.assertEqual(report["tiers"]["quick"]["budget"]["test_count_target"], 330) self.assertEqual(report["tiers"]["quick"]["budget"]["module_count_status"], "within_target") self.assertEqual(report["tiers"]["quick"]["budget"]["test_count_status"], "within_target") - self.assertEqual(report["tiers"]["pr"]["budget"]["module_count_target"], 70) + self.assertEqual(report["tiers"]["pr"]["budget"]["module_count_target"], 86) self.assertEqual(report["tiers"]["pr"]["budget"]["test_count_target"], 500) self.assertEqual(report["tiers"]["pr"]["budget"]["budget_outcome"], "within_target") + self.assertEqual(report["tiers"]["pr"]["budget"]["target_updated_at"], "2026-07-08") + self.assertIn("replacement lanes", report["tiers"]["pr"]["budget"]["replacement_lane_note"]) self.assertIsNone(report["tiers"]["pr"]["budget"]["recommended_action"]) self.assertTrue( any( @@ -277,7 +279,7 @@ def test_tier_report_exposes_module_counts_test_counts_and_top_contributors(self def test_count_budget_surfaces_actionable_drift_without_hard_failing(self) -> None: budget = run_tests.count_budget_for_tier( "pr", - module_count=71, + module_count=87, test_count=501, ) @@ -295,8 +297,8 @@ def test_count_budget_surfaces_actionable_drift_without_hard_failing(self) -> No def test_broad_suite_growth_review_is_non_gating_telemetry(self) -> None: review = run_tests.suite_growth_review_for_tier( "broad-pr", - module_count=331, - test_count=2957, + module_count=381, + test_count=3401, top_modules=[ {"module": "tests.aippocampus.test_docs_health", "test_count": 85} ], @@ -307,6 +309,8 @@ def test_broad_suite_growth_review_is_non_gating_telemetry(self) -> None: self.assertEqual(review["kind"], "non_gating_suite_growth_review") self.assertEqual(review["status"], "review_recommended") self.assertEqual(review["recommended_action"]["mutation_risk"], "planning_only") + self.assertEqual(review["review_decision_date"], "2026-07-08") + self.assertIn("CI/pre-merge", review["review_rationale"]) self.assertIn("must not make broad coverage", review["recommended_action"]["why"]) self.assertIn("Non-gating", review["boundary"]) self.assertIsNone( @@ -403,11 +407,11 @@ def test_compact_tier_report_surfaces_budget_warning_action(self) -> None: "generated_at": "2026-06-29T15:00:00Z", "tiers": { "pr": { - "module_count": 71, + "module_count": 87, "test_count": 501, "budget": run_tests.count_budget_for_tier( "pr", - module_count=71, + module_count=87, test_count=501, ), "growth_review": None, diff --git a/tests/aippocampus/test_search_clean_source.py b/tests/aippocampus/test_search_clean_source.py index 60b4e0bb..c61db08d 100644 --- a/tests/aippocampus/test_search_clean_source.py +++ b/tests/aippocampus/test_search_clean_source.py @@ -1786,7 +1786,7 @@ def test_search_all_phrase_like_absent_query_suppresses_low_coverage_noise(self) self.assertNotIn("suppressed_low_coverage_matches", payload) self.assertNotIn("source_boundary", payload) self.assertNotIn("privacy", payload) - self.assertEqual(payload["suppression_boundary"], "phrase_like_low_coverage_suppressed") + self.assertNotIn("suppression_boundary", payload) self.assertEqual(payload["foreground_action_contract"], "foreground-action-v2") self.assertNotIn("agent_next_action", payload) diff --git a/tests/aippocampus/test_search_clean_source_foreground_projection.py b/tests/aippocampus/test_search_clean_source_foreground_projection.py index 12812a8f..161dabd9 100644 --- a/tests/aippocampus/test_search_clean_source_foreground_projection.py +++ b/tests/aippocampus/test_search_clean_source_foreground_projection.py @@ -113,5 +113,5 @@ def test_current_thread_search_suppresses_low_coverage_phrase_hits(self) -> None self.assertEqual(payload["foreground_action"]["id"], "refine_or_recall") self.assertNotIn("source_boundary", payload) self.assertNotIn("recovery_actions", payload) - self.assertEqual(payload["suppression_boundary"], "phrase_like_low_coverage_suppressed") + self.assertNotIn("suppression_boundary", payload) self.assertEqual(foreground_action_contract_violations(payload), []) diff --git a/tests/aippocampus/test_storage_governance.py b/tests/aippocampus/test_storage_governance.py index 4038f062..c122340f 100644 --- a/tests/aippocampus/test_storage_governance.py +++ b/tests/aippocampus/test_storage_governance.py @@ -309,7 +309,8 @@ def test_dry_run_json_cli_bounds_candidates_by_top_unless_full_requested(self) - self.assertFalse(payload["privacy"]["raw_session_like_ids_emitted"]) self.assertNotIn("session-one", encoded) self.assertNotIn("session:one", encoded) - self.assertEqual(payload["full_audit_flag"], "--full") + self.assertTrue(payload["details_available"]) + self.assertNotIn("full_audit_flag", payload) def test_dry_run_summary_json_is_foreground_bounded(self) -> None: with patch("sys.stdout", new=StringIO()) as stdout: @@ -351,27 +352,20 @@ def test_dry_run_summary_json_is_foreground_bounded(self) -> None: "aippocampus storage gc --dry-run --json --top 1 --cwd .", ) action_ids = [action["id"] for action in payload["safe_next_actions"]] - self.assertEqual(action_ids[:2], ["apply_rebuildable_after_audit", "stop_without_cleanup"]) + self.assertEqual(action_ids, ["stop_without_cleanup"]) self.assertEqual(payload["foreground_action"]["mutation_risk"], "read_only") - self.assertEqual( - payload["safe_next_actions"][0]["mutation_risk"], - "explicit_local_delete_of_rebuildable_cache", - ) - self.assertTrue(payload["safe_next_actions"][0]["requires_prior_audit"]) - self.assertIn("deterministic_checks", payload["safe_next_actions"][0]) - self.assertIn("rollback_or_rebuild_boundary", payload["safe_next_actions"][0]) - self.assertTrue(payload["safe_next_actions"][1]["continue_without_command"]) - self.assertNotIn("command", payload["safe_next_actions"][1]) + self.assertNotIn("claim_boundary", payload["foreground_action"]) + self.assertTrue(payload["safe_next_actions"][0]["continue_without_command"]) + self.assertNotIn("command", payload["safe_next_actions"][0]) + self.assertNotIn("claim_boundary", payload["safe_next_actions"][0]) self.assertTrue(payload["safe_next_action"]["continue_without_command"]) self.assertEqual( payload["safe_next_action"]["instruction"], "Continue the user's foreground work without running storage cleanup.", ) - self.assertEqual( - payload["comparable_metrics_command"], - "aippocampus storage gc --dry-run --json --top 1 --cwd .", - ) - self.assertEqual(payload["risk_boundary"]["apply_requires_explicit_flag"], True) + self.assertTrue(payload["details_available"]) + self.assertNotIn("comparable_metrics_command", payload) + self.assertNotIn("risk_boundary", payload) self.assertFalse(payload["privacy"]["raw_session_like_ids_emitted"]) self.assertNotIn("session-one", encoded) self.assertNotIn("session:one", encoded) @@ -434,30 +428,19 @@ def test_summary_json_without_existing_reports_computes_bounded_capacity_pressur payload["foreground_action"]["command"], "aippocampus storage gc --dry-run --json --top 1 --cwd .", ) - self.assertIn( - "stop_without_cleanup", + self.assertEqual( [action["id"] for action in payload["safe_next_actions"]], + ["stop_without_cleanup"], ) self.assertNotIn( "apply_rebuildable_after_audit", [action["id"] for action in payload["safe_next_actions"]], ) - retention_action = next( - action - for action in payload["safe_next_actions"] - if action["id"] == "generate_retention_report" - ) - self.assertIn("aippocampus_runtime.ops.retention_report", retention_action["command"]) self.assertTrue(payload["safe_next_action"]["continue_without_command"]) self.assertNotIn("command", payload["safe_next_action"]) - self.assertEqual( - payload["comparable_metrics_command"], - "aippocampus storage gc --dry-run --json --top 1 --cwd .", - ) - self.assertEqual( - payload["operator_audit_command"], - "aippocampus storage gc --dry-run --json --full --cwd .", - ) + self.assertTrue(payload["details_available"]) + self.assertNotIn("comparable_metrics_command", payload) + self.assertNotIn("operator_audit_command", payload) def test_default_json_without_existing_reports_is_bounded_foreground_card(self) -> None: with patch("sys.stdout", new=StringIO()) as stdout: @@ -480,16 +463,19 @@ def test_default_json_without_existing_reports_is_bounded_foreground_card(self) self.assertEqual(payload["surface_class"], "foreground_storage_gc_plan") self.assertEqual(payload["candidate_count_total"], 1) self.assertEqual(payload["candidates_returned"], 1) - self.assertEqual(payload["full_audit_flag"], "--full") + self.assertTrue(payload["details_available"]) + self.assertNotIn("full_audit_flag", payload) self.assertFalse(payload["privacy"]["raw_session_like_ids_emitted"]) self.assertEqual(payload["foreground_action"]["id"], "bounded_storage_audit") self.assertTrue(payload["foreground_action"]["continue_without_command"]) self.assertNotIn("command", payload["foreground_action"]) + self.assertNotIn("claim_boundary", payload["foreground_action"]) self.assertEqual(payload["safe_next_actions"][0]["id"], "preview_storage_gc_summary") self.assertEqual( payload["safe_next_actions"][0]["command"], "aippocampus storage gc --dry-run --summary-json --cwd .", ) + self.assertNotIn("claim_boundary", payload["safe_next_actions"][0]) def test_dry_run_falls_back_to_capacity_aggregate_when_retention_report_is_missing( self, @@ -628,16 +614,12 @@ def test_summary_top_budget_does_not_hide_old_generation_cleanup_candidates(self payload["metrics"]["reclaimable_rebuildable_bytes"], 13, ) - self.assertIn( + self.assertTrue(payload["details_available"]) + self.assertEqual(payload["foreground_action"]["id"], "bounded_storage_audit") + self.assertNotIn( "apply_rebuildable_after_audit", [item["id"] for item in payload["safe_next_actions"]], ) - apply_action = next( - item - for item in payload["safe_next_actions"] - if item["id"] == "apply_rebuildable_after_audit" - ) - self.assertIn("--include-active", apply_action["command"]) def test_capacity_fallback_cli_redacts_session_like_ids_by_default(self) -> None: with patch("sys.stdout", new=StringIO()) as stdout: diff --git a/tests/aippocampus/test_update_sync.py b/tests/aippocampus/test_update_sync.py index a2c59fa1..b9ade564 100644 --- a/tests/aippocampus/test_update_sync.py +++ b/tests/aippocampus/test_update_sync.py @@ -1074,9 +1074,17 @@ def test_status_agent_json_splits_action_hint_cache_readiness(self) -> None: item.get("surface") for item in payload.get("safe_next_actions", []) } self.assertNotIn("action_hints", action_surfaces) - self.assertEqual( + action_commands = { + payload["foreground_action"].get("command"), + *(item.get("command") for item in payload.get("safe_next_actions", [])), + } + self.assertIn("aippocampus doctor provider --json", action_commands) + self.assertIn( payload["ambient_recall"]["next_command"], - "aippocampus doctor provider --json", + { + "aippocampus doctor provider --json", + "aippocampus warm status --json", + }, ) self.assertNotIn("refresh-cache --write", raw) self.assertNotIn("action_hints", payload) diff --git a/tools/aippocampus/compact_surface_scan.py b/tools/aippocampus/compact_surface_scan.py index d983d04d..55eb7a59 100644 --- a/tools/aippocampus/compact_surface_scan.py +++ b/tools/aippocampus/compact_surface_scan.py @@ -12,9 +12,14 @@ import argparse import json +import os +import shlex +import signal import subprocess from collections.abc import Mapping, Sequence +from dataclasses import dataclass from pathlib import Path +from time import perf_counter from typing import Any try: @@ -26,19 +31,42 @@ SCHEMA_VERSION = 1 -CLI_COMMANDS = ( - "aippocampus hooks prompt status --last --json", - "aippocampus hooks lifecycle status --json", - "aippocampus hooks action status --json", - "aippocampus hooks claude-code status --json", - "aippocampus agent background --json", - 'aippocampus agent background "compact foreground audit" --json', - 'aippocampus agent aippo --task "AIppocampus host probe schema freshness" --json', - "aippocampus warm status --json", - "aippocampus dream status --json", - "aippocampus update status --json", +DEFAULT_PROBE_TIMEOUT_SECONDS = 10.0 +DEFAULT_SCAN_BUDGET_SECONDS = 30.0 +DEFAULT_SLOW_PROBE_MS = 3000.0 + + +@dataclass(frozen=True) +class SurfaceProbe: + command: str + profile: str = "foreground_compact" + timeout_seconds: float = DEFAULT_PROBE_TIMEOUT_SECONDS + + +CLI_PROBES = ( + SurfaceProbe("aippocampus hooks prompt status --last --json"), + SurfaceProbe("aippocampus hooks lifecycle status --json"), + SurfaceProbe("aippocampus hooks action status --json"), + SurfaceProbe("aippocampus hooks claude-code status --json"), + SurfaceProbe("aippocampus agent background --json"), + SurfaceProbe('aippocampus agent background "compact foreground audit" --json'), + SurfaceProbe('aippocampus agent aippo --task "AIppocampus host probe schema freshness" --json'), + SurfaceProbe("aippocampus warm status --json"), + SurfaceProbe("aippocampus dream status --json"), + SurfaceProbe("aippocampus update status --json"), + SurfaceProbe('aippocampus search --all "compact foreground audit" --json --max 5'), + SurfaceProbe( + 'aippocampus search --all "compact foreground audit" ' + "--search-budget deep --json --max-elapsed-ms 15000 --max 5", + profile="detail_or_full", + ), + SurfaceProbe("aippocampus doctor provider --json"), + SurfaceProbe("aippocampus storage gc --dry-run --summary-json --cwd ."), + SurfaceProbe("aippocampus storage gc --dry-run --json --top 1 --cwd ."), ) +CLI_COMMANDS = tuple(probe.command for probe in CLI_PROBES) + MAX_SAFE_ACTIONS = 1 @@ -105,33 +133,152 @@ def _load_json(stdout: str) -> Any: return json.loads(text) -def scan_cli_command(command: str, *, cwd: Path) -> dict[str, Any]: - proc = subprocess.run( - command, - cwd=str(cwd), - shell=True, - text=True, - capture_output=True, - timeout=60, - ) +def _command_surface(command: str | Sequence[str]) -> str: + if isinstance(command, str): + return command + return shlex.join([str(item) for item in command]) + + +def _command_argv(command: str | Sequence[str]) -> list[str]: + if isinstance(command, str): + return shlex.split(command) + return [str(item) for item in command] + + +def _terminate_process_tree(proc: subprocess.Popen[str]) -> None: + if proc.poll() is not None: + return + if os.name == "nt": + subprocess.run( + ["taskkill", "/F", "/T", "/PID", str(proc.pid)], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=False, + ) + return + try: + os.killpg(proc.pid, signal.SIGTERM) + except ProcessLookupError: + return + try: + proc.wait(timeout=1) + return + except subprocess.TimeoutExpired: + pass try: - payload = _load_json(proc.stdout) + os.killpg(proc.pid, signal.SIGKILL) + except ProcessLookupError: + return + + +def _popen_kwargs() -> dict[str, Any]: + if os.name == "nt": + return {"creationflags": getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)} + return {"start_new_session": True} + + +def scan_cli_command( + command: str | Sequence[str], + *, + cwd: Path, + timeout_seconds: float = DEFAULT_PROBE_TIMEOUT_SECONDS, + profile: str = "foreground_compact", +) -> dict[str, Any]: + surface = _command_surface(command) + argv = _command_argv(command) + started_at = perf_counter() + if timeout_seconds <= 0: + return { + "surface": surface, + "kind": "cli", + "profile": profile, + "ok": False, + "exit_code": None, + "elapsed_ms": 0.0, + "timeout_seconds": timeout_seconds, + "error": "scan_budget_exhausted", + "stderr_preview": "", + } + try: + proc = subprocess.Popen( + argv, + cwd=str(cwd), + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + **_popen_kwargs(), + ) + except OSError as exc: + elapsed_ms = round((perf_counter() - started_at) * 1000, 3) + return { + "surface": surface, + "kind": "cli", + "profile": profile, + "ok": False, + "exit_code": None, + "elapsed_ms": elapsed_ms, + "timeout_seconds": timeout_seconds, + "error": f"process_start_failed:{type(exc).__name__}", + "stderr_preview": str(exc)[:240], + } + try: + stdout, stderr = proc.communicate(timeout=timeout_seconds) + except subprocess.TimeoutExpired: + _terminate_process_tree(proc) + try: + stdout, stderr = proc.communicate(timeout=1) + except subprocess.TimeoutExpired: + stdout, stderr = "", "" + elapsed_ms = round((perf_counter() - started_at) * 1000, 3) + return { + "surface": surface, + "kind": "cli", + "profile": profile, + "ok": False, + "exit_code": None, + "elapsed_ms": elapsed_ms, + "timeout_seconds": timeout_seconds, + "error": "timeout", + "stdout_preview": stdout.strip()[:240], + "stderr_preview": stderr.strip()[:240], + } + elapsed_ms = round((perf_counter() - started_at) * 1000, 3) + try: + payload = _load_json(stdout) except Exception as exc: return { - "surface": command, + "surface": surface, "kind": "cli", + "profile": profile, "ok": False, "exit_code": proc.returncode, + "elapsed_ms": elapsed_ms, + "timeout_seconds": timeout_seconds, "error": f"json_parse_failed:{type(exc).__name__}", - "stderr_preview": proc.stderr.strip()[:240], + "stderr_preview": stderr.strip()[:240], } leaks = denied_field_paths(payload) safe_count = safe_action_count(payload) + if profile == "detail_or_full": + return { + "surface": surface, + "kind": "cli", + "profile": profile, + "ok": True, + "exit_code": proc.returncode, + "elapsed_ms": elapsed_ms, + "timeout_seconds": timeout_seconds, + "diagnostic_field_paths": leaks, + "safe_action_count": safe_count, + } return { - "surface": command, + "surface": surface, "kind": "cli", + "profile": profile, "ok": not leaks and safe_count <= MAX_SAFE_ACTIONS, "exit_code": proc.returncode, + "elapsed_ms": elapsed_ms, + "timeout_seconds": timeout_seconds, "denied_field_paths": leaks, "safe_action_count": safe_count, } @@ -143,6 +290,7 @@ def scan_mcp_runtime_recovery() -> dict[str, Any]: foreground_mcp_runtime_recovery_payload, ) + started_at = perf_counter() payload = foreground_mcp_runtime_recovery_payload( "agent_recall", TypeError("agent_recall() got an unexpected keyword argument 'cue'"), @@ -153,13 +301,16 @@ def scan_mcp_runtime_recovery() -> dict[str, Any]: is_error=True, full_output_boundary="foreground_mcp_runtime_recovery", ) + elapsed_ms = round((perf_counter() - started_at) * 1000, 3) structured = result.get("structuredContent") if isinstance(result, dict) else None leaks = denied_field_paths(structured) safe_count = safe_action_count(structured) return { "surface": "mcp_runtime_recovery:agent_recall", "kind": "mcp", + "profile": "foreground_compact", "ok": isinstance(structured, Mapping) and not leaks and safe_count <= MAX_SAFE_ACTIONS, + "elapsed_ms": elapsed_ms, "has_structured_content": isinstance(structured, Mapping), "text_item_count": len(result.get("content") or []) if isinstance(result, dict) else 0, "denied_field_paths": leaks, @@ -167,14 +318,21 @@ def scan_mcp_runtime_recovery() -> dict[str, Any]: } -def _mcp_tool_structured_content_check(surface: str, result: dict[str, Any]) -> dict[str, Any]: +def _mcp_tool_structured_content_check( + surface: str, + result: dict[str, Any], + *, + elapsed_ms: float, +) -> dict[str, Any]: structured = result.get("structuredContent") if isinstance(result, dict) else None leaks = denied_field_paths(structured) safe_count = safe_action_count(structured) return { "surface": surface, "kind": "mcp", + "profile": "foreground_compact", "ok": isinstance(structured, Mapping) and not leaks and safe_count <= MAX_SAFE_ACTIONS, + "elapsed_ms": elapsed_ms, "has_structured_content": isinstance(structured, Mapping), "text_item_count": len(result.get("content") or []) if isinstance(result, dict) else 0, "denied_field_paths": leaks, @@ -194,19 +352,20 @@ def scan_mcp_key_tool_compact_cards(*, cwd: Path) -> list[dict[str, Any]]: from aippocampus_runtime.mcp import tool_handlers - return [ - _mcp_tool_structured_content_check( + checks: list[dict[str, Any]] = [] + for surface, call in ( + ( "mcp_tool:agent_aippo", - tool_handlers.call_agent_aippo( + lambda: tool_handlers.call_agent_aippo( { "task": "AIppocampus host probe schema freshness", "cwd": str(cwd), } ), ), - _mcp_tool_structured_content_check( + ( "mcp_tool:agent_background", - tool_handlers.call_agent_background( + lambda: tool_handlers.call_agent_background( { "cue": "AIppocampus reviewed background findings smoke", "limit": 1, @@ -214,23 +373,81 @@ def scan_mcp_key_tool_compact_cards(*, cwd: Path) -> list[dict[str, Any]]: } ), ), - _mcp_tool_structured_content_check( + ( "mcp_tool:agent_background_missing_input", - tool_handlers.call_agent_background({}), + lambda: tool_handlers.call_agent_background({}), ), - ] - - -def run_scan(*, cwd: Path, include_cli: bool = True) -> dict[str, Any]: + ): + started_at = perf_counter() + result = call() + elapsed_ms = round((perf_counter() - started_at) * 1000, 3) + checks.append(_mcp_tool_structured_content_check(surface, result, elapsed_ms=elapsed_ms)) + return checks + + +def _first_slow_check( + checks: Sequence[Mapping[str, Any]], + *, + slow_probe_ms: float, +) -> Mapping[str, Any] | None: + for check in checks: + if check.get("error") == "timeout": + return check + elapsed = check.get("elapsed_ms") + if isinstance(elapsed, int | float) and elapsed >= slow_probe_ms: + return check + return None + + +def run_scan( + *, + cwd: Path, + include_cli: bool = True, + probe_timeout_seconds: float = DEFAULT_PROBE_TIMEOUT_SECONDS, + scan_budget_seconds: float = DEFAULT_SCAN_BUDGET_SECONDS, + slow_probe_ms: float = DEFAULT_SLOW_PROBE_MS, +) -> dict[str, Any]: + scan_started_at = perf_counter() + scan_deadline = scan_started_at + scan_budget_seconds checks: list[dict[str, Any]] = [] if include_cli: - checks.extend(scan_cli_command(command, cwd=cwd) for command in CLI_COMMANDS) + for probe in CLI_PROBES: + remaining_seconds = scan_deadline - perf_counter() + timeout_seconds = min(probe.timeout_seconds, probe_timeout_seconds, remaining_seconds) + checks.append( + scan_cli_command( + probe.command, + cwd=cwd, + timeout_seconds=timeout_seconds, + profile=probe.profile, + ) + ) + if checks[-1].get("error") == "scan_budget_exhausted": + break checks.append(scan_mcp_runtime_recovery()) checks.extend(scan_mcp_key_tool_compact_cards(cwd=cwd)) failures = [item for item in checks if not item.get("ok")] + elapsed_ms = round((perf_counter() - scan_started_at) * 1000, 3) + first_slow = _first_slow_check(checks, slow_probe_ms=slow_probe_ms) return { "schema_version": SCHEMA_VERSION, "ok": not failures, + "elapsed_ms": elapsed_ms, + "scan_budget_seconds": scan_budget_seconds, + "probe_timeout_seconds": probe_timeout_seconds, + "slow_probe_ms": slow_probe_ms, + "first_slow_surface": first_slow.get("surface") if first_slow else None, + "first_slow_elapsed_ms": first_slow.get("elapsed_ms") if first_slow else None, + "slow_probe_count": sum( + 1 + for item in checks + if item.get("error") == "timeout" + or ( + isinstance(item.get("elapsed_ms"), int | float) + and float(item["elapsed_ms"]) >= slow_probe_ms + ) + ), + "timeout_count": sum(1 for item in checks if item.get("error") == "timeout"), "checked_count": len(checks), "failure_count": len(failures), "checks": checks, @@ -245,8 +462,32 @@ def main(argv: list[str] | None = None) -> int: action="store_true", help="Only scan the in-process MCP recovery payload; used by focused unit tests.", ) + parser.add_argument( + "--probe-timeout-seconds", + type=float, + default=DEFAULT_PROBE_TIMEOUT_SECONDS, + help="Per CLI probe timeout; timeouts are reported as failed checks.", + ) + parser.add_argument( + "--scan-budget-seconds", + type=float, + default=DEFAULT_SCAN_BUDGET_SECONDS, + help="Total local scan budget for representative CLI probes.", + ) + parser.add_argument( + "--slow-probe-ms", + type=float, + default=DEFAULT_SLOW_PROBE_MS, + help="Elapsed time threshold for first_slow_surface reporting.", + ) args = parser.parse_args(argv) - report = run_scan(cwd=PATHS.repo_root, include_cli=not args.no_cli) + report = run_scan( + cwd=PATHS.repo_root, + include_cli=not args.no_cli, + probe_timeout_seconds=args.probe_timeout_seconds, + scan_budget_seconds=args.scan_budget_seconds, + slow_probe_ms=args.slow_probe_ms, + ) if args.json_output: print(json.dumps(report, ensure_ascii=False, indent=2)) else: diff --git a/tools/aippocampus/run_tests.py b/tools/aippocampus/run_tests.py index 8436934a..b6fb153f 100644 --- a/tools/aippocampus/run_tests.py +++ b/tools/aippocampus/run_tests.py @@ -52,25 +52,60 @@ RUNNER_VERSION = "aippocampus-run-tests-v2" TIMINGS_SCHEMA_VERSION = 2 QUICK_BUDGET = { - "module_count_target": 46, + "module_count_target": 50, "test_count_target": 330, "elapsed_seconds_target": 30.0, + "target_updated_at": "2026-07-08", + "target_owner": "verification-steward", + "target_rationale": ( + "Quick has grown by small, focused guard modules while staying far under " + "the test-count and timing targets; keep it as the local inner loop " + "instead of moving focused follow-through coverage to broad lanes." + ), + "replacement_lane_note": ( + "Modules that outgrow quick should move to pr, smoke, integration, or a " + "planner-named focused command with an owner note." + ), } PR_BUDGET = { - "module_count_target": 70, + "module_count_target": 86, "test_count_target": 500, "elapsed_seconds_target": 180.0, + "target_updated_at": "2026-07-08", + "target_owner": "verification-steward", + "target_rationale": ( + "PR keeps the canonical local closeout lane for planner, MCP, compact, " + "and recall follow-through guards; current growth is module-shaped, not " + "test-count-shaped, and remains below the PR timing budget." + ), + "replacement_lane_note": ( + "Broad-pr, benchmark-smoke, integration, and focused planner commands " + "remain the replacement lanes for heavier coverage." + ), } BROAD_SUITE_REVIEW_TARGETS: dict[str, BroadSuiteReviewTarget] = { "broad-pr": { - "module_count_review_threshold": 300, - "test_count_review_threshold": 2500, + "module_count_review_threshold": 380, + "test_count_review_threshold": 3400, "label": "Broad PR", + "review_owner": "verification-steward", + "review_decision_date": "2026-07-08", + "review_rationale": ( + "Accepted as a CI/pre-merge breadth lane, not an ordinary local " + "agent ritual; changed-surface planning must still name focused " + "commands before escalating here." + ), }, "full": { - "module_count_review_threshold": 400, - "test_count_review_threshold": 3500, + "module_count_review_threshold": 460, + "test_count_review_threshold": 4200, "label": "Full", + "review_owner": "verification-steward", + "review_decision_date": "2026-07-08", + "review_rationale": ( + "Accepted as the complete catalog/release audit lane after recent " + "coverage growth; it remains deferred from default local closeout." + ), }, } @@ -97,6 +132,9 @@ class BroadSuiteReviewTarget(TypedDict): module_count_review_threshold: int test_count_review_threshold: int label: str + review_owner: str + review_decision_date: str + review_rationale: str def _target_status(actual: int | float, target: int | float) -> str: @@ -159,6 +197,10 @@ def count_budget_for_tier( ), "note": note, "tier_label": label, + "target_updated_at": budget["target_updated_at"], + "target_owner": budget["target_owner"], + "target_rationale": budget["target_rationale"], + "replacement_lane_note": budget["replacement_lane_note"], } @@ -186,6 +228,9 @@ def suite_growth_review_for_tier( "module_count_status": _target_status(module_count, module_threshold), "test_count_status": _target_status(test_count, test_threshold), "top_contributors": top_modules[:5], + "review_owner": target["review_owner"], + "review_decision_date": target["review_decision_date"], + "review_rationale": target["review_rationale"], "recommended_action": ( { "id": f"review_{normalized_tier.replace('-', '_')}_suite_growth", @@ -726,6 +771,10 @@ def _compact_budget(budget: object) -> dict[str, object] | None: "module_count_status": budget.get("module_count_status"), "test_count_status": budget.get("test_count_status"), "budget_outcome": budget.get("budget_outcome"), + "target_updated_at": budget.get("target_updated_at"), + "target_owner": budget.get("target_owner"), + "target_rationale": budget.get("target_rationale"), + "replacement_lane_note": budget.get("replacement_lane_note"), } return {key: value for key, value in row.items() if value not in (None, "", [], {})} @@ -737,6 +786,9 @@ def _compact_growth_review(review: object) -> dict[str, object] | None: "status": review.get("status"), "module_count_review_threshold": review.get("module_count_review_threshold"), "test_count_review_threshold": review.get("test_count_review_threshold"), + "review_owner": review.get("review_owner"), + "review_decision_date": review.get("review_decision_date"), + "review_rationale": review.get("review_rationale"), } return {key: value for key, value in row.items() if value not in (None, "", [], {})} From 14d4db51f8fdc19b5e70aa8f06c7a0ed09e7c434 Mon Sep 17 00:00:00 2001 From: Sapientropic Date: Wed, 8 Jul 2026 16:24:27 +0800 Subject: [PATCH 2/4] Fix CI closeout regressions --- .../architecture-debt-register.md | 11 ++-- .../aippocampus_runtime/cli/chooser_accept.py | 12 ++--- .../aippocampus_runtime/cli/recovery.py | 8 +-- .../aippocampus_runtime/macro/perturbation.py | 4 +- .../macro/total_encoder.py | 27 ++++++---- .../recall/agent_continuity.py | 9 +++- .../recall/registry_source_apw_candidates.py | 2 +- .../registry/reachability_audit.py | 16 +++--- .../source/current_source_window.py | 5 +- .../update/plugin_public_summary.py | 52 +++++++++++-------- tests/aippocampus/test_cli_recovery_cards.py | 8 ++- tests/aippocampus/test_frontier_probe.py | 6 ++- tests/aippocampus/test_plugin_installer.py | 7 ++- tools/aippocampus/run_tests.py | 34 ++++++++---- 14 files changed, 125 insertions(+), 76 deletions(-) diff --git a/docs/architecture/architecture-debt-register.md b/docs/architecture/architecture-debt-register.md index 28237b95..5bed0b0c 100644 --- a/docs/architecture/architecture-debt-register.md +++ b/docs/architecture/architecture-debt-register.md @@ -93,7 +93,7 @@ this main action queue. | `skills/aippocampus/scripts/aippocampus_runtime/hooks/claude_code.py` | 810 | 840 | P1 claude-hook-host-risk | Split settings mutation helpers, event-status projection, or handler-command resolution before adding another Claude Code event family or host-version lane. | #2010 keeps status, dry-run, explicit install/uninstall, synthetic smoke, and fail-open scoped handlers together because they guard the same Claude settings contract. PostToolUse/PostCompact support or real-host firing evidence should extract a focused owner instead of growing this file. | | `skills/aippocampus/scripts/aippocampus_runtime/mcp/recall_navigation.py` | 914 | 1040 | P1 agent-recall-risk | Split source-window projection or malformed-handle diagnostics before adding another route family, handle family, or freshness policy. | Fixed-vocabulary route-topic labels now live in `source/route_topics.py` so MCP and current clean-source search share one navigation-only topic owner. #1278 and #1188 still own display-vs-callable deepen diagnostics in the progressive MCP recall owner because they directly shape the same `recall_context -> recall_deepen` packet contract. #2026 kept latest-reply reopen in this owner because it is another handle-resolution branch, not a new recall surface. #2541-#2542 moved strict source-ref matching and public source-window projection into `mcp/source_ref_matching.py`; keep future reopen matching changes there instead of spending this recovered headroom. | | `skills/aippocampus/scripts/aippocampus_runtime/mcp/memory_health_recovery.py` | 350 | 380 | P1 memory-health-foreground-risk | Split recall-capability probing from public recovery-card projection before adding another health, maintenance, or host-readiness branch. | #2129 keeps exception recovery and compact health-card recall-first overlay together because both prevent MCP `memory_health` from hiding a source-backed `agent_recall -> agent_deepen` path. Do not add maintenance planning, indexing policy, or host install status here; keep those in their existing owners and only project the foreground split state. | -| `skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py` | 528 | 700 | P1 recall-projection-risk | Split source-open/exact-search action selection before adding another compact recall action family, posture field family, or low-confidence recovery lane. | #2679 moved compact route receipts to `mcp/agent_recall_route_projection.py` and final compact assembly/debug stripping to `mcp/agent_recall_result_assembly.py`. #2923 moved APW fallback/policy compaction and recovery-action promotion to `mcp/agent_recall_recovery_projection.py`, leaving this file as orchestration for one foreground action. Keep proof/follow-through material in deepen/open, detail, tests, or PR notes rather than default compact output. | +| `skills/aippocampus/scripts/aippocampus_runtime/mcp/agent_recall_projection.py` | 580 | 700 | P1 recall-projection-risk | Split source-open/exact-search action selection before adding another compact recall action family, posture field family, or low-confidence recovery lane. | #2679 moved compact route receipts to `mcp/agent_recall_route_projection.py` and final compact assembly/debug stripping to `mcp/agent_recall_result_assembly.py`. #2923 moved APW fallback/policy compaction and recovery-action promotion to `mcp/agent_recall_recovery_projection.py`, leaving this file as orchestration for one foreground action. Keep proof/follow-through material in deepen/open, detail, tests, or PR notes rather than default compact output. | | `skills/aippocampus/scripts/aippocampus_runtime/mcp/tool_handlers.py` | 1074 | 1120 | P1 mcp-tool-handler-risk | Split recall, source-search, health, or Telepathy handler families before adding another MCP tool family or write-like setup lane. | #2530 moved tool argument parsing, handler execution, and text/error result shaping out of `mcp/server.py` so the server stays JSON-RPC/stdio glue. This helper is the new MCP tool-family owner; do not let it become the next host facade by adding unrelated protocol behavior here. | | `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity_cli.py` | 293 | 780 | P1 agent-continuity-cli-risk | Split selector/cache error projection or subcommand-family dispatch before adding another CLI recovery branch, help-card family, or command mode. | #2530 moved parser/help/command dispatch out of `agent_continuity.py`; #2676 keeps corrupt last-recall error handling here as thin dispatch only, delegating compact/full projection to `mcp/agent_deepen_projection.py` and recovery payload shaping to `agent_continuity_cli_support.py`. This row is freeze-except-delete/split for future behavior growth. | | `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_recall_cache.py` | 645 | 680 | P1 last-recall-cache-risk | Split selector-cache persistence, cache diagnostic shaping, or handle materialization before adding another last-recall route family or foreground recovery lane. | #2676 keeps same-machine last-recall selectors, handle snapshots, and request lookup here while shared cache-read status lives in `recall/cache_read_diagnostics.py`; compact/full recovery projection belongs in `agent_continuity_cli_support.py` and `mcp/agent_deepen_projection.py`. Do not reintroduce broad cache-read fallbacks or compact proof fields here. | @@ -145,7 +145,7 @@ of letting the action queue become an archive. | `skills/aippocampus/scripts/aippocampus_runtime/recall/ambient_cache.py` | 666 | 820 | #2676 | Ambient thread-card cache persistence, related-cache matching, and residue export only; topic signal accumulation and its typed read diagnostics now live in `recall/ambient_signal_accumulator.py`. | | `skills/aippocampus/scripts/aippocampus_runtime/mcp/server.py` | 304 | 320 | #2530 | JSON-RPC/stdio handshake, initialization, request validation, and runtime-recovery wrapping only; tool argument parsing and execution now live in `mcp/tool_handlers.py`. | | `skills/aippocampus/scripts/aippocampus_runtime/ops/maintenance.py` | 753 | 780 | #2530 | Maintenance CLI orchestration, consent/apply dispatch, and health preflight; plan/status/action-card projection now lives in `ops/maintenance_projection.py`. | -| `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py` | 797 | 1500 | #2530/#2631 | Source-backed deepen/explain behavior, AIppo/macro packet building, low-authority feedback capture, and source reopen semantics; parser/help/command dispatch lives in `recall/agent_continuity_cli.py`, and the recall hot path now delegates load/evaluate/recovery/render stages to `recall/agent_recall_pipeline.py`. | +| `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py` | 804 | 1500 | #2530/#2631 | Source-backed deepen/explain behavior, AIppo/macro packet building, low-authority feedback capture, and source reopen semantics; parser/help/command dispatch lives in `recall/agent_continuity_cli.py`, and the recall hot path now delegates load/evaluate/recovery/render stages to `recall/agent_recall_pipeline.py`. | | `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_recall_pipeline.py` | 635 | 640 | #2631 / #2951 parked | Agent recall hot-path stages: resolve inputs, load/bias routes, build memory packets/deepen requests, attach source gates, build weak-route recovery, select foreground action, and assemble the full payload. The 2026-06-28 recall closeout kept action selection here only to align CLI/MCP follow-through with the shared foreground-action card; #2631 is history-only, and #2951 parks this row as freeze-except-delete/split until scan/gate/action selection moves before another recall route family or compact recovery lane. | | `skills/aippocampus/scripts/aippocampus_runtime/recall/agent_recall_primitives.py` | 389 | 520 | #2631 | Shared agent-recall primitives used by the thin facade and staged recall pipeline: route-to-packet shaping, compact byte helpers, already-opened route markers, clean-source checks, and macro-state projection loading. Keep this source-free of CLI/deepen/render orchestration so it cannot become another facade/pipeline cycle. | | `skills/aippocampus/scripts/aippocampus_runtime/source/last_recall_search.py` | 64 | 640 | #2631 | Facade only: human rendering and CLI exit policy for last-recall source search. Selector/cache/source scanning and result projection live in `source/last_recall_search_pipeline.py`; do not reintroduce search or recovery branching here. | @@ -159,7 +159,7 @@ of letting the action queue become an archive. ## Test, Benchmark, And Tool Debt Budgets -Last counted: 2026-06-27. +Last counted: 2026-07-08. Counting method: `script_line_count()` from `tests/aippocampus/test_architecture_boundaries.py`: nonblank lines excluding lines whose first non-space character is `#`. @@ -192,9 +192,9 @@ Current non-runtime action rows: | `tests/aippocampus/test_continuity_domains.py` | 1506 | 1750 | #973 / #926 | Split continuity-domain producer fixtures or route-projection assertions before adding another domain family, source-opening surface, or MCP handle contract. | | `tests/aippocampus/test_agent_opt_in_recall_routes.py` | 1617 | 1650 | #2587 / #2605 | Keep opt-in recall route projection, source-anchor gating, foreground action cards, and deepen request contracts together while the recall usefulness gate settles. Split clean-source fixture builders or source-anchor scenario catalogs before adding another foreground-action family or route-ranking gate. | | `tests/aippocampus/test_aippocampus_mcp_server_recall.py` | 1682 | 1700 | #2561 / #2606 / #2609 | Keep MCP recall/deepen JSON-RPC parity, selector-cache handoff, cwd registry resolution, and APW source follow-through together while the foreground MCP contract settles. Split the JSON-RPC harness or registry/source-chain fixture builders before adding another MCP recall route family. | -| `tests/aippocampus/test_agent_recall_compact_projection.py` | 1633 | 1650 | #2799-#2802 / #2824 / #2855 | Keep compact recall foreground-action projection, safe-action budget enforcement, route actionability labels, and source-opening follow-through together while the compact/detail boundary settles. Split representative route fixtures or source-window scenario builders before adding another route-family actionability case. | +| `tests/aippocampus/test_agent_recall_compact_projection.py` | 1680 | 1700 | #2799-#2802 / #2824 / #2855 / #2969 | Keep compact recall foreground-action projection, safe-action budget enforcement, route actionability labels, and source-opening follow-through together while the compact/detail boundary settles. #2969 added the route-note-only blind-deepen regression here because it exercises the same compact projection actionability boundary. Split representative route fixtures or source-window scenario builders before adding another route-family actionability case. | | `tests/aippocampus/test_search_clean_source.py` | 1781 | 1800 | #2611 | Keep exact/current clean-source search, artifact demotion, and canonical-origin route regressions together while source-anchor follow-through settles. Split real-history cue catalogs or source-artifact fixture builders before adding another search/ranking family. | -| `tests/aippocampus/test_update_sync.py` | 2249 | 2290 | #1307 / #1335 / #1431 / #1563 / #1748 / #1755 / #1810 / #1982 / #2078 / #2183 / #2203-#2226 | Codex config/cache, plugin-package, hook-duplicate, and dirty-worktree fixture builders now live in `tests/aippocampus/update_sync_fixtures.py`. Split plugin apply/status assertion helpers or rollback result fixtures before adding another host install/sync lane; keep one local-update contract suite until marketplace/cache/rollback/no-write status behavior stabilizes. | +| `tests/aippocampus/test_update_sync.py` | 2257 | 2290 | #1307 / #1335 / #1431 / #1563 / #1748 / #1755 / #1810 / #1982 / #2078 / #2183 / #2203-#2226 | Codex config/cache, plugin-package, hook-duplicate, and dirty-worktree fixture builders now live in `tests/aippocampus/update_sync_fixtures.py`. Split plugin apply/status assertion helpers or rollback result fixtures before adding another host install/sync lane; keep one local-update contract suite until marketplace/cache/rollback/no-write status behavior stabilizes. | | `tests/aippocampus/test_aippocampus_cli.py` | 1748 | 1800 | #1756 / #1843 / #1838 / #1845 / #1774-#1823 / #2066-#2078 / #1986-#1996 / #2014-#2015 / #2173 / #2183 / #2203-#2226 / #2266 / #2337 | Split subprocess runner helpers or command-family smoke tests into focused modules before adding more top-level facade, personal-control, package-facade, or MCP command discovery cases; keep the first-five-minutes path visible in one smoke owner. Command-family recovery/help contracts live in `tests/aippocampus/test_cli_recovery_cards.py`; #2266 kept config doctor shortcut smoke here because the public facade owns `aippocampus config`; #2337 kept compact/full config doctor split coverage here for the same public `aippocampus config` alias. Behavioral-record cleanup moved to `tests/aippocampus/test_learning_loop_behavioral_records.py`; #2535 lowered the broad allowance after the cleanup wave so drift does not become background noise. | | `tests/aippocampus/test_aippocampus_health.py` | 1628 | 1960 | #248 / #1809 / #1782 / #1890 / #2081 / #2178 | Split storage-pressure fixture builders, host-state fixture builders, exit-code policy fixtures, or compact/full projection assertion helpers before adding another health lane; keep generated-cache pressure advisory, host-state confound separation, compact agent cards, first-recall-blocked exit semantics, and live-delta readiness together until that contract stabilizes. #2400 full-detail timing/deferred-expensive cases now live in `tests/aippocampus/test_health_operator_diagnostics.py` instead of expanding this broad health contract. | | `tests/aippocampus/test_docs_health.py` | 1541 | 1700 | #672 / #1837 / #1845 / #1846 / #1887 / #2079 / #2087 / #2276 / #2277 | Shared docs-health fixture writers now live in `tests/aippocampus/docs_health_fixtures.py`. Split public-doc command lint, executable memory-card examples, bilingual surface guards, evidence-ledger guards, diagnostic-prerequisite guards, or product-profile guards into focused helpers before adding another docs-health domain; keep one repository health smoke owner for cross-doc drift. #2276/#2277 kept first-useful-recall and bilingual IA guards here because they catch public first-use documentation drift across docs, llms.txt, and the canonical origin essay. #2535 lowered the stale broad allowance after the cleanup wave. | @@ -207,6 +207,7 @@ Current non-runtime action rows: | `benchmarks/aippocampus/benchmark_continuous_memory_arms.py` | 1954 | 2000 | #378 / #1968 | Preregistered repeat readout and registration projection live in `continuous_memory_preregistered_slices.py`; #1968 added the public context-loss cohort in the same comparative-arm runner so no extra scoring layer was introduced. Split arm fixture catalog, context-loss cohort construction, or cost/harm scoring before adding more arms or private-history adapters. | | `benchmarks/aippocampus/benchmark_hippocampal_hard_negatives.py` | 1364 | 1400 | #517 / #1972 | Split currentness/supersession fixture catalog or source-open scoring projection before adding another hard-negative family, LoCoMo adapter, or answer-time currentness claim. | | `tests/aippocampus/test_warm_ambient_recall.py` | 2086 | 2300 | #657 / #1281 / #2009 | Clean-thread and registry fixture writers now live in `tests/aippocampus/warm_ambient_fixtures.py`. Keep warm ambient hook, registry reconciliation, cache, scout, detached-job, and optional queue-status tests together while stale queue recovery stays part of the scheduler contract. Split hook-seen ledger fixtures, queue-status fixtures, or scout-card assertion helpers before adding another scheduler/lifecycle lane. | +| `tools/aippocampus/run_tests.py` | 1151 | 1200 | #2971 / #2972 | Keep tier planning, count/timing budget projection, report JSON, and shard/timing artifact handling together while quick/pr and broad/full ownership is being re-baselined. Split budget report projection or manifest/timing artifact helpers before adding another lane-family policy. | | `tools/aippocampus/agent_slop_guard.py` | 1217 | 1260 | #2689 / #2636 | Keep this CLI as the changed-surface anti-workaround guard over central owner contracts. Split rule-family scanners, baseline inventory loading, or fixture/report projection before adding another owner-layer contract family. | | `tools/aippocampus/docs/check_docs_health.py` | 1436 | 1460 | #672 / #1887 / #2203-#2226 | Product profile guards now live in `tools/aippocampus/docs/product_profile_guard.py`; foreground continuity/report-router/currentness guards now live in `tools/aippocampus/docs/foreground_continuity_guard.py`; evidence-ledger payload line checks and diagnostic-prerequisite wording guards stayed here because they are repository docs-health invariants. Split another focused check group or shared markdown/path scanner before adding more public-readiness domains; keep single CLI output stable. | | `tools/aippocampus/docs/debt_report.py` | 1153 | 1180 | #2628 / #2629 / #2636 / #2651 / #2678 | Keep this CLI as the changed-surface acceptance gate, but split helper-duplication, exception, debug-literal, instruction-surface, or guard-pressure scanners into focused helpers before adding another debt family. #2678 moved guard-pressure owner metadata and touched-pressure detection into `tools/aippocampus/docs/guard_pressure.py`; this row remains freeze-except-delete/split even with recovered headroom. Do not grow it into a second product-readiness report or foreground proof surface. | diff --git a/skills/aippocampus/scripts/aippocampus_runtime/cli/chooser_accept.py b/skills/aippocampus/scripts/aippocampus_runtime/cli/chooser_accept.py index de56176f..a9520a38 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/cli/chooser_accept.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/cli/chooser_accept.py @@ -67,12 +67,12 @@ def chooser_tail_supported(tail: Sequence[str]) -> bool: def _action_candidates(payload: Mapping[str, Any]) -> list[dict[str, Any]]: candidates: list[dict[str, Any]] = [] - for raw in [ - payload.get("foreground_action"), - *(payload.get("safe_next_actions") or []), - *(payload.get("choices") if isinstance(payload.get("choices"), list) else []), - *(payload.get("write_actions") if isinstance(payload.get("write_actions"), list) else []), - ]: + raw_candidates: list[Any] = [payload.get("foreground_action")] + for key in ("safe_next_actions", "choices", "write_actions"): + value = payload.get(key) + if isinstance(value, list): + raw_candidates.extend(value) + for raw in raw_candidates: if not isinstance(raw, Mapping): continue action = normalize_foreground_action(raw) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/cli/recovery.py b/skills/aippocampus/scripts/aippocampus_runtime/cli/recovery.py index 1bc2ba79..2431f25f 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/cli/recovery.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/cli/recovery.py @@ -186,7 +186,7 @@ def module_exception_payload( else [] ) action = search_actions[0] if search_actions else script_recovery_action(script_name, args) - payload = ( + payload: dict[str, Any] = ( { "ok": False, "error": cli_error_object( @@ -231,7 +231,8 @@ def module_exception_payload( def render_module_exception_text(payload: dict[str, Any]) -> str: - error = payload.get("error") if isinstance(payload.get("error"), dict) else {} + raw_error = payload.get("error") + error: Mapping[str, Any] = raw_error if isinstance(raw_error, Mapping) else {} action = payload.get("foreground_action") action_map = action if isinstance(action, Mapping) else {} lines = [ @@ -252,7 +253,8 @@ def handle_module_exception( stderr_text: str = "", ) -> int: payload = module_exception_payload(script_name, args, exc, stderr_text=stderr_text) - error = payload.get("error") if isinstance(payload.get("error"), dict) else {} + raw_error = payload.get("error") + error: Mapping[str, Any] = raw_error if isinstance(raw_error, Mapping) else {} exit_code = cli_exit_code_for_error_code(str(error.get("code") or "runtime_error")) if args_request_json(args): emit_public_text(json.dumps(payload, ensure_ascii=False, indent=2), end="") diff --git a/skills/aippocampus/scripts/aippocampus_runtime/macro/perturbation.py b/skills/aippocampus/scripts/aippocampus_runtime/macro/perturbation.py index 5347cffd..fea01d10 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/macro/perturbation.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/macro/perturbation.py @@ -172,7 +172,9 @@ def build_perturbation_packet( reason_codes.append("inversion_requires_source_reopen_or_conflict_review") if policy.get("stale_conflict_checks_required"): reason_codes.append("stale_conflict_checks_required") - for reason in layer_diagnostics["reason_codes"]: + raw_layer_reasons = layer_diagnostics.get("reason_codes") + layer_reasons = raw_layer_reasons if isinstance(raw_layer_reasons, list) else [] + for reason in layer_reasons: if reason not in reason_codes: reason_codes.append(str(reason)) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/macro/total_encoder.py b/skills/aippocampus/scripts/aippocampus_runtime/macro/total_encoder.py index d9aa5ad3..438360b9 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/macro/total_encoder.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/macro/total_encoder.py @@ -194,13 +194,13 @@ def build_total_hexagram_encoding( if explicit_reviewed_state is None: observed: dict[int, list[dict[str, Any]]] = {line: [] for line in range(1, 7)} for signal in line_signals: - line = _line_index(signal.get("line")) + line_index = _line_index(signal.get("line")) value = _line_value(signal.get("value")) - if line is None or value is None: + if line_index is None or value is None: continue if _private_scope(signal): blocked = True - observed[line].append( + observed[line_index].append( { "value": value, "authority": "blocked", @@ -210,7 +210,7 @@ def build_total_hexagram_encoding( ) continue refs = _safe_refs(signal.get("source_refs")) - observed[line].append( + observed[line_index].append( { "value": value, "authority": "source_backed" if refs else "unknown", @@ -221,8 +221,8 @@ def build_total_hexagram_encoding( for line, rows in observed.items(): if not rows: continue - values = {row["value"] for row in rows} - if len(values) > 1: + row_values = {row["value"] for row in rows} + if len(row_values) > 1: ambiguous = True line_slots[line] = { "line": line, @@ -292,15 +292,24 @@ def build_total_hexagram_encoding( "symbolic_advice_included": False, }, } - if complete: - lines_tuple = tuple(int(value) for value in known_values) # type: ignore[arg-type] + complete_values = [value for value in known_values if value in (0, 1)] + if complete and len(complete_values) == 6: + line_values = tuple(int(value) for value in complete_values) + lines_tuple = ( + line_values[0], + line_values[1], + line_values[2], + line_values[3], + line_values[4], + line_values[5], + ) hexagram = hexagram_from_lines(lines_tuple) hexagram_projection = hexagram.to_public_dict() hexagram_projection["bits_bottom_to_top"] = hexagram_projection["bitstring_bottom_to_top"] result["hexagram"] = hexagram_projection result["macro_state_hint"] = _state_hint( project=project, - lines=lines_tuple, # type: ignore[arg-type] + lines=lines_tuple, changing_lines=changing_lines, active_layer=active_layer, momentum=momentum, diff --git a/skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py b/skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py index 65c48e91..2d8e3325 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/recall/agent_continuity.py @@ -8,6 +8,7 @@ from __future__ import annotations from collections.abc import Callable, Mapping +from datetime import datetime, timezone from pathlib import Path from typing import Any, cast @@ -753,6 +754,7 @@ def explain( def build_macro_orientation_packet_fixture_report() -> dict[str, Any]: + fixture_now = datetime(2026, 6, 12, tzinfo=timezone.utc) moving = macro_state.build_macro_orientation_state( project="AIppocampus", hexagram="乾", @@ -769,10 +771,15 @@ def build_macro_orientation_packet_fixture_report() -> dict[str, Any]: updated_at="2026-06-11T11:00:00Z", active_layer="人", ) - current = macro_state.latest_project_macro_orientation([moving], project="AIppocampus") + current = macro_state.latest_project_macro_orientation( + [moving], + project="AIppocampus", + now=fixture_now, + ) standing_projection = macro_state.latest_project_macro_orientation( [standing], project="AIppocampus", + now=fixture_now, ) packet = _macro_memory_packet(current, project="AIppocampus") standing_packet = _macro_memory_packet(standing_projection, project="AIppocampus") diff --git a/skills/aippocampus/scripts/aippocampus_runtime/recall/registry_source_apw_candidates.py b/skills/aippocampus/scripts/aippocampus_runtime/recall/registry_source_apw_candidates.py index 467feec4..ba55d73a 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/recall/registry_source_apw_candidates.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/recall/registry_source_apw_candidates.py @@ -145,7 +145,7 @@ def _candidate_from_match( "candidate_id": f"registry-clean-source:{public_id}", "thread_key": _compact(thread_key, 120), "route_terms": route_terms, - "query_anchor_terms": unique_preserve(anchor_terms, limit=12), + "query_anchor_terms": unique_preserve(list(anchor_terms), limit=12), "actual_source_matched_terms": route_terms, "meaningful_matched_terms": route_terms, "anchor_quality": anchor_quality, diff --git a/skills/aippocampus/scripts/aippocampus_runtime/registry/reachability_audit.py b/skills/aippocampus/scripts/aippocampus_runtime/registry/reachability_audit.py index 67b968c5..29585c99 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/registry/reachability_audit.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/registry/reachability_audit.py @@ -31,7 +31,8 @@ def _clean_source_messages_path(paths: Mapping[str, Any]) -> str: def _row_state(entry: Mapping[str, Any]) -> dict[str, bool]: - paths = entry.get("paths") if isinstance(entry.get("paths"), Mapping) else {} + raw_paths = entry.get("paths") + paths: Mapping[str, Any] = raw_paths if isinstance(raw_paths, Mapping) else {} provider_known = bool( str(entry.get("source_provider") or "").strip() or (isinstance(entry.get("session_meta"), Mapping) and entry.get("session_meta")) @@ -57,8 +58,10 @@ def _count(rows: list[dict[str, bool]], key: str) -> int: def render_reachability_audit(report: Mapping[str, Any]) -> str: - counts = report.get("counts") if isinstance(report.get("counts"), Mapping) else {} - action = report.get("foreground_action") if isinstance(report.get("foreground_action"), Mapping) else {} + raw_counts = report.get("counts") + counts: Mapping[str, Any] = raw_counts if isinstance(raw_counts, Mapping) else {} + raw_action = report.get("foreground_action") + action: Mapping[str, Any] = raw_action if isinstance(raw_action, Mapping) else {} next_action = action.get("command") or action.get("command_template") lines = [ f"registry source reachability: {report.get('status')}", @@ -162,12 +165,7 @@ def registry_source_reachability_audit( } for entry, state in zip(entries, row_states, strict=True) ] - report.update( - canonical_foreground_action_fields( - actions[0] if actions else None, - safe_next_actions=actions, - ) - ) + report.update(canonical_foreground_action_fields(actions[0], safe_next_actions=actions)) return report if include_paths else redact_private_paths(report) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/source/current_source_window.py b/skills/aippocampus/scripts/aippocampus_runtime/source/current_source_window.py index 9df33bfa..b4cb0058 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/source/current_source_window.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/source/current_source_window.py @@ -201,12 +201,13 @@ def render_current_source_window_text( result: Mapping[str, Any], *, snippet_chars: int ) -> str: if not result.get("ok"): - error = result.get("error") if isinstance(result.get("error"), Mapping) else {} + raw_error = result.get("error") + error_map: Mapping[str, Any] = raw_error if isinstance(raw_error, Mapping) else {} return "\n".join( [ "AIppocampus current-thread source window", f"status: {result.get('status') or 'cannot_verify'}", - f"error: {error.get('code') or 'unknown'}", + f"error: {error_map.get('code') or 'unknown'}", "boundary: rerun search for a fresh source-backed route before quoting.", ] ) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/update/plugin_public_summary.py b/skills/aippocampus/scripts/aippocampus_runtime/update/plugin_public_summary.py index 9de62560..08ea732e 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/update/plugin_public_summary.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/update/plugin_public_summary.py @@ -94,12 +94,19 @@ def _trusted_codex_next_actions(*, ok: bool, action_required: bool) -> list[dict def public_install_summary(result: dict[str, Any]) -> dict[str, Any]: """Return the user-reportable install/probe summary without local paths.""" - plugin = result.get("plugin") if isinstance(result.get("plugin"), dict) else {} - host_probe = result.get("host_probe") if isinstance(result.get("host_probe"), dict) else {} - warning_summary = host_probe.get("warning_summary") if isinstance(host_probe, dict) else {} + raw_plugin = result.get("plugin") + raw_host_probe = result.get("host_probe") + plugin_map: dict[str, Any] = raw_plugin if isinstance(raw_plugin, dict) else {} + host_probe_map: dict[str, Any] = ( + raw_host_probe if isinstance(raw_host_probe, dict) else {} + ) + warning_summary = host_probe_map.get("warning_summary") + warning_summary_map = warning_summary if isinstance(warning_summary, dict) else {} + mcp_status = host_probe_map.get("mcp_status") + mcp_status_map = mcp_status if isinstance(mcp_status, dict) else {} tool_names = [ str(item) - for item in ((host_probe.get("mcp_status") or {}).get("tool_names") or []) + for item in (mcp_status_map.get("tool_names") or []) ] key_tools = [ name @@ -113,19 +120,18 @@ def public_install_summary(result: dict[str, Any]) -> dict[str, Any]: if name in tool_names ] key_tool_smokes = [ - item for item in host_probe.get("key_tool_smokes") or [] if isinstance(item, dict) + item for item in host_probe_map.get("key_tool_smokes") or [] if isinstance(item, dict) ] key_tool_failures = [item for item in key_tool_smokes if not item.get("ok")] ok = bool(result.get("ok")) agent_callable_status = result.get("agent_callable_status") - warning_counts = _warning_summary_counts(warning_summary) - mcp_preflight = ( - result.get("mcp_command_preflight") - if isinstance(result.get("mcp_command_preflight"), dict) - else {} + warning_counts = _warning_summary_counts(warning_summary_map) + raw_mcp_preflight = result.get("mcp_command_preflight") + mcp_preflight_map: dict[str, Any] = ( + raw_mcp_preflight if isinstance(raw_mcp_preflight, dict) else {} ) action_required = _aippocampus_action_required(warning_counts, ok=ok) or _mcp_preflight_action_required( - mcp_preflight + mcp_preflight_map ) return { "kind": "aippocampus_plugin_install_public_summary", @@ -138,21 +144,21 @@ def public_install_summary(result: dict[str, Any]) -> dict[str, Any]: ok=ok, action_required=action_required, agent_callable_status=agent_callable_status, - mcp_preflight=mcp_preflight, + mcp_preflight=mcp_preflight_map, ), "trusted_codex_next_actions": _trusted_codex_next_actions( ok=ok, action_required=action_required, ), "plugin": { - "id": plugin.get("id"), - "version": plugin.get("version"), - "action": plugin.get("action"), - "installed": bool(plugin.get("installed")), - "enabled": bool(plugin.get("enabled")), + "id": plugin_map.get("id"), + "version": plugin_map.get("version"), + "action": plugin_map.get("action"), + "installed": bool(plugin_map.get("installed")), + "enabled": bool(plugin_map.get("enabled")), }, "host_probe": { - "validation_ok": bool(host_probe.get("validation_ok")), + "validation_ok": bool(host_probe_map.get("validation_ok")), "tool_count": len(tool_names), "key_tools_present": key_tools, "key_tools_callable": None @@ -162,11 +168,11 @@ def public_install_summary(result: dict[str, Any]) -> dict[str, Any]: "warning_summary": warning_counts, }, "mcp_command_preflight": { - "status": mcp_preflight.get("status"), - "command": mcp_preflight.get("command"), - "resolves": bool(mcp_preflight.get("resolves")), - "primary_repair_command": mcp_preflight.get("primary_repair_command"), - "repair_options": list(mcp_preflight.get("repair_options") or [])[:3], + "status": mcp_preflight_map.get("status"), + "command": mcp_preflight_map.get("command"), + "resolves": bool(mcp_preflight_map.get("resolves")), + "primary_repair_command": mcp_preflight_map.get("primary_repair_command"), + "repair_options": list(mcp_preflight_map.get("repair_options") or [])[:3], "resolved_path_emitted": False, }, "current_thread_mcp_boundary": { diff --git a/tests/aippocampus/test_cli_recovery_cards.py b/tests/aippocampus/test_cli_recovery_cards.py index 0c1be08a..0d80d43a 100644 --- a/tests/aippocampus/test_cli_recovery_cards.py +++ b/tests/aippocampus/test_cli_recovery_cards.py @@ -1317,8 +1317,12 @@ def test_compact_foreground_cards_detail_gate_operator_diagnostics(self) -> None self.assertEqual(provider_payload["foreground_action_contract"], "foreground-action-v2") self.assertNotIn("cannot_claim", provider_payload) self.assertNotIn("boundary_detail", provider_payload) - self.assertIn("boundary_summary", provider_payload) - self.assertTrue(provider_payload["boundary_summary"]["full_detail_owns_diagnostics"]) + self.assertNotIn("boundary_summary", provider_payload) + self.assertTrue(provider_payload["details_available"]) + self.assertEqual( + provider_payload["safe_next_actions"][-1]["id"], + "inspect_provider_doctor_detail", + ) self.assertEqual(aippo.returncode, 2, aippo.stderr) aippo_payload = parse_cli_json(self, aippo) diff --git a/tests/aippocampus/test_frontier_probe.py b/tests/aippocampus/test_frontier_probe.py index 36cba738..c192dabc 100644 --- a/tests/aippocampus/test_frontier_probe.py +++ b/tests/aippocampus/test_frontier_probe.py @@ -155,7 +155,11 @@ def test_frontier_probe_converts_to_non_foreground_dream_seed(self) -> None: ("source refs", "Active Path Packet", "same_decision_space", 0.94, "verified"), ] ) - probes = frontier_probe.build_frontier_probes([journey_row()], graph) + probes = frontier_probe.build_frontier_probes( + [journey_row()], + graph, + now="2026-06-07T00:00:00Z", + ) seeds = frontier_probe.frontier_probes_to_dream_seeds(probes) diff --git a/tests/aippocampus/test_plugin_installer.py b/tests/aippocampus/test_plugin_installer.py index 99e9f954..7abb9841 100644 --- a/tests/aippocampus/test_plugin_installer.py +++ b/tests/aippocampus/test_plugin_installer.py @@ -389,7 +389,10 @@ def test_codex_install_refreshes_existing_configured_marketplace_root(self) -> N self.assertTrue(result["ok"]) self.assertEqual(result["marketplace_source"], "configured_codex_marketplace") - self.assertEqual(Path(result["marketplace"]["root"]), configured_marketplace) + self.assertEqual( + Path(result["marketplace"]["root"]).resolve(), + configured_marketplace.resolve(), + ) self.assertTrue( ( configured_marketplace @@ -404,7 +407,7 @@ def test_codex_install_refreshes_existing_configured_marketplace_root(self) -> N "plugin", "marketplace", "add", - str(configured_marketplace), + str(configured_marketplace.resolve()), ], runner.command_tails, ) diff --git a/tools/aippocampus/run_tests.py b/tools/aippocampus/run_tests.py index b6fb153f..3d29a7b0 100644 --- a/tools/aippocampus/run_tests.py +++ b/tools/aippocampus/run_tests.py @@ -51,7 +51,28 @@ TIER_REPORT_TOP_LIMIT = 10 RUNNER_VERSION = "aippocampus-run-tests-v2" TIMINGS_SCHEMA_VERSION = 2 -QUICK_BUDGET = { + + +class TierCountBudget(TypedDict): + module_count_target: int + test_count_target: int + elapsed_seconds_target: float + target_updated_at: str + target_owner: str + target_rationale: str + replacement_lane_note: str + + +class BroadSuiteReviewTarget(TypedDict): + module_count_review_threshold: int + test_count_review_threshold: int + label: str + review_owner: str + review_decision_date: str + review_rationale: str + + +QUICK_BUDGET: TierCountBudget = { "module_count_target": 50, "test_count_target": 330, "elapsed_seconds_target": 30.0, @@ -67,7 +88,7 @@ "planner-named focused command with an owner note." ), } -PR_BUDGET = { +PR_BUDGET: TierCountBudget = { "module_count_target": 86, "test_count_target": 500, "elapsed_seconds_target": 180.0, @@ -128,15 +149,6 @@ class ModuleTimingRow(TypedDict): ok: bool -class BroadSuiteReviewTarget(TypedDict): - module_count_review_threshold: int - test_count_review_threshold: int - label: str - review_owner: str - review_decision_date: str - review_rationale: str - - def _target_status(actual: int | float, target: int | float) -> str: return "within_target" if actual <= target else "over_target" From f10506a2c084fe1650d11590a4bff739df90ce6e Mon Sep 17 00:00:00 2001 From: Sapientropic Date: Wed, 8 Jul 2026 16:59:20 +0800 Subject: [PATCH 3/4] Address PR closeout review feedback --- .../mcp/current_source_route_policy.py | 21 ++++++- .../aippocampus/test_compact_surface_scan.py | 57 ++++++++++++++++++ .../test_route_note_recall_paths.py | 58 +++++++++++++++++++ tests/aippocampus/test_run_tests_tiers.py | 21 +++++++ tools/aippocampus/compact_surface_scan.py | 25 ++++++-- tools/aippocampus/run_tests.py | 12 ++-- 6 files changed, 183 insertions(+), 11 deletions(-) diff --git a/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py b/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py index 6aa1c3b3..dd5af5eb 100644 --- a/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py +++ b/skills/aippocampus/scripts/aippocampus_runtime/mcp/current_source_route_policy.py @@ -6,6 +6,7 @@ from typing import Any from aippocampus_runtime.recall import associative_path_foreground_gate as apw_gate +from aippocampus_runtime.source.io_kernel import source_ref_identity_key from aippocampus_runtime.source.query_match_gate import query_match_gate @@ -56,10 +57,26 @@ def _route_note_like(packet: Mapping[str, Any]) -> bool: def _route_note_has_joined_source_ref(packet: Mapping[str, Any]) -> bool: for key in ("joined_evidence_refs", "source_refs"): value = packet.get(key) - if isinstance(value, list) and any(isinstance(item, Mapping) for item in value): + if isinstance(value, list) and any(_route_note_ref_reopenable(item) for item in value): return True source_ref = packet.get("source_ref") - return isinstance(source_ref, Mapping) and bool(source_ref.get("message_id")) + return _source_ref_reopenable(source_ref) + + +def _route_note_ref_reopenable(value: Any) -> bool: + if not isinstance(value, Mapping): + return False + nested = value.get("source_ref") + if _source_ref_reopenable(nested): + return True + return _source_ref_reopenable(value) + + +def _source_ref_reopenable(value: Any) -> bool: + if not isinstance(value, Mapping): + return False + source_id, thread_key, message_id, turn_id, line = source_ref_identity_key(value) + return bool((source_id or thread_key) and (message_id or turn_id or line)) def apw_card_allows_primary( diff --git a/tests/aippocampus/test_compact_surface_scan.py b/tests/aippocampus/test_compact_surface_scan.py index c6a06a41..90ccb36f 100644 --- a/tests/aippocampus/test_compact_surface_scan.py +++ b/tests/aippocampus/test_compact_surface_scan.py @@ -2,6 +2,7 @@ import sys import unittest +from unittest import mock from tools.aippocampus import compact_surface_scan @@ -105,6 +106,62 @@ def test_run_scan_reports_first_slow_surface(self) -> None: self.assertGreater(report["checked_count"], 0) self.assertIsNotNone(report["first_slow_surface"]) + def test_run_scan_clamps_exhausted_budget_before_probe(self) -> None: + observed_timeouts: list[float] = [] + + def fake_scan_cli_command(*args: object, **kwargs: object) -> dict[str, object]: + observed_timeouts.append(float(kwargs["timeout_seconds"])) + return { + "surface": "fake slow probe", + "kind": "cli", + "profile": "foreground_compact", + "ok": False, + "error": "scan_budget_exhausted", + "elapsed_ms": 0.0, + } + + with ( + mock.patch.object( + compact_surface_scan, + "CLI_PROBES", + (compact_surface_scan.SurfaceProbe("fake slow probe"),), + ), + mock.patch.object( + compact_surface_scan, + "perf_counter", + side_effect=[0.0, 1.0, 1.0], + ), + mock.patch.object( + compact_surface_scan, + "scan_cli_command", + side_effect=fake_scan_cli_command, + ), + mock.patch.object( + compact_surface_scan, + "scan_mcp_runtime_recovery", + return_value={ + "surface": "mcp_runtime_recovery:agent_recall", + "kind": "mcp", + "profile": "foreground_compact", + "ok": True, + "elapsed_ms": 0.0, + }, + ), + mock.patch.object( + compact_surface_scan, + "scan_mcp_key_tool_compact_cards", + return_value=[], + ), + ): + report = compact_surface_scan.run_scan( + cwd=compact_surface_scan.PATHS.repo_root, + scan_budget_seconds=0.5, + ) + + self.assertEqual(observed_timeouts, [0.0]) + self.assertFalse(report["ok"], report) + self.assertEqual(report["checks"][0]["error"], "scan_budget_exhausted") + def test_mcp_key_tool_compact_scan_covers_aippo_and_background(self) -> None: checks = compact_surface_scan.scan_mcp_key_tool_compact_cards( cwd=compact_surface_scan.PATHS.repo_root, diff --git a/tests/aippocampus/test_route_note_recall_paths.py b/tests/aippocampus/test_route_note_recall_paths.py index f5e28690..ac8007fc 100644 --- a/tests/aippocampus/test_route_note_recall_paths.py +++ b/tests/aippocampus/test_route_note_recall_paths.py @@ -8,6 +8,9 @@ from unittest.mock import patch from aippocampus_runtime import core +from aippocampus_runtime.mcp.current_source_route_policy import ( + primary_deepen_followthrough_reopenable, +) from aippocampus_runtime.source import search from aippocampus_runtime.source.current_source_window import open_current_thread_source_window from tests.aippocampus.product_probe_helpers import call_mcp_tool_payload @@ -192,6 +195,61 @@ def test_mcp_recall_deepens_joined_route_note_source_anchor(self) -> None: ) self.assertIn("Joined final anchor RN-42", window_text) + def test_route_note_primary_deepen_rejects_malformed_source_ref_containers(self) -> None: + for packet in ( + { + "output_mode": "reopenable_route", + "route_kind": "route_note", + "source_refs": [{}], + }, + { + "output_mode": "reopenable_route", + "route_kind": "route_note", + "joined_evidence_refs": [{"kind": "not_a_source"}], + }, + { + "output_mode": "reopenable_route", + "route_kind": "route_note", + "source_refs": [{"line": 12}], + }, + ): + self.assertFalse(primary_deepen_followthrough_reopenable([packet])) + + self.assertTrue( + primary_deepen_followthrough_reopenable( + [ + { + "output_mode": "reopenable_route", + "route_kind": "route_note", + "joined_evidence_refs": [ + { + "source_ref": { + "thread_key": "session:route-note", + "message_id": "msg-route-note", + } + } + ], + } + ] + ) + ) + self.assertTrue( + primary_deepen_followthrough_reopenable( + [ + { + "output_mode": "reopenable_route", + "route_kind": "route_note", + "source_refs": [ + { + "source_id": "clean:route-note", + "message_id": "msg-route-note", + } + ], + } + ] + ) + ) + if __name__ == "__main__": unittest.main() diff --git a/tests/aippocampus/test_run_tests_tiers.py b/tests/aippocampus/test_run_tests_tiers.py index 98f0cfe1..b7ca28d9 100644 --- a/tests/aippocampus/test_run_tests_tiers.py +++ b/tests/aippocampus/test_run_tests_tiers.py @@ -322,6 +322,27 @@ def test_broad_suite_growth_review_is_non_gating_telemetry(self) -> None: ) ) + def test_broad_suite_thresholds_cover_current_pr_head_catalog(self) -> None: + broad_pr = run_tests.suite_growth_review_for_tier( + "broad-pr", + module_count=372, + test_count=3508, + top_modules=[], + ) + full = run_tests.suite_growth_review_for_tier( + "full", + module_count=456, + test_count=4314, + top_modules=[], + ) + + self.assertIsNotNone(broad_pr) + self.assertIsNotNone(full) + assert broad_pr is not None + assert full is not None + self.assertEqual(broad_pr["status"], "within_review_threshold") + self.assertEqual(full["status"], "within_review_threshold") + def test_report_json_cli_prints_compact_report_without_running_tests(self) -> None: report = { "kind": "aippocampus_test_tier_report", diff --git a/tools/aippocampus/compact_surface_scan.py b/tools/aippocampus/compact_surface_scan.py index 55eb7a59..e8ba913a 100644 --- a/tools/aippocampus/compact_surface_scan.py +++ b/tools/aippocampus/compact_surface_scan.py @@ -16,6 +16,7 @@ import shlex import signal import subprocess +import sys from collections.abc import Mapping, Sequence from dataclasses import dataclass from pathlib import Path @@ -141,8 +142,22 @@ def _command_surface(command: str | Sequence[str]) -> str: def _command_argv(command: str | Sequence[str]) -> list[str]: if isinstance(command, str): - return shlex.split(command) - return [str(item) for item in command] + raw = shlex.split(command) + else: + raw = [str(item) for item in command] + if raw and raw[0] == "aippocampus": + return [sys.executable, "-m", "aippocampus_runtime.cli.facade", *raw[1:]] + return raw + + +def _subprocess_env() -> dict[str, str]: + env = dict(os.environ) + existing = env.get("PYTHONPATH") + parts = [str(PATHS.skill_scripts)] + if existing: + parts.append(existing) + env["PYTHONPATH"] = os.pathsep.join(parts) + return env def _terminate_process_tree(proc: subprocess.Popen[str]) -> None: @@ -188,6 +203,7 @@ def scan_cli_command( argv = _command_argv(command) started_at = perf_counter() if timeout_seconds <= 0: + clamped_timeout_seconds = max(0.0, timeout_seconds) return { "surface": surface, "kind": "cli", @@ -195,7 +211,7 @@ def scan_cli_command( "ok": False, "exit_code": None, "elapsed_ms": 0.0, - "timeout_seconds": timeout_seconds, + "timeout_seconds": clamped_timeout_seconds, "error": "scan_budget_exhausted", "stderr_preview": "", } @@ -206,6 +222,7 @@ def scan_cli_command( text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + env=_subprocess_env(), **_popen_kwargs(), ) except OSError as exc: @@ -412,7 +429,7 @@ def run_scan( checks: list[dict[str, Any]] = [] if include_cli: for probe in CLI_PROBES: - remaining_seconds = scan_deadline - perf_counter() + remaining_seconds = max(0.0, scan_deadline - perf_counter()) timeout_seconds = min(probe.timeout_seconds, probe_timeout_seconds, remaining_seconds) checks.append( scan_cli_command( diff --git a/tools/aippocampus/run_tests.py b/tools/aippocampus/run_tests.py index 3d29a7b0..72011ad4 100644 --- a/tools/aippocampus/run_tests.py +++ b/tools/aippocampus/run_tests.py @@ -107,25 +107,27 @@ class BroadSuiteReviewTarget(TypedDict): BROAD_SUITE_REVIEW_TARGETS: dict[str, BroadSuiteReviewTarget] = { "broad-pr": { "module_count_review_threshold": 380, - "test_count_review_threshold": 3400, + "test_count_review_threshold": 3600, "label": "Broad PR", "review_owner": "verification-steward", "review_decision_date": "2026-07-08", "review_rationale": ( "Accepted as a CI/pre-merge breadth lane, not an ordinary local " - "agent ritual; changed-surface planning must still name focused " - "commands before escalating here." + "agent ritual; the current PR-head catalog has been re-counted with " + "reviewer-observed generated test cases included, and changed-surface " + "planning must still name focused commands before escalating here." ), }, "full": { "module_count_review_threshold": 460, - "test_count_review_threshold": 4200, + "test_count_review_threshold": 4400, "label": "Full", "review_owner": "verification-steward", "review_decision_date": "2026-07-08", "review_rationale": ( "Accepted as the complete catalog/release audit lane after recent " - "coverage growth; it remains deferred from default local closeout." + "coverage growth and reviewer-observed PR-head recounts; it remains " + "deferred from default local closeout." ), }, } From 974b4ad18e8571e06043c2683791c7e43fb4ee85 Mon Sep 17 00:00:00 2001 From: Sapientropic Date: Wed, 8 Jul 2026 17:10:25 +0800 Subject: [PATCH 4/4] Keep manual full tests out of PR checks --- .github/workflows/aippocampus-ci.yml | 18 -------------- .github/workflows/full-tests.yml | 29 +++++++++++++++++++++++ tests/aippocampus/test_run_tests_tiers.py | 16 +++++++++++++ 3 files changed, 45 insertions(+), 18 deletions(-) create mode 100644 .github/workflows/full-tests.yml diff --git a/.github/workflows/aippocampus-ci.yml b/.github/workflows/aippocampus-ci.yml index 3e3c5ed5..5f2b2dc4 100644 --- a/.github/workflows/aippocampus-ci.yml +++ b/.github/workflows/aippocampus-ci.yml @@ -269,21 +269,3 @@ jobs: run: | # Covers recurring #140/#242/#402 /var and /private/var path-identity regressions without repeating the full PR tier. python -m unittest tests.aippocampus.test_path_identity tests.aippocampus.test_run_tests_tiers tests.aippocampus.test_macos_install_smoke_workflow -v - - full-tests: - if: github.event_name == 'workflow_dispatch' - runs-on: ubuntu-latest - - steps: - - name: Check out repository - uses: actions/checkout@v5 - - - name: Set up Python - uses: actions/setup-python@v6 - with: - python-version: "3.12" - cache: pip - cache-dependency-path: pyproject.toml - - - name: Full test tier - run: python tools/aippocampus/run_tests.py --tier full diff --git a/.github/workflows/full-tests.yml b/.github/workflows/full-tests.yml new file mode 100644 index 00000000..c0e124c5 --- /dev/null +++ b/.github/workflows/full-tests.yml @@ -0,0 +1,29 @@ +name: AIppocampus Full Test Tier + +on: + workflow_dispatch: + +permissions: + contents: read + +jobs: + full-tests: + name: Full test tier + runs-on: ubuntu-latest + + steps: + - name: Check out repository + uses: actions/checkout@v5 + + - name: Set up Python + uses: actions/setup-python@v6 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: pyproject.toml + + - name: Install full test dependencies + run: python -m pip install -e ".[dev,benchmark,openai-agents-smoke]" + + - name: Full test tier + run: python tools/aippocampus/run_tests.py --tier full diff --git a/tests/aippocampus/test_run_tests_tiers.py b/tests/aippocampus/test_run_tests_tiers.py index b7ca28d9..90c90905 100644 --- a/tests/aippocampus/test_run_tests_tiers.py +++ b/tests/aippocampus/test_run_tests_tiers.py @@ -1068,6 +1068,22 @@ def test_benchmark_smoke_tier_is_exposed_in_cli_and_ci(self) -> None: self.assertNotIn("coverage-${{ matrix.python-version }}", workflow) self.assertNotIn("python benchmarks/aippocampus/benchmark_suite.py", workflow) + def test_manual_full_tier_does_not_emit_skipped_pr_check(self) -> None: + pr_workflow = (REPO_ROOT / ".github" / "workflows" / "aippocampus-ci.yml").read_text( + encoding="utf-8", + ) + full_workflow = (REPO_ROOT / ".github" / "workflows" / "full-tests.yml").read_text( + encoding="utf-8", + ) + + self.assertNotIn("github.event_name == 'workflow_dispatch'", pr_workflow) + self.assertNotIn("github.event_name == \"workflow_dispatch\"", pr_workflow) + self.assertNotIn("full-tests:", pr_workflow) + self.assertIn("workflow_dispatch:", full_workflow) + self.assertNotIn("pull_request:", full_workflow) + self.assertIn("full-tests:", full_workflow) + self.assertIn("--tier full", full_workflow) + def test_quick_and_pr_count_budgets_report_drift_without_hard_gating(self) -> None: for tier in ("quick", "pr"): modules = run_tests.modules_for_tier(tier)