From 57f9e5f4c7f4ae191048d193c062a191f4ca6d08 Mon Sep 17 00:00:00 2001 From: murraylovecode Date: Sat, 15 Aug 2026 06:30:24 -0700 Subject: [PATCH 1/2] Add governed agent identity evidence batch --- .../correction-retraction-review.json | 101 ++ .../cross-project-mappings.json | 110 ++ .../batches/BATCH-2026-005/doi-validation.csv | 10 + .../duplicate-report-review.json | 21 + .../evidence-quality-assessments.json | 137 ++ .../BATCH-2026-005/evidence-statements.json | 1348 +++++++++++++++++ .../BATCH-2026-005/funding-and-conflicts.csv | 10 + .../BATCH-2026-005/institutional-authors.json | 46 + staging/batches/BATCH-2026-005/limitations.md | 15 + .../batches/BATCH-2026-005/locator-review.csv | 23 + staging/batches/BATCH-2026-005/manifest.yml | 116 ++ .../BATCH-2026-005/methodology-matrix.csv | 10 + .../BATCH-2026-005/methodology-records.json | 272 ++++ .../person-author-relationships.json | 902 +++++++++++ .../BATCH-2026-005/rights-review-records.json | 110 ++ .../batches/BATCH-2026-005/rights-review.md | 14 + .../BATCH-2026-005/source-analyses.json | 313 ++++ .../BATCH-2026-005/source-proposals.json | 1189 +++++++++++++++ .../BATCH-2026-005/source-version-review.csv | 10 + .../BATCH-2026-005/statistics-review.csv | 8 + .../BATCH-2026-005/validation-results.md | 24 + 21 files changed, 4789 insertions(+) create mode 100644 staging/batches/BATCH-2026-005/correction-retraction-review.json create mode 100644 staging/batches/BATCH-2026-005/cross-project-mappings.json create mode 100644 staging/batches/BATCH-2026-005/doi-validation.csv create mode 100644 staging/batches/BATCH-2026-005/duplicate-report-review.json create mode 100644 staging/batches/BATCH-2026-005/evidence-quality-assessments.json create mode 100644 staging/batches/BATCH-2026-005/evidence-statements.json create mode 100644 staging/batches/BATCH-2026-005/funding-and-conflicts.csv create mode 100644 staging/batches/BATCH-2026-005/institutional-authors.json create mode 100644 staging/batches/BATCH-2026-005/limitations.md create mode 100644 staging/batches/BATCH-2026-005/locator-review.csv create mode 100644 staging/batches/BATCH-2026-005/manifest.yml create mode 100644 staging/batches/BATCH-2026-005/methodology-matrix.csv create mode 100644 staging/batches/BATCH-2026-005/methodology-records.json create mode 100644 staging/batches/BATCH-2026-005/person-author-relationships.json create mode 100644 staging/batches/BATCH-2026-005/rights-review-records.json create mode 100644 staging/batches/BATCH-2026-005/rights-review.md create mode 100644 staging/batches/BATCH-2026-005/source-analyses.json create mode 100644 staging/batches/BATCH-2026-005/source-proposals.json create mode 100644 staging/batches/BATCH-2026-005/source-version-review.csv create mode 100644 staging/batches/BATCH-2026-005/statistics-review.csv create mode 100644 staging/batches/BATCH-2026-005/validation-results.md diff --git a/staging/batches/BATCH-2026-005/correction-retraction-review.json b/staging/batches/BATCH-2026-005/correction-retraction-review.json new file mode 100644 index 0000000..229636a --- /dev/null +++ b/staging/batches/BATCH-2026-005/correction-retraction-review.json @@ -0,0 +1,101 @@ +[ + { + "source_id": "source-BATCH-2026-005-001", + "canonical_title": "Summary Analysis of Responses to the Request for Information Regarding Security Considerations for AI Agents", + "version_reviewed": "official HTML publication page accessed 2026-08-15", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_canonical_page_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-002", + "canonical_title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile", + "version_reviewed": "July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded", + "correction_status": "no_correction_notice; PDF metadata modification noted", + "retraction_status": "none_identified_in_pdf_or_official_doi_resolution_as_of_2026-08-15", + "correction_or_retraction_locator": "PDF publication-history page and file metadata", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-003", + "canonical_title": "AI Agent Authentication and Authorization", + "version_reviewed": "revision 02; expires 2026-12-03", + "correction_status": "none_identified", + "retraction_status": "active_work_in_progress_not_retracted_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-004", + "canonical_title": "Identity Management for Agentic AI", + "version_reviewed": "October 2025", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_official_pdf_or_publisher_url_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-005", + "canonical_title": "Identity Management for Agentic AI: The new frontier of authorization, authentication, and security for an AI agent world", + "version_reviewed": "v1 submitted 2025-10-29", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-006", + "canonical_title": "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents", + "version_reviewed": "v3; updated after a Llama implementation bug fix and travel-suite update", + "correction_status": "version_update_with_bug_fix", + "retraction_status": "no_retraction_identified;_v3_documents_bug_fix_update", + "correction_or_retraction_locator": "arXiv submission history comment for v3", + "substantive_effect": "v3 fixes a Llama implementation bug and updates the travel suite; earlier numerical results should not be used without version qualification", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-007", + "canonical_title": "Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents", + "version_reviewed": "v4; paper and arXiv page state accepted at ICLR 2025", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-008", + "canonical_title": "Teams of LLM Agents can Exploit Zero-Day Vulnerabilities", + "version_reviewed": "v2 submitted 2025-03-30", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-009", + "canonical_title": "Practices for Governing Agentic AI Systems", + "version_reviewed": "2023; PDF metadata created 2023-12-18", + "correction_status": "none_identified", + "retraction_status": "none_identified_on_official_pdf_url_as_of_2026-08-15", + "correction_or_retraction_locator": "canonical publisher or arXiv version page", + "substantive_effect": "none identified", + "checked_at": "2026-08-15", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/cross-project-mappings.json b/staging/batches/BATCH-2026-005/cross-project-mappings.json new file mode 100644 index 0000000..ba48895 --- /dev/null +++ b/staging/batches/BATCH-2026-005/cross-project-mappings.json @@ -0,0 +1,110 @@ +[ + { + "mapping_id": "mapping-BATCH-2026-005-001", + "source_id": "source-BATCH-2026-005-001", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-002", + "source_id": "source-BATCH-2026-005-002", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-003", + "source_id": "source-BATCH-2026-005-003", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-004", + "source_id": "source-BATCH-2026-005-004", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-005", + "source_id": "source-BATCH-2026-005-005", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-006", + "source_id": "source-BATCH-2026-005-006", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-007", + "source_id": "source-BATCH-2026-005-007", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-008", + "source_id": "source-BATCH-2026-005-008", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + }, + { + "mapping_id": "mapping-BATCH-2026-005-009", + "source_id": "source-BATCH-2026-005-009", + "external_project_id": "executive-ai-research", + "external_snapshot_commit": "d205a6b2e6f4", + "match_status": "unmatched", + "external_record_id": null, + "ownership_status": "third_party", + "off_relationship": "none_identified", + "review_note": "Pinned snapshot manifest contains no production report records; no OFF-owned match can be asserted.", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/doi-validation.csv b/staging/batches/BATCH-2026-005/doi-validation.csv new file mode 100644 index 0000000..4ed894a --- /dev/null +++ b/staging/batches/BATCH-2026-005/doi-validation.csv @@ -0,0 +1,10 @@ +source_id,doi,validation_method,resolution_status,resolved_target,note +source-BATCH-2026-005-001,not reported,canonical source metadata reviewed; no DOI asserted,not_applicable,not applicable,A plausible NIST-style DOI was not assigned because it resolved to a missing PDF and was absent from canonical metadata. +source-BATCH-2026-005-002,10.6028/NIST.AI.600-1,HTTPS GET via doi.org on 2026-08-15,resolved_http_200,https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf, +source-BATCH-2026-005-003,not reported,canonical source metadata reviewed; no DOI asserted,not_applicable,not applicable, +source-BATCH-2026-005-004,not reported,canonical source metadata reviewed; no DOI asserted,not_applicable,not applicable, +source-BATCH-2026-005-005,10.48550/arXiv.2510.25819,HTTPS GET via doi.org on 2026-08-15,resolved_http_200,https://arxiv.org/abs/2510.25819, +source-BATCH-2026-005-006,10.48550/arXiv.2406.13352,HTTPS GET via doi.org on 2026-08-15,resolved_http_200,https://arxiv.org/abs/2406.13352, +source-BATCH-2026-005-007,10.48550/arXiv.2410.02644,HTTPS GET via doi.org on 2026-08-15,resolved_http_200,https://arxiv.org/abs/2410.02644, +source-BATCH-2026-005-008,10.48550/arXiv.2406.01637,HTTPS GET via doi.org on 2026-08-15,resolved_http_200,https://arxiv.org/abs/2406.01637, +source-BATCH-2026-005-009,not reported,canonical source metadata reviewed; no DOI asserted,not_applicable,not applicable, diff --git a/staging/batches/BATCH-2026-005/duplicate-report-review.json b/staging/batches/BATCH-2026-005/duplicate-report-review.json new file mode 100644 index 0000000..e14ef87 --- /dev/null +++ b/staging/batches/BATCH-2026-005/duplicate-report-review.json @@ -0,0 +1,21 @@ +{ + "batch_id": "BATCH-2026-005", + "checked_at": "2026-08-15", + "exact_hash_duplicate_groups": [], + "related_rendition_groups": [ + { + "group_id": "related-BATCH-2026-005-001", + "normalized_work_identity": "Identity Management for Agentic AI", + "source_ids": [ + "source-BATCH-2026-005-004", + "source-BATCH-2026-005-005" + ], + "relationship": "official OpenID rendition and arXiv v1 preprint of substantially the same intellectual work", + "hashes_identical": false, + "resolution": "Retain both source identities for bibliographic provenance but do not double-count their claims as independent evidence.", + "human_review_status": "pending" + } + ], + "discovery_duplicate_queue_checked": true, + "note": "No selected source collides with the five alternate-rendition groups in BATCH-2026-001; the OpenID/arXiv relationship is newly made explicit here." +} diff --git a/staging/batches/BATCH-2026-005/evidence-quality-assessments.json b/staging/batches/BATCH-2026-005/evidence-quality-assessments.json new file mode 100644 index 0000000..9464816 --- /dev/null +++ b/staging/batches/BATCH-2026-005/evidence-quality-assessments.json @@ -0,0 +1,137 @@ +[ + { + "assessment_id": "quality-BATCH-2026-005-001", + "source_id": "source-BATCH-2026-005-001", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "low_for_accessible_version", + "independence": "government_publisher", + "bibliographic_stability": "high", + "evidence_strength": "moderate_for_describing_commenter_themes_not_population_prevalence", + "unresolved": "Full report body and response-level method were not available from the canonical page." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-002", + "source_id": "source-BATCH-2026-005-002", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "moderate_for_consensus_process", + "independence": "government_publisher", + "bibliographic_stability": "high_doi_resolved", + "evidence_strength": "strong_normative_reference_not_empirical_outcome_evidence", + "unresolved": "PDF metadata shows a 2025 modification date without a visible new edition statement." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-003", + "source_id": "source-BATCH-2026-005-003", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "not_applicable_non_empirical", + "independence": "multi_vendor_author_group", + "bibliographic_stability": "medium_due_to_expiring_draft", + "evidence_strength": "strong_for_current_proposal_weak_for_interoperability_outcomes", + "unresolved": "Future revisions may change normative language or identifier choices." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-004", + "source_id": "source-BATCH-2026-005-004", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_white_paper_but_source_selection_not_reported", + "independence": "standards_community_publisher", + "bibliographic_stability": "high_official_pdf", + "evidence_strength": "moderate_normative_synthesis_not_empirical", + "unresolved": "Substantially the same intellectual work appears as arXiv:2510.25819v1." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-005", + "source_id": "source-BATCH-2026-005-005", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_preprint_synthesis", + "independence": "multi_author_preprint", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_normative_synthesis_not_empirical", + "unresolved": "Treat as a related rendition, not an independent evidentiary replication of the OpenID report." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-006", + "source_id": "source-BATCH-2026-005-006", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "high", + "independence": "academic_with_disclosed_industry_affiliations", + "bibliographic_stability": "high_arxiv_and_neurips_records", + "evidence_strength": "strong_for_benchmark_conditions_not_real_world_incidence", + "unresolved": "Confidence-interval construction and raw row numerators are not stated in the reviewed tables." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-007", + "source_id": "source-BATCH-2026-005-007", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "high_for_benchmark_structure", + "independence": "academic", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_to_strong_for_benchmark_conditions", + "unresolved": "Venue acceptance is author/arXiv-reported; OpenReview verification was unavailable in this run. Headline average lacks uncertainty and a compact denominator statement." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-008", + "source_id": "source-BATCH-2026-005-008", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "moderate", + "independence": "academic_with_platform_coordination_disclosed", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_for_selected_sandbox_cases", + "unresolved": "Nonrelease of code/prompts limits replication; funding and conflict disclosures are absent." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + }, + { + "assessment_id": "quality-BATCH-2026-005-009", + "source_id": "source-BATCH-2026-005-009", + "dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_policy_analysis", + "independence": "company_published", + "bibliographic_stability": "high_official_pdf", + "evidence_strength": "moderate_normative_policy_analysis_not_empirical", + "unresolved": "Publication day, funding, affiliations, conflicts, and external review status are not reported in the PDF." + }, + "composite_score_used": false, + "reviewer_note": "Dimensions remain separate; no composite score or publication approval.", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/evidence-statements.json b/staging/batches/BATCH-2026-005/evidence-statements.json new file mode 100644 index 0000000..54397f0 --- /dev/null +++ b/staging/batches/BATCH-2026-005/evidence-statements.json @@ -0,0 +1,1348 @@ +[ + { + "statement_id": "statement-BATCH-2026-005-001", + "source_id": "source-BATCH-2026-005-001", + "person_id": null, + "institutional_author": "NIST Center for AI Standards and Innovation", + "speaker_role_at_source_time": null, + "organization_at_source_time": "National Institute of Standards and Technology", + "statement_type": "empirical_finding", + "neutral_paraphrase": "The NIST authors report broad agreement among RFI commenters that AI agents introduce novel security threats and that security concerns impede adoption.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Official HTML publication page, Abstract, paragraphs 1-2, accessed 2026-08-15", + "locator_type": "html_section", + "source_date": "2026-05-18", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "United States", + "global issues discussed" + ], + "evidence_character": "qualitative observation reported by government authors", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Self-selected RFI commenters summarized by NIST, not a representative population.", + "uncertainty": "Response count, composition, coding method, and prevalence threshold for 'widely agreed' are not reported on the accessible page.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.nist.gov/publications/summary-analysis-responses-request-information-regarding-security-considerations-ai", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Official HTML publication page, Abstract, paragraphs 1-2, accessed 2026-08-15", + "content_hash": "32c07aeeb8f49694dd407080eab757c905dc4395871b52edd46a113208064a03", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-002", + "source_id": "source-BATCH-2026-005-001", + "person_id": null, + "institutional_author": "NIST Center for AI Standards and Innovation", + "speaker_role_at_source_time": null, + "organization_at_source_time": "National Institute of Standards and Technology", + "statement_type": "interpretation", + "neutral_paraphrase": "The NIST authors identify implementation guidance, information sharing, and standards promotion as possible roles for government action.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Official HTML publication page, Abstract, final sentence, accessed 2026-08-15", + "locator_type": "html_section", + "source_date": "2026-05-18", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "United States", + "global issues discussed" + ], + "evidence_character": "author interpretation and policy option", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Options synthesized from the RFI response corpus.", + "uncertainty": "The accessible page does not quantify support for each option.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.nist.gov/publications/summary-analysis-responses-request-information-regarding-security-considerations-ai", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Official HTML publication page, Abstract, final sentence, accessed 2026-08-15", + "content_hash": "32c07aeeb8f49694dd407080eab757c905dc4395871b52edd46a113208064a03", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-003", + "source_id": "source-BATCH-2026-005-002", + "person_id": null, + "institutional_author": "National Institute of Standards and Technology", + "speaker_role_at_source_time": null, + "organization_at_source_time": "National Institute of Standards and Technology", + "statement_type": "recommendation", + "neutral_paraphrase": "Organizations should establish policies for documenting the origin and history of training and generated data while balancing proprietary constraints.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "PDF file p. 18 (document p. 14), table GOVERN 1.2, Action GV-1.2-001, July 2024 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "United States", + "cross-sectoral global applicability" + ], + "evidence_character": "government standards recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Voluntary cross-sectoral generative-AI risk management.", + "uncertainty": "Normative guidance; effectiveness is not evaluated.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "PDF file p. 18 (document p. 14), table GOVERN 1.2, Action GV-1.2-001, July 2024 version", + "content_hash": "6e73620ab6b64e90ef2c04bf0e0d6246185a2f4b1b13cab0df494496cff89b6a", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-004", + "source_id": "source-BATCH-2026-005-002", + "person_id": null, + "institutional_author": "National Institute of Standards and Technology", + "speaker_role_at_source_time": null, + "organization_at_source_time": "National Institute of Standards and Technology", + "statement_type": "methodological_claim", + "neutral_paraphrase": "NIST warns that laboratory benchmarks, in-silico tests, and prompt-jailbreak tests may fail to represent deployment contexts or real-world impacts.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "PDF file p. 53 (document p. 49), Appendix A.1.4, 'Limitations of Current Pre-deployment Test Approaches', July 2024 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "United States", + "cross-sectoral global applicability" + ], + "evidence_character": "methodological limitation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Pre-deployment testing and evaluation for generative AI.", + "uncertainty": "The warning is qualitative and not tied to a measured generalization gap.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "PDF file p. 53 (document p. 49), Appendix A.1.4, 'Limitations of Current Pre-deployment Test Approaches', July 2024 version", + "content_hash": "6e73620ab6b64e90ef2c04bf0e0d6246185a2f4b1b13cab0df494496cff89b6a", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-005", + "source_id": "source-BATCH-2026-005-003", + "person_id": null, + "institutional_author": "IETF Internet-Draft authors", + "speaker_role_at_source_time": null, + "organization_at_source_time": "Internet Engineering Task Force", + "statement_type": "definition", + "neutral_paraphrase": "The draft models an AI agent as a workload that requires an identifier and credentials for authentication by interacting systems and services.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Revision 02 HTML, Section 3 'Agents are workloads', paragraphs following Figure 1", + "locator_type": "html_section", + "source_date": "2026-06-01", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global standards context" + ], + "evidence_character": "normative standards definition", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Internet-Draft revision 02 conceptual model.", + "uncertainty": "Work in progress; later revisions may change the model.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Revision 02 HTML, Section 3 'Agents are workloads', paragraphs following Figure 1", + "content_hash": "e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-006", + "source_id": "source-BATCH-2026-005-003", + "person_id": null, + "institutional_author": "IETF Internet-Draft authors", + "speaker_role_at_source_time": null, + "organization_at_source_time": "Internet Engineering Task Force", + "statement_type": "recommendation", + "neutral_paraphrase": "The draft requires each participating agent to have exactly one WIMSE identifier and requires credentials that cryptographically bind to that identifier.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Revision 02 HTML, Sections 5 'Agent Identifier' and 6 'Agent Credentials'", + "locator_type": "html_section", + "source_date": "2026-06-01", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global standards context" + ], + "evidence_character": "normative standards requirement", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Proposed interoperable identity framework in revision 02.", + "uncertainty": "Not yet an RFC and not validated by deployment evidence.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Revision 02 HTML, Sections 5 'Agent Identifier' and 6 'Agent Credentials'", + "content_hash": "e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-007", + "source_id": "source-BATCH-2026-005-003", + "person_id": null, + "institutional_author": "IETF Internet-Draft authors", + "speaker_role_at_source_time": null, + "organization_at_source_time": "Internet Engineering Task Force", + "statement_type": "recommendation", + "neutral_paraphrase": "When an agent acts for a user or system, delegation context should inform authorization decisions and be preserved in audit trails.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Revision 02 HTML, Section 3 'Agents are workloads', paragraph after Figure 1", + "locator_type": "html_section", + "source_date": "2026-06-01", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global standards context" + ], + "evidence_character": "normative standards recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Agent acting on behalf of a user or system.", + "uncertainty": "Proposed behavior in an expiring Internet-Draft.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Revision 02 HTML, Section 3 'Agents are workloads', paragraph after Figure 1", + "content_hash": "e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-008", + "source_id": "source-BATCH-2026-005-004", + "person_id": null, + "institutional_author": "OpenID Foundation", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenID Foundation", + "statement_type": "interpretation", + "neutral_paraphrase": "The authors judge existing identity patterns strongest for synchronous agents inside one trust domain and less adequate for cross-domain, asynchronous, and recursive delegation.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "PDF file p. 15 (document p. 14), Section 2.14 'Summary of Immediate Solutions', October 2025 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global technology standards context" + ], + "evidence_character": "author interpretation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Large foundation-model-based agent systems considered in the white paper.", + "uncertainty": "Conceptual assessment without comparative implementation testing.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "PDF file p. 15 (document p. 14), Section 2.14 'Summary of Immediate Solutions', October 2025 version", + "content_hash": "e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-009", + "source_id": "source-BATCH-2026-005-004", + "person_id": null, + "institutional_author": "OpenID Foundation", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenID Foundation", + "statement_type": "recommendation", + "neutral_paraphrase": "The report recommends standard protocols, authenticated agent interactions, least privilege, automated lifecycle management, clear audit trails, and adaptable interoperability.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "PDF file p. 17 (document p. 16), 'Best Practices', October 2025 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global technology standards context" + ], + "evidence_character": "technical recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Enterprise agent identity and authorization programs.", + "uncertainty": "No outcomes study establishes relative effectiveness.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "PDF file p. 17 (document p. 16), 'Best Practices', October 2025 version", + "content_hash": "e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-010", + "source_id": "source-BATCH-2026-005-004", + "person_id": null, + "institutional_author": "OpenID Foundation", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenID Foundation", + "statement_type": "recommendation", + "neutral_paraphrase": "For recursive delegation, the report recommends multi-hop chains that progressively narrow permissions at each step.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "PDF file p. 27 (document p. 26), Section 4.4 'Recursive Delegation in Dynamic Agent Networks', October 2025 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global technology standards context" + ], + "evidence_character": "technical recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Future multi-agent networks with delegated authority.", + "uncertainty": "Forward-looking architecture; implementation maturity is not established.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "PDF file p. 27 (document p. 26), Section 4.4 'Recursive Delegation in Dynamic Agent Networks', October 2025 version", + "content_hash": "e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-011", + "source_id": "source-BATCH-2026-005-005", + "person_id": null, + "institutional_author": "Tobin South", + "speaker_role_at_source_time": null, + "organization_at_source_time": "arXiv", + "statement_type": "recommendation", + "neutral_paraphrase": "The preprint proposes scope attenuation so that a subagent receives only a narrowed subset of its parent agent's authority.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2510.25819v1 PDF file p. 27 (document p. 26), Section 4.4", + "locator_type": "pdf_page", + "source_date": "2025-10-29", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global technology standards context" + ], + "evidence_character": "technical recommendation in preprint rendition", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Recursive delegation in future multi-agent networks.", + "uncertainty": "Same intellectual work as the OpenID rendition; do not count as independent evidence.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2510.25819", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2510.25819v1 PDF file p. 27 (document p. 26), Section 4.4", + "content_hash": "b019c9420565afe6a3f81fcd8e9b3375f546003cecab93ecfa25d965b2f68bb2", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-012", + "source_id": "source-BATCH-2026-005-006", + "person_id": null, + "institutional_author": "Edoardo Debenedetti", + "speaker_role_at_source_time": null, + "organization_at_source_time": "NeurIPS 2024 Datasets and Benchmarks Track", + "statement_type": "methodological_claim", + "neutral_paraphrase": "AgentDojo v3 contains 97 user tasks and 629 security test cases in four stateful simulated environments.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.13352v3 PDF p. 1, Abstract; environment counts visually checked in Table 1 on PDF p. 6", + "locator_type": "pdf_page", + "source_date": "2024-06-19", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic environments; no human geography" + ], + "evidence_character": "directly reported benchmark design", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Version 3 benchmark composition.", + "uncertainty": "Curated synthetic cases are not a probability sample of deployed agents.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.13352", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.13352v3 PDF p. 1, Abstract; environment counts visually checked in Table 1 on PDF p. 6", + "content_hash": "349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-013", + "source_id": "source-BATCH-2026-005-006", + "person_id": null, + "institutional_author": "Edoardo Debenedetti", + "speaker_role_at_source_time": null, + "organization_at_source_time": "NeurIPS 2024 Datasets and Benchmarks Track", + "statement_type": "empirical_finding", + "neutral_paraphrase": "For the listed GPT-4o experiments, Table 5 reports targeted attack success of 57.69% without a defense and 6.84% with tool filtering, with reported 95% confidence-interval half-widths of 3.9 and 2.0 percentage points respectively.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.13352v3 PDF p. 20, Table 5", + "locator_type": "pdf_page", + "source_date": "2024-06-19", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic environments; no human geography" + ], + "evidence_character": "direct empirical benchmark finding", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "GPT-4o in AgentDojo v3 security cases and the listed defense configurations.", + "uncertainty": "Raw numerators and interval construction are not reported in the table; values are not production incident rates.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.13352", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.13352v3 PDF p. 20, Table 5", + "content_hash": "349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-014", + "source_id": "source-BATCH-2026-005-006", + "person_id": null, + "institutional_author": "Edoardo Debenedetti", + "speaker_role_at_source_time": null, + "organization_at_source_time": "NeurIPS 2024 Datasets and Benchmarks Track", + "statement_type": "methodological_claim", + "neutral_paraphrase": "The authors report releasing code, model outputs, conversations, documentation, and exact dependencies for the benchmark.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.13352v3 PDF pp. 21-23, Appendix E 'Dataset-related supplementary material'", + "locator_type": "pdf_page", + "source_date": "2024-06-19", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic environments; no human geography" + ], + "evidence_character": "replicability statement", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "AgentDojo benchmark artifacts current to version 3.", + "uncertainty": "External artifact availability can change after the access date.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.13352", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.13352v3 PDF pp. 21-23, Appendix E 'Dataset-related supplementary material'", + "content_hash": "349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-015", + "source_id": "source-BATCH-2026-005-007", + "person_id": null, + "institutional_author": "Hanrong Zhang", + "speaker_role_at_source_time": null, + "organization_at_source_time": "ICLR 2025 / arXiv", + "statement_type": "methodological_claim", + "neutral_paraphrase": "ASB v4 spans 10 scenarios and agents, 400 tasks, more than 400 tools, 27 attack or defense methods, 13 LLM backbones, and seven evaluation metrics.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2410.02644v4 PDF pp. 1-2, Abstract and Introduction; Table 3 on PDF p. 8 visually checked", + "locator_type": "pdf_page", + "source_date": "2024-10-03", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic benchmark; no human geography" + ], + "evidence_character": "directly reported benchmark design", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "ASB version 4 benchmark composition.", + "uncertainty": "Researcher-constructed synthetic benchmark, not a sampled production population.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2410.02644", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2410.02644v4 PDF pp. 1-2, Abstract and Introduction; Table 3 on PDF p. 8 visually checked", + "content_hash": "e20155df01b3a1f6c0a947c4e9e4871957cff2e507939f7ea21a5fe967d84504", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-016", + "source_id": "source-BATCH-2026-005-007", + "person_id": null, + "institutional_author": "Hanrong Zhang", + "speaker_role_at_source_time": null, + "organization_at_source_time": "ICLR 2025 / arXiv", + "statement_type": "empirical_finding", + "neutral_paraphrase": "The authors report a highest average attack success rate of 84.30% among evaluated ASB configurations.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2410.02644v4 PDF p. 1, Abstract", + "locator_type": "pdf_page", + "source_date": "2024-10-03", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic benchmark; no human geography" + ], + "evidence_character": "direct empirical benchmark finding", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Evaluated attacks, agents, and 13 LLM backbones in ASB v4.", + "uncertainty": "The abstract does not state a raw numerator, compact denominator, or confidence interval for the headline average.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2410.02644", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2410.02644v4 PDF p. 1, Abstract", + "content_hash": "e20155df01b3a1f6c0a947c4e9e4871957cff2e507939f7ea21a5fe967d84504", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-017", + "source_id": "source-BATCH-2026-005-008", + "person_id": null, + "institutional_author": "Yuxuan Zhu", + "speaker_role_at_source_time": null, + "organization_at_source_time": "arXiv", + "statement_type": "methodological_claim", + "neutral_paraphrase": "The HPTSA study uses 14 recent, reproducible, manually exploitable open-source web vulnerabilities selected after the tested GPT-4 knowledge cutoff.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.01637v2 PDF p. 4, Section 4 and Tables 1-2", + "locator_type": "pdf_page", + "source_date": "2024-06-02", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic sandbox; no human geography" + ], + "evidence_character": "directly reported benchmark design", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Purposively selected sandboxed web vulnerabilities.", + "uncertainty": "Narrow, non-random sample of vulnerabilities.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.01637", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.01637v2 PDF p. 4, Section 4 and Tables 1-2", + "content_hash": "f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-018", + "source_id": "source-BATCH-2026-005-008", + "person_id": null, + "institutional_author": "Yuxuan Zhu", + "speaker_role_at_source_time": null, + "organization_at_source_time": "arXiv", + "statement_type": "empirical_finding", + "neutral_paraphrase": "HPTSA with GPT-4 reports 42% pass@5 and 18% pass@1 on the 14-vulnerability benchmark.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2", + "locator_type": "pdf_page", + "source_date": "2024-06-02", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic sandbox; no human geography" + ], + "evidence_character": "direct empirical benchmark finding", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Fourteen sandboxed web vulnerabilities and the described HPTSA configuration.", + "uncertainty": "Rounded rates, raw numerators, and confidence intervals are not reported; no extrapolation to all vulnerabilities.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.01637", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2", + "content_hash": "f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-019", + "source_id": "source-BATCH-2026-005-008", + "person_id": null, + "institutional_author": "Yuxuan Zhu", + "speaker_role_at_source_time": null, + "organization_at_source_time": "arXiv", + "statement_type": "methodological_claim", + "neutral_paraphrase": "The authors acknowledge that focusing on open-source web vulnerabilities may bias the study sample and that withheld code and prompts limit replication.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "arXiv:2406.01637v2 PDF pp. 8-9, Section 10 'Limitations, Ethical Considerations'", + "locator_type": "pdf_page", + "source_date": "2024-06-02", + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "synthetic sandbox; no human geography" + ], + "evidence_character": "author limitation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "External validity and replicability of the HPTSA experiment.", + "uncertainty": "The direction and magnitude of bias are not quantified.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.01637", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "arXiv:2406.01637v2 PDF pp. 8-9, Section 10 'Limitations, Ethical Considerations'", + "content_hash": "f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-020", + "source_id": "source-BATCH-2026-005-009", + "person_id": null, + "institutional_author": "Yonadav Shavit", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenAI", + "statement_type": "policy_position", + "neutral_paraphrase": "The authors argue that at least one human legal entity should remain accountable for every uncompensated direct harm caused by an agentic AI system.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Official PDF p. 3, Section 1 'Introduction', 2023 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global policy context" + ], + "evidence_character": "policy position", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Allocation of accountability for direct harms from agentic systems.", + "uncertainty": "Normative position, not a legal conclusion or empirical finding.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Official PDF p. 3, Section 1 'Introduction', 2023 version", + "content_hash": "22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-021", + "source_id": "source-BATCH-2026-005-009", + "person_id": null, + "institutional_author": "Yonadav Shavit", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenAI", + "statement_type": "recommendation", + "neutral_paraphrase": "For high-stakes interactions, the paper suggests that counterparties could require a unique agent identifier linked to a human principal and key accountability information.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Official PDF pp. 13-14, Section 4.6 'Attributability', 2023 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global policy context" + ], + "evidence_character": "policy recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "High-stakes interactions such as private-data or financial transactions.", + "uncertainty": "The authors also identify surveillance, privacy, and spoofing risks; no implementation is evaluated.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Official PDF pp. 13-14, Section 4.6 'Attributability', 2023 version", + "content_hash": "22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "statement_id": "statement-BATCH-2026-005-022", + "source_id": "source-BATCH-2026-005-009", + "person_id": null, + "institutional_author": "Yonadav Shavit", + "speaker_role_at_source_time": null, + "organization_at_source_time": "OpenAI", + "statement_type": "recommendation", + "neutral_paraphrase": "The paper treats interruptibility as a critical backstop and proposes that user shutdown authority extend recursively to subagents.", + "direct_quote": null, + "direct_quote_rights_note": null, + "exact_locator": "Official PDF pp. 14-15, Section 4.7 'Interruptibility and Maintaining Control', 2023 version", + "locator_type": "pdf_page", + "source_date": null, + "topic_ids": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_role_context": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "industry_context": [ + "cross-industry" + ], + "geographic_context": [ + "global policy context" + ], + "evidence_character": "policy recommendation", + "factual_verification_status": "verified_against_exact_source_version", + "statement_scope": "Agent systems and subagents under user or deployer control.", + "uncertainty": "The paper notes safety-critical exceptions and implementation tradeoffs.", + "extraction_method": "manual reading assisted by text extraction and visual PDF verification", + "machine_extraction_confidence": 0.94, + "independent_agent_review_status": "not_started", + "publication_status": "staging_only", + "workflow_status": "statements_extracted", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "manual reading assisted by text extraction and visual PDF verification", + "exact_locator": "Official PDF pp. 14-15, Section 4.7 'Interruptibility and Maintaining Control', 2023 version", + "content_hash": "22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + } +] diff --git a/staging/batches/BATCH-2026-005/funding-and-conflicts.csv b/staging/batches/BATCH-2026-005/funding-and-conflicts.csv new file mode 100644 index 0000000..265c545 --- /dev/null +++ b/staging/batches/BATCH-2026-005/funding-and-conflicts.csv @@ -0,0 +1,10 @@ +source_id,source_classification,funding,sponsor,sponsor_involvement,author_conflicts,employer_or_affiliation_context,disclosure_locator,review_status +source-BATCH-2026-005-001,government report,U.S. government publication; separate research funding not reported,National Institute of Standards and Technology,NIST/CAISI issued the RFI and published the synthesis,not reported,NIST CAISI,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-002,government standards profile,U.S. government publication; separate funding not reported,National Institute of Standards and Technology,"NIST convened the process, authored, reviewed, and published the profile",not reported; government disclaimer states named commercial entities are not endorsements,National Institute of Standards and Technology,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-003,standards document; Internet-Draft,not reported,not reported,not reported,"no conflict statement; employer affiliations include identity, cloud, and AI vendors",Defakto Security | AWS | Zscaler | Ping Identity | OpenAI | Okta,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-004,nonprofit/company white paper,not reported,OpenID Foundation publisher; separate sponsorship not reported,OpenID Foundation publication context; further role not reported,not reported,not reported,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-005,preprint,not reported,not reported,not reported,not reported,not reported,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-006,peer-reviewed conference paper,armasuisse support and SNSF grant 214838 disclosed; no institution explicitly funded benchmark creation,no benchmark sponsor reported,no benchmark sponsor reported,no conflict statement; three authors disclose Invariant Labs affiliations,ETH Zurich | ETH Zurich; Invariant Labs,PDF p. 10 Acknowledgments and p. 23 Funding Sources,machine_checked_human_review_pending +source-BATCH-2026-005-007,"conference paper; acceptance marker verified in version, venue record not independently retrieved",not reported,not reported,not reported,not reported,Zhejiang University | Rutgers University,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending +source-BATCH-2026-005-008,preprint,not reported,not reported; authors state OpenAI requested code and prompts remain confidential,"OpenAI requested confidentiality of agents, code, and prompts; authors disclosed findings to OpenAI",not reported,University of Illinois Urbana-Champaign | independent affiliation listed by email only,PDF pp. 8-9 Limitations/Ethical Considerations,machine_checked_human_review_pending +source-BATCH-2026-005-009,company white paper and policy analysis,not reported,OpenAI publisher; separate sponsor involvement not reported,OpenAI hosts the paper; further involvement not reported,not reported,not reported in paper,"title page, acknowledgements, methodology, and canonical metadata reviewed; no additional disclosure located",machine_checked_human_review_pending diff --git a/staging/batches/BATCH-2026-005/institutional-authors.json b/staging/batches/BATCH-2026-005/institutional-authors.json new file mode 100644 index 0000000..64169bd --- /dev/null +++ b/staging/batches/BATCH-2026-005/institutional-authors.json @@ -0,0 +1,46 @@ +[ + { + "institutional_author_id": "institution-BATCH-2026-005-nist-caisi", + "canonical_name": "NIST Center for AI Standards and Innovation", + "organization": "National Institute of Standards and Technology", + "source_ids": [ + "source-BATCH-2026-005-001" + ], + "author_role": "institutional research unit", + "identity_status": "verified", + "human_review_status": "pending" + }, + { + "institutional_author_id": "institution-BATCH-2026-005-nist", + "canonical_name": "National Institute of Standards and Technology", + "organization": "U.S. Department of Commerce", + "source_ids": [ + "source-BATCH-2026-005-002" + ], + "author_role": "institutional author and publisher", + "identity_status": "verified", + "human_review_status": "pending" + }, + { + "institutional_author_id": "institution-BATCH-2026-005-ietf-draft-authors", + "canonical_name": "IETF Internet-Draft authors", + "organization": "Internet Engineering Task Force", + "source_ids": [ + "source-BATCH-2026-005-003" + ], + "author_role": "collective technical authorship", + "identity_status": "verified", + "human_review_status": "pending" + }, + { + "institutional_author_id": "institution-BATCH-2026-005-openid", + "canonical_name": "OpenID Foundation", + "organization": "OpenID Foundation", + "source_ids": [ + "source-BATCH-2026-005-004" + ], + "author_role": "publisher context", + "identity_status": "verified", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/limitations.md b/staging/batches/BATCH-2026-005/limitations.md new file mode 100644 index 0000000..e5ee8b3 --- /dev/null +++ b/staging/batches/BATCH-2026-005/limitations.md @@ -0,0 +1,15 @@ +# Limitations — BATCH-2026-005 + +This evidence batch contains nine source records representing eight intellectual works because the OpenID report and arXiv preprint are related renditions of substantially the same work. They are retained for provenance but must not be counted as independent corroboration. + +The corpus is English-language and concentrated in U.S.- and European-linked standards, research, and vendor institutions. It has no OFF-owned report and no consulting report. The pinned Executive AI Research snapshot contains no production report records, so every crosswalk remains explicitly unmatched. + +Methodologically, four sources are normative or conceptual rather than empirical. The NIST RFI synthesis is response-based, but the accessible canonical page does not disclose the response count, respondent composition, coding process, or full report body. AgentDojo and ASB are synthetic benchmarks; HPTSA uses a purposive sample of 14 reproducible web vulnerabilities. Their percentages are not production prevalence estimates. + +AgentDojo v3 supersedes earlier versions for extracted statistics because the authors report a Llama implementation bug fix and travel-suite update. NIST AI 600-1 has a July 2024 publication statement but later PDF modification metadata without a visible revision label. No retractions were identified. + +Funding and conflicts are incompletely reported. AgentDojo contains the only explicit funding disclosure in the selected set. Several sources disclose author employer or company affiliations without a formal conflict-of-interest statement. + +Two initially considered nonprofit reports were not selected because the linked full text required authentication or returned a permission-denied page. No access control was bypassed, and no statistics from those unavailable reports were extracted. + +All analyses and statements are machine drafted, staging only, and pending named human review. diff --git a/staging/batches/BATCH-2026-005/locator-review.csv b/staging/batches/BATCH-2026-005/locator-review.csv new file mode 100644 index 0000000..654f63d --- /dev/null +++ b/staging/batches/BATCH-2026-005/locator-review.csv @@ -0,0 +1,23 @@ +statement_id,source_id,locator,locator_type,source_version,visual_pdf_check,result,human_review_status +statement-BATCH-2026-005-001,source-BATCH-2026-005-001,"Official HTML publication page, Abstract, paragraphs 1-2, accessed 2026-08-15",html_section,official HTML publication page accessed 2026-08-15,not applicable; official HTML section checked,pass,pending +statement-BATCH-2026-005-002,source-BATCH-2026-005-001,"Official HTML publication page, Abstract, final sentence, accessed 2026-08-15",html_section,official HTML publication page accessed 2026-08-15,not applicable; official HTML section checked,pass,pending +statement-BATCH-2026-005-003,source-BATCH-2026-005-002,"PDF file p. 18 (document p. 14), table GOVERN 1.2, Action GV-1.2-001, July 2024 version",pdf_page,"PDF file p. 18 (document p. 14), table GOVERN 1.2, Action GV-1.2-001, July 2024 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-004,source-BATCH-2026-005-002,"PDF file p. 53 (document p. 49), Appendix A.1.4, 'Limitations of Current Pre-deployment Test Approaches', July 2024 version",pdf_page,"PDF file p. 53 (document p. 49), Appendix A.1.4, 'Limitations of Current Pre-deployment Test Approaches', July 2024 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-005,source-BATCH-2026-005-003,"Revision 02 HTML, Section 3 'Agents are workloads', paragraphs following Figure 1",html_section,"Revision 02 HTML, Section 3 'Agents are workloads', paragraphs following Figure 1",not applicable; official HTML section checked,pass,pending +statement-BATCH-2026-005-006,source-BATCH-2026-005-003,"Revision 02 HTML, Sections 5 'Agent Identifier' and 6 'Agent Credentials'",html_section,"Revision 02 HTML, Sections 5 'Agent Identifier' and 6 'Agent Credentials'",not applicable; official HTML section checked,pass,pending +statement-BATCH-2026-005-007,source-BATCH-2026-005-003,"Revision 02 HTML, Section 3 'Agents are workloads', paragraph after Figure 1",html_section,"Revision 02 HTML, Section 3 'Agents are workloads', paragraph after Figure 1",not applicable; official HTML section checked,pass,pending +statement-BATCH-2026-005-008,source-BATCH-2026-005-004,"PDF file p. 15 (document p. 14), Section 2.14 'Summary of Immediate Solutions', October 2025 version",pdf_page,"PDF file p. 15 (document p. 14), Section 2.14 'Summary of Immediate Solutions', October 2025 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-009,source-BATCH-2026-005-004,"PDF file p. 17 (document p. 16), 'Best Practices', October 2025 version",pdf_page,"PDF file p. 17 (document p. 16), 'Best Practices', October 2025 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-010,source-BATCH-2026-005-004,"PDF file p. 27 (document p. 26), Section 4.4 'Recursive Delegation in Dynamic Agent Networks', October 2025 version",pdf_page,"PDF file p. 27 (document p. 26), Section 4.4 'Recursive Delegation in Dynamic Agent Networks', October 2025 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-011,source-BATCH-2026-005-005,"arXiv:2510.25819v1 PDF file p. 27 (document p. 26), Section 4.4",pdf_page,"arXiv:2510.25819v1 PDF file p. 27 (document p. 26), Section 4.4","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-012,source-BATCH-2026-005-006,"arXiv:2406.13352v3 PDF p. 1, Abstract; environment counts visually checked in Table 1 on PDF p. 6",pdf_page,"arXiv:2406.13352v3 PDF p. 1, Abstract; environment counts visually checked in Table 1 on PDF p. 6","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-013,source-BATCH-2026-005-006,"arXiv:2406.13352v3 PDF p. 20, Table 5",pdf_page,"arXiv:2406.13352v3 PDF p. 20, Table 5","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-014,source-BATCH-2026-005-006,"arXiv:2406.13352v3 PDF pp. 21-23, Appendix E 'Dataset-related supplementary material'",pdf_page,"arXiv:2406.13352v3 PDF pp. 21-23, Appendix E 'Dataset-related supplementary material'","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-015,source-BATCH-2026-005-007,"arXiv:2410.02644v4 PDF pp. 1-2, Abstract and Introduction; Table 3 on PDF p. 8 visually checked",pdf_page,"arXiv:2410.02644v4 PDF pp. 1-2, Abstract and Introduction; Table 3 on PDF p. 8 visually checked","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-016,source-BATCH-2026-005-007,"arXiv:2410.02644v4 PDF p. 1, Abstract",pdf_page,"arXiv:2410.02644v4 PDF p. 1, Abstract","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-017,source-BATCH-2026-005-008,"arXiv:2406.01637v2 PDF p. 4, Section 4 and Tables 1-2",pdf_page,"arXiv:2406.01637v2 PDF p. 4, Section 4 and Tables 1-2","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-018,source-BATCH-2026-005-008,"arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2",pdf_page,"arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-019,source-BATCH-2026-005-008,"arXiv:2406.01637v2 PDF pp. 8-9, Section 10 'Limitations, Ethical Considerations'",pdf_page,"arXiv:2406.01637v2 PDF pp. 8-9, Section 10 'Limitations, Ethical Considerations'","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-020,source-BATCH-2026-005-009,"Official PDF p. 3, Section 1 'Introduction', 2023 version",pdf_page,"Official PDF p. 3, Section 1 'Introduction', 2023 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-021,source-BATCH-2026-005-009,"Official PDF pp. 13-14, Section 4.6 'Attributability', 2023 version",pdf_page,"Official PDF pp. 13-14, Section 4.6 'Attributability', 2023 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending +statement-BATCH-2026-005-022,source-BATCH-2026-005-009,"Official PDF pp. 14-15, Section 4.7 'Interruptibility and Maintaining Control', 2023 version",pdf_page,"Official PDF pp. 14-15, Section 4.7 'Interruptibility and Maintaining Control', 2023 version","completed for relevant page; table/figure labels, axes, footnotes, and page number checked where applicable",pass,pending diff --git a/staging/batches/BATCH-2026-005/manifest.yml b/staging/batches/BATCH-2026-005/manifest.yml new file mode 100644 index 0000000..2464626 --- /dev/null +++ b/staging/batches/BATCH-2026-005/manifest.yml @@ -0,0 +1,116 @@ +batch_id: BATCH-2026-005 +prompt_id: OEII-EVIDENCE-RESEARCH +prompt_version: "2.0" +topic: Governed identities for AI agents +topic_slug: governed-agent-identities +branch: research/evidence-governed-agent-identities-BATCH-2026-005 +pull_request: null +base_branch: research/discovery-governed-agent-identities-BATCH-2026-001 +execution_date: 2026-08-15 +agent_or_researcher: OpenAI Codex; machine-assisted evidence review; human review pending +model_disclosure: AI-assisted metadata verification, methodology coding, statistical-context review, PDF visual inspection, and statement drafting; no human approval or publication approval occurred. +accepted_evidence_source_candidate_ids: + - candidate-BATCH-2026-001-005 + - candidate-BATCH-2026-001-007 + - candidate-BATCH-2026-001-009 + - candidate-BATCH-2026-001-027 + - candidate-BATCH-2026-001-045 + - candidate-BATCH-2026-001-050 + - candidate-BATCH-2026-001-054 + - candidate-BATCH-2026-001-058 + - candidate-BATCH-2026-001-085 +selected_source_count: 9 +source_classification_counts: + government_reports: 1 + government_standards_profiles: 1 + standards_working_documents: 1 + nonprofit_or_company_white_papers: 2 + preprints: 2 + verified_peer_reviewed_conference_papers: 1 + conference_papers_with_qualified_venue_verification: 1 + consulting_reports: 0 + vendor_sponsored_reports: 0 + off_owned_reports: 0 +statement_count: 22 +statistical_claim_count: 7 +institutional_author_record_count: 4 +person_author_relationship_count: 90 +funding_disclosure_count: 1 +explicit_conflict_disclosure_count: 0 +sample_size_undisclosed_for_empirical_or_response_based_source_count: 1 +weak_methodology_count: 1 +correction_or_version_issue_count: 2 +retraction_count: 0 +production_records_created: 0 +workflow_status: statements_extracted +machine_review_status: machine_checked +human_review_status: pending +output_files: + - source-proposals.json + - institutional-authors.json + - person-author-relationships.json + - methodology-records.json + - source-analyses.json + - evidence-quality-assessments.json + - evidence-statements.json + - methodology-matrix.csv + - statistics-review.csv + - funding-and-conflicts.csv + - correction-retraction-review.json + - duplicate-report-review.json + - cross-project-mappings.json + - rights-review-records.json + - rights-review.md + - doi-validation.csv + - source-version-review.csv + - locator-review.csv + - limitations.md + - validation-results.md +validation_required: + - DOI validation + - source version validation + - duplicate report detection + - methodology-field validation + - sample-context validation + - statistics completeness checks + - page and table locator checks + - funding disclosure checks + - correction and retraction checks + - OFF crosswalk checks + - rights checks + - statement validation + - staging-leak checks +validation_commands: + - node work/evidence-research-2026-005/validate_batch.mjs + - node --import tsx scripts/cli.ts validate data + - node --import tsx scripts/cli.ts validate provenance + - node --import tsx scripts/cli.ts validate attributions + - node --import tsx scripts/cli.ts validate statements + - node --import tsx scripts/cli.ts validate propositions + - node --import tsx scripts/cli.ts validate book-editions + - node --import tsx scripts/cli.ts validate source-locators + - node --import tsx scripts/cli.ts validate rights + - node --import tsx scripts/cli.ts validate review-status + - node --import tsx scripts/cli.ts validate publication + - node --import tsx scripts/cli.ts validate content + - node --import tsx scripts/cli.ts audit duplicates + - node --import tsx scripts/cli.ts audit concentration + - node --import tsx scripts/cli.ts audit coverage + - npm test -- --run + - git diff --check + - staging isolation search across data docs exports and public +validation_results: + - PASS; 9 source proposals and 22 evidence statements conform to canonical schemas. + - PASS; methodology, sample context, statistics, funding, version, correction/retraction, duplicate, OFF crosswalk, rights, and locator checks. + - PASS; five asserted DOIs resolved by HTTPS GET; no guessed DOI retained for NIST AI 800-5. + - PASS; relevant PDF tables, figures, axes, labels, footnotes, and page numbers visually reviewed. + - PASS; canonical repository validators and audits. + - PASS; 24 tests across four test files. + - PASS; staging isolation and whitespace checks. +limitations: + - Two initially considered nonprofit reports were excluded because their full downloads required login or returned a permission-blocked page; they were not silently treated as reviewed. + - BATCH-2026-001 discovery acceptance did not constitute source approval; this batch narrows to exact, reproducibly reviewable versions. + - Four selected sources are non-empirical normative documents and must not be counted as outcome evidence. + - One response-synthesis source lacks sample and coding details on its accessible canonical page. + - Academic benchmarks use synthetic or purposively selected cases and do not estimate production incident prevalence. + - All records remain staging-only pending named human review. diff --git a/staging/batches/BATCH-2026-005/methodology-matrix.csv b/staging/batches/BATCH-2026-005/methodology-matrix.csv new file mode 100644 index 0000000..9104d9f --- /dev/null +++ b/staging/batches/BATCH-2026-005/methodology-matrix.csv @@ -0,0 +1,10 @@ +methodology_id,source_id,candidate_id,source_classification,source_version,research_question,study_design,data_source,sample_size,sampling_method,population,inclusion_criteria,exclusion_criteria,fieldwork_period,geography,industries,executive_roles,response_rate,weighting,statistical_method,qualitative_method,limitations,missing_data,funding,sponsor_involvement,author_conflicts,replication_data_availability,methodology_review_status +methodology-BATCH-2026-005-001,source-BATCH-2026-005-001,candidate-BATCH-2026-001-005,government report,official HTML publication page accessed 2026-08-15,"What security threats, mitigation and assessment practices, and government support priorities did respondents identify for AI agents?",qualitative synthesis of responses to a U.S. government request for information,public RFI responses,not reported on the canonical publication page,self-selected RFI respondents; response recruitment and inclusion process not reported on the canonical page,organizations and individuals responding to the CAISI RFI; composition not reported,not reported,not reported,not reported,United States RFI with globally relevant subject matter; respondent geography not reported,not reported,not reported,not reported,not reported,not applicable; no statistical analysis reported on the canonical page,summary analysis; coding and synthesis procedure not reported on the canonical page,"The accessible page does not disclose the number or composition of responses, coding procedure, dissent prevalence, or source-level traceability.","response count, respondent composition, inclusion rules, coding procedure, and full report body not exposed",U.S. government publication; separate research funding not reported,NIST/CAISI issued the RFI and published the synthesis,not reported,"individual public comments may exist in the RFI docket, but the canonical publication page does not provide a replication package",machine_checked_human_review_pending +methodology-BATCH-2026-005-002,source-BATCH-2026-005-002,candidate-BATCH-2026-001-007,government standards profile,July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded,How should organizations apply the NIST AI RMF to risks unique to or exacerbated by generative AI?,non-empirical cross-sectoral risk-management profile informed by multistakeholder public working-group feedback and public comments,"NIST Generative AI Public Working Group, public consultations, and cited literature",not applicable for a normative profile; contributor and commenter counts not reported,open multistakeholder process; selection details not reported,"organizations designing, developing, deploying, or using generative AI","four primary considerations: governance, content provenance, pre-deployment testing, and incident disclosure",other AI RMF subcategories deferred to future revisions,not reported,U.S.-published cross-sectoral guidance,cross-industry,"governance, deployment, legal, security, and technical leadership",not reported,not applicable,not applicable,multistakeholder synthesis; formal coding method not reported,"Normative profile, not outcome evidence; the document itself warns that laboratory benchmarks and in-silico tests may not extrapolate to real deployment contexts.","working-group size, comment coding, and evidence-weighting procedure not reported",U.S. government publication; separate funding not reported,"NIST convened the process, authored, reviewed, and published the profile",not reported; government disclaimer states named commercial entities are not endorsements,not applicable; no study dataset,machine_checked_human_review_pending +methodology-BATCH-2026-005-003,source-BATCH-2026-005-003,candidate-BATCH-2026-001-009,standards document; Internet-Draft,revision 02; expires 2026-12-03,"How can existing workload-identity, OAuth, and security-event standards be composed for AI-agent authentication and authorization?",non-empirical technical standards proposal,"existing IETF, SPIFFE, WIMSE, OAuth, and OpenID specifications",not applicable,not applicable,"AI-agent workloads and the tools, services, models, systems, and users that interact with them",existing and emerging identity standards relevant to agent workloads,tool-to-tool communication and implementation-specific deployment details are out of scope,not applicable,global internet standards context,cross-industry technology infrastructure,"CISO, CIO, CTO",not applicable,not applicable,not applicable,technical synthesis and protocol composition; selection method not reported,"Internet-Draft status; may be updated, replaced, or expire and must be cited as work in progress.","implementation testing, deployment outcomes, funding, and conflicts not reported",not reported,not reported,"no conflict statement; employer affiliations include identity, cloud, and AI vendors",public source repository and issue tracker disclosed,machine_checked_human_review_pending +methodology-BATCH-2026-005-004,source-BATCH-2026-005-004,candidate-BATCH-2026-001-027,nonprofit/company white paper,October 2025,"Which identity and authorization mechanisms work for present-day agents, and what gaps emerge for autonomous, cross-domain, asynchronous, and multi-agent systems?",non-empirical technical white paper and standards synthesis,"existing OAuth, OpenID, SCIM, MCP, A2A, SPIFFE, and related specifications and examples",not applicable,not applicable,AI-agent systems interacting with protected resources and other agents,"identity, authentication, authorization, delegation, lifecycle, auditability, and agent-commerce mechanisms",paper narrows focus primarily to large foundation-model-based agents,not applicable,global standards context,cross-industry technology infrastructure,"CISO, CIO, CTO",not applicable,not applicable,not applicable,technical and policy synthesis; source-selection procedure not reported,"No empirical implementation study, market sample, or outcomes; some recommendations concern emerging drafts and anticipated architectures.","funding, sponsor role, conflicts, formal review process, and evidence-selection method not reported",not reported,OpenID Foundation publication context; further role not reported,not reported,not applicable; no dataset,machine_checked_human_review_pending +methodology-BATCH-2026-005-005,source-BATCH-2026-005-005,candidate-BATCH-2026-001-045,preprint,v1 submitted 2025-10-29,Same as source-BATCH-2026-005-004; this is a preprint rendition of substantially the same work.,non-empirical technical preprint and standards synthesis,existing identity and agent-protocol specifications,not applicable,not applicable,AI-agent systems,not reported as a systematic review protocol,not reported,not applicable,global standards context,cross-industry,"CISO, CIO, CTO",not applicable,not applicable,not applicable,technical synthesis; selection procedure not reported,Same non-empirical and version-duplication limitations as the OpenID rendition.,"funding, conflicts, review process, and source-selection method not reported",not reported,not reported,not reported,not applicable,machine_checked_human_review_pending +methodology-BATCH-2026-005-006,source-BATCH-2026-005-006,candidate-BATCH-2026-001-050,peer-reviewed conference paper,v3; updated after a Llama implementation bug fix and travel-suite update,"How robust are tool-using LLM agents to indirect prompt injection, and how do attacks and defenses trade off utility and security?",controlled benchmark experiments in stateful synthetic tool environments,"four curated environments, 74 tools, 97 user tasks, 27 injection tasks, and their compatible cross-product yielding 629 security cases",97 user tasks; 27 injection tasks; 629 security test cases; seven listed agent/model configurations in Table 3,researcher-curated realistic tasks and synthetic data; compatible task cross-product,"LLM agents executing tools over simulated workspace, Slack, travel, and banking environments",tasks with executable utility and security checks in supported environments,real user data and uncontrolled production environments,not reported; v3 results reflect the paper's 2024 model/API versions,not applicable,"productivity, communications, travel, and banking simulations",CISO and CTO,not applicable,not reported,"benign utility, utility under attack, targeted and untargeted attack success rates; 95% confidence intervals reported, interval construction not specified in the reviewed text",manual inspection of LLM-generated dummy data; no human-subject interviews,Synthetic tasks and model/API snapshots may not generalize to production; benchmark is dynamic; some defenses reduce utility; no human subjects.,row-specific raw numerators and confidence-interval method not reported in Tables 3-5,armasuisse support and SNSF grant 214838 disclosed; no institution explicitly funded benchmark creation,no benchmark sponsor reported,no conflict statement; three authors disclose Invariant Labs affiliations,"code, conversations, model outputs, and documentation reported available; MIT license for the authors' code",machine_checked_human_review_pending +methodology-BATCH-2026-005-007,source-BATCH-2026-005-007,candidate-BATCH-2026-001-054,"conference paper; acceptance marker verified in version, venue record not independently retrieved",v4; paper and arXiv page state accepted at ICLR 2025,"How vulnerable are LLM-based agents across prompts, tools, planning, and memory, and how effective are corresponding defenses?",multi-scenario controlled security benchmark,"10 scenarios and agents, 400 tasks, more than 400 tools, 27 attack/defense methods, 13 LLM backbones, and 7 evaluation metrics",400 tasks; 10 scenarios; 10 agents; more than 400 tools; 27 attack/defense methods; 13 LLM backbones,"researcher-constructed scenarios, tasks, tools, and attacks; selection procedure not probabilistic",LLM-agent configurations under synthetic attacks and defenses,"agent stages covering system prompt, user prompt, tool usage, planning, and memory retrieval",production incidents and human users,not reported; model versions correspond to experiments before v4 dated 2025-05-30,not applicable,"10 synthetic scenarios including e-commerce, autonomous driving, finance, academic advising, and counseling",CISO and CTO,not applicable,not reported,"attack success, refusal, performance, false-negative, false-positive, and net resilient performance metrics; no confidence intervals reported in the reviewed headline result",not applicable,"Synthetic tools and tasks, researcher-selected attacks, model-version dependence, and no population sampling; external validity is limited.","funding, conflicts, raw denominator for headline average, uncertainty intervals, and fieldwork dates not reported",not reported,not reported,not reported,"code, configurations, Docker setup, and attack scripts reported available on GitHub",machine_checked_human_review_pending +methodology-BATCH-2026-005-008,source-BATCH-2026-005-008,candidate-BATCH-2026-001-058,preprint,v2 submitted 2025-03-30,Can a hierarchical team of task-specific LLM agents autonomously exploit reproducible real-world vulnerabilities without being given vulnerability descriptions?,controlled cybersecurity benchmark experiment in sandboxed reproductions,"14 post-knowledge-cutoff, reproducible open-source web vulnerabilities of medium-or-higher severity",14 vulnerabilities; pass@1 and pass@5 trials; exact total run count not summarized,"purposive selection of recent, reproducible web vulnerabilities with clear success triggers and manual exploitability",reproducible open-source web vulnerabilities; not all vulnerabilities or deployed systems,"after GPT-4 knowledge cutoff, reproducible, web-based, clear trigger, manually exploitable, medium-or-higher severity","non-web vulnerabilities, irreproducible packages, unclear success conditions, and vulnerabilities not manually exploitable",not reported,not applicable,open-source web applications,CISO and CTO,not applicable,each selected vulnerability appears equally weighted in pass metrics; not explicitly stated,pass@1 and pass@5 success rates; manually checked traces; no confidence intervals reported,manual trace verification and case studies,"Small purposive web-vulnerability sample, potential selection bias, model-version dependence, and unreleased code/prompts.","funding, conflicts, exact run count summary, and uncertainty intervals not reported",not reported,"OpenAI requested confidentiality of agents, code, and prompts; authors disclosed findings to OpenAI",not reported,"benchmark vulnerabilities described, but code and prompts intentionally not released",machine_checked_human_review_pending +methodology-BATCH-2026-005-009,source-BATCH-2026-005-009,candidate-BATCH-2026-001-085,company white paper and policy analysis,2023; PDF metadata created 2023-12-18,"What baseline practices could keep agentic AI systems safe and accountable, which lifecycle actors should implement them, and what open questions remain?",non-empirical company white paper and policy analysis,"conceptual analysis, literature, and illustrative scenarios",not applicable,not applicable,"model developers, system deployers, users, and third parties in agentic-system lifecycles",direct safety/accountability practices and indirect societal impacts,paper states many operationalization questions remain open and does not prescribe a complete regulatory system,not applicable,global policy context; examples include U.S. and international governance,cross-industry,"board, CEO, general counsel, CISO, CIO, CTO",not applicable,not applicable,not applicable,conceptual policy analysis; source-selection method not reported,Hypothetical scenarios and normative proposals; no outcome evaluation; many practices remain open questions.,"funding, sponsor involvement, author affiliations, conflicts, and review process not reported in the PDF",not reported,OpenAI hosts the paper; further involvement not reported,not reported,not applicable,machine_checked_human_review_pending diff --git a/staging/batches/BATCH-2026-005/methodology-records.json b/staging/batches/BATCH-2026-005/methodology-records.json new file mode 100644 index 0000000..f4c9838 --- /dev/null +++ b/staging/batches/BATCH-2026-005/methodology-records.json @@ -0,0 +1,272 @@ +[ + { + "methodology_id": "methodology-BATCH-2026-005-001", + "source_id": "source-BATCH-2026-005-001", + "candidate_id": "candidate-BATCH-2026-001-005", + "source_classification": "government report", + "source_version": "official HTML publication page accessed 2026-08-15", + "research_question": "What security threats, mitigation and assessment practices, and government support priorities did respondents identify for AI agents?", + "study_design": "qualitative synthesis of responses to a U.S. government request for information", + "data_source": "public RFI responses", + "sample_size": "not reported on the canonical publication page", + "sampling_method": "self-selected RFI respondents; response recruitment and inclusion process not reported on the canonical page", + "population": "organizations and individuals responding to the CAISI RFI; composition not reported", + "inclusion_criteria": "not reported", + "exclusion_criteria": "not reported", + "fieldwork_period": "not reported", + "geography": "United States RFI with globally relevant subject matter; respondent geography not reported", + "industries": "not reported", + "executive_roles": "not reported", + "response_rate": "not reported", + "weighting": "not reported", + "statistical_method": "not applicable; no statistical analysis reported on the canonical page", + "qualitative_method": "summary analysis; coding and synthesis procedure not reported on the canonical page", + "limitations": "The accessible page does not disclose the number or composition of responses, coding procedure, dissent prevalence, or source-level traceability.", + "missing_data": "response count, respondent composition, inclusion rules, coding procedure, and full report body not exposed", + "funding": "U.S. government publication; separate research funding not reported", + "sponsor_involvement": "NIST/CAISI issued the RFI and published the synthesis", + "author_conflicts": "not reported", + "replication_data_availability": "individual public comments may exist in the RFI docket, but the canonical publication page does not provide a replication package", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-002", + "source_id": "source-BATCH-2026-005-002", + "candidate_id": "candidate-BATCH-2026-001-007", + "source_classification": "government standards profile", + "source_version": "July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded", + "research_question": "How should organizations apply the NIST AI RMF to risks unique to or exacerbated by generative AI?", + "study_design": "non-empirical cross-sectoral risk-management profile informed by multistakeholder public working-group feedback and public comments", + "data_source": "NIST Generative AI Public Working Group, public consultations, and cited literature", + "sample_size": "not applicable for a normative profile; contributor and commenter counts not reported", + "sampling_method": "open multistakeholder process; selection details not reported", + "population": "organizations designing, developing, deploying, or using generative AI", + "inclusion_criteria": "four primary considerations: governance, content provenance, pre-deployment testing, and incident disclosure", + "exclusion_criteria": "other AI RMF subcategories deferred to future revisions", + "fieldwork_period": "not reported", + "geography": "U.S.-published cross-sectoral guidance", + "industries": "cross-industry", + "executive_roles": "governance, deployment, legal, security, and technical leadership", + "response_rate": "not reported", + "weighting": "not applicable", + "statistical_method": "not applicable", + "qualitative_method": "multistakeholder synthesis; formal coding method not reported", + "limitations": "Normative profile, not outcome evidence; the document itself warns that laboratory benchmarks and in-silico tests may not extrapolate to real deployment contexts.", + "missing_data": "working-group size, comment coding, and evidence-weighting procedure not reported", + "funding": "U.S. government publication; separate funding not reported", + "sponsor_involvement": "NIST convened the process, authored, reviewed, and published the profile", + "author_conflicts": "not reported; government disclaimer states named commercial entities are not endorsements", + "replication_data_availability": "not applicable; no study dataset", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-003", + "source_id": "source-BATCH-2026-005-003", + "candidate_id": "candidate-BATCH-2026-001-009", + "source_classification": "standards document; Internet-Draft", + "source_version": "revision 02; expires 2026-12-03", + "research_question": "How can existing workload-identity, OAuth, and security-event standards be composed for AI-agent authentication and authorization?", + "study_design": "non-empirical technical standards proposal", + "data_source": "existing IETF, SPIFFE, WIMSE, OAuth, and OpenID specifications", + "sample_size": "not applicable", + "sampling_method": "not applicable", + "population": "AI-agent workloads and the tools, services, models, systems, and users that interact with them", + "inclusion_criteria": "existing and emerging identity standards relevant to agent workloads", + "exclusion_criteria": "tool-to-tool communication and implementation-specific deployment details are out of scope", + "fieldwork_period": "not applicable", + "geography": "global internet standards context", + "industries": "cross-industry technology infrastructure", + "executive_roles": "CISO, CIO, CTO", + "response_rate": "not applicable", + "weighting": "not applicable", + "statistical_method": "not applicable", + "qualitative_method": "technical synthesis and protocol composition; selection method not reported", + "limitations": "Internet-Draft status; may be updated, replaced, or expire and must be cited as work in progress.", + "missing_data": "implementation testing, deployment outcomes, funding, and conflicts not reported", + "funding": "not reported", + "sponsor_involvement": "not reported", + "author_conflicts": "no conflict statement; employer affiliations include identity, cloud, and AI vendors", + "replication_data_availability": "public source repository and issue tracker disclosed", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-004", + "source_id": "source-BATCH-2026-005-004", + "candidate_id": "candidate-BATCH-2026-001-027", + "source_classification": "nonprofit/company white paper", + "source_version": "October 2025", + "research_question": "Which identity and authorization mechanisms work for present-day agents, and what gaps emerge for autonomous, cross-domain, asynchronous, and multi-agent systems?", + "study_design": "non-empirical technical white paper and standards synthesis", + "data_source": "existing OAuth, OpenID, SCIM, MCP, A2A, SPIFFE, and related specifications and examples", + "sample_size": "not applicable", + "sampling_method": "not applicable", + "population": "AI-agent systems interacting with protected resources and other agents", + "inclusion_criteria": "identity, authentication, authorization, delegation, lifecycle, auditability, and agent-commerce mechanisms", + "exclusion_criteria": "paper narrows focus primarily to large foundation-model-based agents", + "fieldwork_period": "not applicable", + "geography": "global standards context", + "industries": "cross-industry technology infrastructure", + "executive_roles": "CISO, CIO, CTO", + "response_rate": "not applicable", + "weighting": "not applicable", + "statistical_method": "not applicable", + "qualitative_method": "technical and policy synthesis; source-selection procedure not reported", + "limitations": "No empirical implementation study, market sample, or outcomes; some recommendations concern emerging drafts and anticipated architectures.", + "missing_data": "funding, sponsor role, conflicts, formal review process, and evidence-selection method not reported", + "funding": "not reported", + "sponsor_involvement": "OpenID Foundation publication context; further role not reported", + "author_conflicts": "not reported", + "replication_data_availability": "not applicable; no dataset", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-005", + "source_id": "source-BATCH-2026-005-005", + "candidate_id": "candidate-BATCH-2026-001-045", + "source_classification": "preprint", + "source_version": "v1 submitted 2025-10-29", + "research_question": "Same as source-BATCH-2026-005-004; this is a preprint rendition of substantially the same work.", + "study_design": "non-empirical technical preprint and standards synthesis", + "data_source": "existing identity and agent-protocol specifications", + "sample_size": "not applicable", + "sampling_method": "not applicable", + "population": "AI-agent systems", + "inclusion_criteria": "not reported as a systematic review protocol", + "exclusion_criteria": "not reported", + "fieldwork_period": "not applicable", + "geography": "global standards context", + "industries": "cross-industry", + "executive_roles": "CISO, CIO, CTO", + "response_rate": "not applicable", + "weighting": "not applicable", + "statistical_method": "not applicable", + "qualitative_method": "technical synthesis; selection procedure not reported", + "limitations": "Same non-empirical and version-duplication limitations as the OpenID rendition.", + "missing_data": "funding, conflicts, review process, and source-selection method not reported", + "funding": "not reported", + "sponsor_involvement": "not reported", + "author_conflicts": "not reported", + "replication_data_availability": "not applicable", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-006", + "source_id": "source-BATCH-2026-005-006", + "candidate_id": "candidate-BATCH-2026-001-050", + "source_classification": "peer-reviewed conference paper", + "source_version": "v3; updated after a Llama implementation bug fix and travel-suite update", + "research_question": "How robust are tool-using LLM agents to indirect prompt injection, and how do attacks and defenses trade off utility and security?", + "study_design": "controlled benchmark experiments in stateful synthetic tool environments", + "data_source": "four curated environments, 74 tools, 97 user tasks, 27 injection tasks, and their compatible cross-product yielding 629 security cases", + "sample_size": "97 user tasks; 27 injection tasks; 629 security test cases; seven listed agent/model configurations in Table 3", + "sampling_method": "researcher-curated realistic tasks and synthetic data; compatible task cross-product", + "population": "LLM agents executing tools over simulated workspace, Slack, travel, and banking environments", + "inclusion_criteria": "tasks with executable utility and security checks in supported environments", + "exclusion_criteria": "real user data and uncontrolled production environments", + "fieldwork_period": "not reported; v3 results reflect the paper's 2024 model/API versions", + "geography": "not applicable", + "industries": "productivity, communications, travel, and banking simulations", + "executive_roles": "CISO and CTO", + "response_rate": "not applicable", + "weighting": "not reported", + "statistical_method": "benign utility, utility under attack, targeted and untargeted attack success rates; 95% confidence intervals reported, interval construction not specified in the reviewed text", + "qualitative_method": "manual inspection of LLM-generated dummy data; no human-subject interviews", + "limitations": "Synthetic tasks and model/API snapshots may not generalize to production; benchmark is dynamic; some defenses reduce utility; no human subjects.", + "missing_data": "row-specific raw numerators and confidence-interval method not reported in Tables 3-5", + "funding": "armasuisse support and SNSF grant 214838 disclosed; no institution explicitly funded benchmark creation", + "sponsor_involvement": "no benchmark sponsor reported", + "author_conflicts": "no conflict statement; three authors disclose Invariant Labs affiliations", + "replication_data_availability": "code, conversations, model outputs, and documentation reported available; MIT license for the authors' code", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-007", + "source_id": "source-BATCH-2026-005-007", + "candidate_id": "candidate-BATCH-2026-001-054", + "source_classification": "conference paper; acceptance marker verified in version, venue record not independently retrieved", + "source_version": "v4; paper and arXiv page state accepted at ICLR 2025", + "research_question": "How vulnerable are LLM-based agents across prompts, tools, planning, and memory, and how effective are corresponding defenses?", + "study_design": "multi-scenario controlled security benchmark", + "data_source": "10 scenarios and agents, 400 tasks, more than 400 tools, 27 attack/defense methods, 13 LLM backbones, and 7 evaluation metrics", + "sample_size": "400 tasks; 10 scenarios; 10 agents; more than 400 tools; 27 attack/defense methods; 13 LLM backbones", + "sampling_method": "researcher-constructed scenarios, tasks, tools, and attacks; selection procedure not probabilistic", + "population": "LLM-agent configurations under synthetic attacks and defenses", + "inclusion_criteria": "agent stages covering system prompt, user prompt, tool usage, planning, and memory retrieval", + "exclusion_criteria": "production incidents and human users", + "fieldwork_period": "not reported; model versions correspond to experiments before v4 dated 2025-05-30", + "geography": "not applicable", + "industries": "10 synthetic scenarios including e-commerce, autonomous driving, finance, academic advising, and counseling", + "executive_roles": "CISO and CTO", + "response_rate": "not applicable", + "weighting": "not reported", + "statistical_method": "attack success, refusal, performance, false-negative, false-positive, and net resilient performance metrics; no confidence intervals reported in the reviewed headline result", + "qualitative_method": "not applicable", + "limitations": "Synthetic tools and tasks, researcher-selected attacks, model-version dependence, and no population sampling; external validity is limited.", + "missing_data": "funding, conflicts, raw denominator for headline average, uncertainty intervals, and fieldwork dates not reported", + "funding": "not reported", + "sponsor_involvement": "not reported", + "author_conflicts": "not reported", + "replication_data_availability": "code, configurations, Docker setup, and attack scripts reported available on GitHub", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-008", + "source_id": "source-BATCH-2026-005-008", + "candidate_id": "candidate-BATCH-2026-001-058", + "source_classification": "preprint", + "source_version": "v2 submitted 2025-03-30", + "research_question": "Can a hierarchical team of task-specific LLM agents autonomously exploit reproducible real-world vulnerabilities without being given vulnerability descriptions?", + "study_design": "controlled cybersecurity benchmark experiment in sandboxed reproductions", + "data_source": "14 post-knowledge-cutoff, reproducible open-source web vulnerabilities of medium-or-higher severity", + "sample_size": "14 vulnerabilities; pass@1 and pass@5 trials; exact total run count not summarized", + "sampling_method": "purposive selection of recent, reproducible web vulnerabilities with clear success triggers and manual exploitability", + "population": "reproducible open-source web vulnerabilities; not all vulnerabilities or deployed systems", + "inclusion_criteria": "after GPT-4 knowledge cutoff, reproducible, web-based, clear trigger, manually exploitable, medium-or-higher severity", + "exclusion_criteria": "non-web vulnerabilities, irreproducible packages, unclear success conditions, and vulnerabilities not manually exploitable", + "fieldwork_period": "not reported", + "geography": "not applicable", + "industries": "open-source web applications", + "executive_roles": "CISO and CTO", + "response_rate": "not applicable", + "weighting": "each selected vulnerability appears equally weighted in pass metrics; not explicitly stated", + "statistical_method": "pass@1 and pass@5 success rates; manually checked traces; no confidence intervals reported", + "qualitative_method": "manual trace verification and case studies", + "limitations": "Small purposive web-vulnerability sample, potential selection bias, model-version dependence, and unreleased code/prompts.", + "missing_data": "funding, conflicts, exact run count summary, and uncertainty intervals not reported", + "funding": "not reported", + "sponsor_involvement": "OpenAI requested confidentiality of agents, code, and prompts; authors disclosed findings to OpenAI", + "author_conflicts": "not reported", + "replication_data_availability": "benchmark vulnerabilities described, but code and prompts intentionally not released", + "methodology_review_status": "machine_checked_human_review_pending" + }, + { + "methodology_id": "methodology-BATCH-2026-005-009", + "source_id": "source-BATCH-2026-005-009", + "candidate_id": "candidate-BATCH-2026-001-085", + "source_classification": "company white paper and policy analysis", + "source_version": "2023; PDF metadata created 2023-12-18", + "research_question": "What baseline practices could keep agentic AI systems safe and accountable, which lifecycle actors should implement them, and what open questions remain?", + "study_design": "non-empirical company white paper and policy analysis", + "data_source": "conceptual analysis, literature, and illustrative scenarios", + "sample_size": "not applicable", + "sampling_method": "not applicable", + "population": "model developers, system deployers, users, and third parties in agentic-system lifecycles", + "inclusion_criteria": "direct safety/accountability practices and indirect societal impacts", + "exclusion_criteria": "paper states many operationalization questions remain open and does not prescribe a complete regulatory system", + "fieldwork_period": "not applicable", + "geography": "global policy context; examples include U.S. and international governance", + "industries": "cross-industry", + "executive_roles": "board, CEO, general counsel, CISO, CIO, CTO", + "response_rate": "not applicable", + "weighting": "not applicable", + "statistical_method": "not applicable", + "qualitative_method": "conceptual policy analysis; source-selection method not reported", + "limitations": "Hypothetical scenarios and normative proposals; no outcome evaluation; many practices remain open questions.", + "missing_data": "funding, sponsor involvement, author affiliations, conflicts, and review process not reported in the PDF", + "funding": "not reported", + "sponsor_involvement": "OpenAI hosts the paper; further involvement not reported", + "author_conflicts": "not reported", + "replication_data_availability": "not applicable", + "methodology_review_status": "machine_checked_human_review_pending" + } +] diff --git a/staging/batches/BATCH-2026-005/person-author-relationships.json b/staging/batches/BATCH-2026-005/person-author-relationships.json new file mode 100644 index 0000000..887baad --- /dev/null +++ b/staging/batches/BATCH-2026-005/person-author-relationships.json @@ -0,0 +1,902 @@ +[ + { + "person_candidate_id": "person-BATCH-2026-005-jared-riggs", + "person_name": "Jared Riggs", + "source_id": "source-BATCH-2026-005-001", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "NIST CAISI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-maia-hamin", + "person_name": "Maia Hamin", + "source_id": "source-BATCH-2026-005-001", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "NIST CAISI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-neil-perry", + "person_name": "Neil Perry", + "source_id": "source-BATCH-2026-005-001", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "NIST CAISI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-benjamin-edelman", + "person_name": "Benjamin Edelman", + "source_id": "source-BATCH-2026-005-001", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "NIST CAISI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-peter-cihon", + "person_name": "Peter Cihon", + "source_id": "source-BATCH-2026-005-001", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "NIST CAISI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-pieter-kasselman", + "person_name": "Pieter Kasselman", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "Defakto Security", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jeff-lombardo", + "person_name": "Jeff Lombardo", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "AWS", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-yaron-rosomakho", + "person_name": "Yaron Rosomakho", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "Zscaler", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-brian-campbell", + "person_name": "Brian Campbell", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "Ping Identity", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-nick-steele", + "person_name": "Nick Steele", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "OpenAI", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-aaron-parecki", + "person_name": "Aaron Parecki", + "source_id": "source-BATCH-2026-005-003", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "Okta", + "verification_basis": "official HTML author metadata", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-tobin-south", + "person_name": "Tobin South", + "source_id": "source-BATCH-2026-005-004", + "relationship": "lead_editor", + "author_order": 1, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-subramanya-nagabhushanaradhya", + "person_name": "Subramanya Nagabhushanaradhya", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 2, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ayesha-dissanayaka", + "person_name": "Ayesha Dissanayaka", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 3, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-sarah-cecchetti", + "person_name": "Sarah Cecchetti", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 4, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-george-fletcher", + "person_name": "George Fletcher", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 5, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-victor-lu", + "person_name": "Victor Lu", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 6, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-aldo-pietropaolo", + "person_name": "Aldo Pietropaolo", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 7, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-dean-h-saxe", + "person_name": "Dean H. Saxe", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 8, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jeff-lombardo", + "person_name": "Jeff Lombardo", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 9, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-abhishek-shivalingaiah", + "person_name": "Abhishek Shivalingaiah", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 10, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-stan-bounev", + "person_name": "Stan Bounev", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 11, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alex-keisner", + "person_name": "Alex Keisner", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 12, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-andor-kesselman", + "person_name": "Andor Kesselman", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 13, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-zack-proser", + "person_name": "Zack Proser", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 14, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ginny-fahs", + "person_name": "Ginny Fahs", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 15, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-andrew-bunyea", + "person_name": "Andrew Bunyea", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 16, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ben-moskowitz", + "person_name": "Ben Moskowitz", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 17, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-atul-tulshibagwale", + "person_name": "Atul Tulshibagwale", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 18, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-dazza-greenwood", + "person_name": "Dazza Greenwood", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 19, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jiaxin-pei", + "person_name": "Jiaxin Pei", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 20, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alex-pentland", + "person_name": "Alex Pentland", + "source_id": "source-BATCH-2026-005-004", + "relationship": "contributor", + "author_order": 21, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-tobin-south", + "person_name": "Tobin South", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-subramanya-nagabhushanaradhya", + "person_name": "Subramanya Nagabhushanaradhya", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ayesha-dissanayaka", + "person_name": "Ayesha Dissanayaka", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-sarah-cecchetti", + "person_name": "Sarah Cecchetti", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-george-fletcher", + "person_name": "George Fletcher", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-victor-lu", + "person_name": "Victor Lu", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-aldo-pietropaolo", + "person_name": "Aldo Pietropaolo", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 7, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-dean-h-saxe", + "person_name": "Dean H. Saxe", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 8, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jeff-lombardo", + "person_name": "Jeff Lombardo", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 9, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-abhishek-shivalingaiah", + "person_name": "Abhishek Shivalingaiah", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 10, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-stan-bounev", + "person_name": "Stan Bounev", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 11, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alex-keisner", + "person_name": "Alex Keisner", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 12, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-andor-kesselman", + "person_name": "Andor Kesselman", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 13, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-zack-proser", + "person_name": "Zack Proser", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 14, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ginny-fahs", + "person_name": "Ginny Fahs", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 15, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-andrew-bunyea", + "person_name": "Andrew Bunyea", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 16, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-ben-moskowitz", + "person_name": "Ben Moskowitz", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 17, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-atul-tulshibagwale", + "person_name": "Atul Tulshibagwale", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 18, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-dazza-greenwood", + "person_name": "Dazza Greenwood", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 19, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jiaxin-pei", + "person_name": "Jiaxin Pei", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 20, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alex-pentland", + "person_name": "Alex Pentland", + "source_id": "source-BATCH-2026-005-005", + "relationship": "author", + "author_order": 21, + "affiliation_at_source_time": "not reported", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-edoardo-debenedetti", + "person_name": "Edoardo Debenedetti", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "ETH Zurich", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jie-zhang", + "person_name": "Jie Zhang", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "ETH Zurich", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-mislav-balunovic", + "person_name": "Mislav Balunovic", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "ETH Zurich; Invariant Labs", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-luca-beurer-kellner", + "person_name": "Luca Beurer-Kellner", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "ETH Zurich; Invariant Labs", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-marc-fischer", + "person_name": "Marc Fischer", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "ETH Zurich; Invariant Labs", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-florian-tramer", + "person_name": "Florian Tramer", + "source_id": "source-BATCH-2026-005-006", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "ETH Zurich", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-hanrong-zhang", + "person_name": "Hanrong Zhang", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "Zhejiang University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-jingyuan-huang", + "person_name": "Jingyuan Huang", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "Rutgers University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-kai-mei", + "person_name": "Kai Mei", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "Rutgers University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-yifei-yao", + "person_name": "Yifei Yao", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "Zhejiang University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-zhenting-wang", + "person_name": "Zhenting Wang", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "Rutgers University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-chenlu-zhan", + "person_name": "Chenlu Zhan", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "Zhejiang University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-hongwei-wang", + "person_name": "Hongwei Wang", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 7, + "affiliation_at_source_time": "Zhejiang University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-yongfeng-zhang", + "person_name": "Yongfeng Zhang", + "source_id": "source-BATCH-2026-005-007", + "relationship": "author", + "author_order": 8, + "affiliation_at_source_time": "Rutgers University", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-yuxuan-zhu", + "person_name": "Yuxuan Zhu", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-antony-kellermann", + "person_name": "Antony Kellermann", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "independent affiliation listed by email only", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-akul-gupta", + "person_name": "Akul Gupta", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-philip-li", + "person_name": "Philip Li", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-richard-fang", + "person_name": "Richard Fang", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-rohan-bindu", + "person_name": "Rohan Bindu", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-daniel-kang", + "person_name": "Daniel Kang", + "source_id": "source-BATCH-2026-005-008", + "relationship": "author", + "author_order": 7, + "affiliation_at_source_time": "University of Illinois Urbana-Champaign", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-yonadav-shavit", + "person_name": "Yonadav Shavit", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 1, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-sandhini-agarwal", + "person_name": "Sandhini Agarwal", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 2, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-miles-brundage", + "person_name": "Miles Brundage", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 3, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-steven-adler", + "person_name": "Steven Adler", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 4, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-cullen-o-keefe", + "person_name": "Cullen O'Keefe", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 5, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-rosie-campbell", + "person_name": "Rosie Campbell", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 6, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-teddy-lee", + "person_name": "Teddy Lee", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 7, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-pamela-mishkin", + "person_name": "Pamela Mishkin", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 8, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-tyna-eloundou", + "person_name": "Tyna Eloundou", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 9, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alan-hickey", + "person_name": "Alan Hickey", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 10, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-katarina-slama", + "person_name": "Katarina Slama", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 11, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-lama-ahmad", + "person_name": "Lama Ahmad", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 12, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-paul-mcmillan", + "person_name": "Paul McMillan", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 13, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alex-beutel", + "person_name": "Alex Beutel", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 14, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-alexandre-passos", + "person_name": "Alexandre Passos", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 15, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + }, + { + "person_candidate_id": "person-BATCH-2026-005-david-g-robinson", + "person_name": "David G. Robinson", + "source_id": "source-BATCH-2026-005-009", + "relationship": "author", + "author_order": 16, + "affiliation_at_source_time": "not reported in paper", + "verification_basis": "PDF title or contributor page", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/rights-review-records.json b/staging/batches/BATCH-2026-005/rights-review-records.json new file mode 100644 index 0000000..e8fd349 --- /dev/null +++ b/staging/batches/BATCH-2026-005/rights-review-records.json @@ -0,0 +1,110 @@ +[ + { + "source_id": "source-BATCH-2026-005-001", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Official NIST publication page and its analytical abstract reviewed; no full report file was exposed by the canonical page.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-002", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete official 64-page PDF reviewed; governance action tables and Appendix A limitation text visually inspected.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "NIST government publication; conservative link-and-paraphrase treatment retained.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-003", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete official IETF HTML rendition reviewed with stable section and paragraph locators.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "IETF Trust legal provisions apply to the source; no source text is republished.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-004", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete official 33-page PDF reviewed; identity, delegation, auditability, and use-case sections inspected visually.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-005", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete arXiv v1 PDF reviewed and compared with the OpenID rendition; the content is substantially the same intellectual work with rendition-level differences.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-006", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete 26-page v3 conference paper reviewed; benchmark composition, figures, Tables 1 and 3-5, confidence intervals, limitations, and funding inspected visually.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-007", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete 36-page v4 paper reviewed; attack/defense definitions, scenario tables, metric table, result tables, and reproducibility appendix inspected visually.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-008", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete 10-page v2 preprint reviewed; Tables 1-2, Figures 2-3, methods, results, and limitation text inspected visually.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + }, + { + "source_id": "source-BATCH-2026-005-009", + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "access_basis": "Complete official 23-page PDF reviewed; definition, accountability, monitoring, attribution, interruptibility, indirect impacts, and acknowledgements inspected visually.", + "stored_material": "metadata, original analytical summaries, brief statistical cells, and exact locators only", + "full_text_stored_in_repository": false, + "tables_or_figures_republished": false, + "direct_quotes_staged": false, + "special_note": "Third-party rights retained by source owners.", + "human_review_status": "pending" + } +] diff --git a/staging/batches/BATCH-2026-005/rights-review.md b/staging/batches/BATCH-2026-005/rights-review.md new file mode 100644 index 0000000..2d725b6 --- /dev/null +++ b/staging/batches/BATCH-2026-005/rights-review.md @@ -0,0 +1,14 @@ +# Rights review — BATCH-2026-005 + +All nine selected sources are third-party works. This batch stores bibliographic metadata, original analytical summaries, neutral paraphrases, exact locators, limited statistical cells, and links. It does not store or republish source PDFs, complete tables, figures, abstracts, or substantial source text. + +## Decisions + +- NIST sources: retained conservatively as `link_and_paraphrase`; no reliance on a broader public-domain determination is needed for this batch. +- IETF Internet-Draft: IETF Trust legal provisions remain controlling; the batch stores only paraphrase, metadata, and section locators. +- OpenID, arXiv, NeurIPS/ICLR-linked, University of Illinois, and OpenAI works: third-party rights retained by their owners; no repository license claim is made. +- PDF visual review was conducted locally for research quality control. No downloaded source file or rendered page is staged in the repository. +- No direct quotations are proposed for publication. +- The two access-blocked nonprofit candidates considered during scoping were excluded rather than bypassing access controls. + +Human rights review remains pending. diff --git a/staging/batches/BATCH-2026-005/source-analyses.json b/staging/batches/BATCH-2026-005/source-analyses.json new file mode 100644 index 0000000..a0878c2 --- /dev/null +++ b/staging/batches/BATCH-2026-005/source-analyses.json @@ -0,0 +1,313 @@ +[ + { + "source_id": "source-BATCH-2026-005-001", + "candidate_id": "candidate-BATCH-2026-001-005", + "source_classification": "government report", + "canonical_title": "Summary Analysis of Responses to the Request for Information Regarding Security Considerations for AI Agents", + "source_version": "official HTML publication page accessed 2026-08-15", + "research_question": "What security concerns and government actions emerged from an RFI on AI-agent security?", + "methodology_summary": "NIST describes a qualitative synthesis of RFI responses, but the accessible canonical page omits the response count, respondent characteristics, coding process, and full analytical method.", + "major_findings": [ + "The authors report broad agreement among commenters that agent security creates adoption barriers and requires adaptation of established cybersecurity practice." + ], + "author_interpretations": [ + "The authors identify implementation guidance, information sharing, and standards promotion as possible government roles." + ], + "important_limitations": [ + "No disclosed denominator, response composition, or qualitative coding method on the accessible page.", + "RFI respondents are self-selected and cannot be treated as a representative population." + ], + "applicability_to_executive_roles": [ + "CISOs and CTOs can use the themes as agenda-setting input, not prevalence estimates.", + "General counsel and boards should treat proposed government roles as policy options." + ], + "geography_limits": "U.S. government process; respondent geography not disclosed.", + "industry_limits": "Respondent industry mix not disclosed.", + "funding_and_sponsor_context": "NIST/CAISI designed and published the work; separate funding and conflicts were not reported.", + "relevance_to_books_and_media": "Provides an institutional counterweight to practitioner media and conceptual books in the discovery corpus.", + "evidence_strength_assessment": "Moderate for reporting the publisher's synthesis; weak for any claim about how common a view is beyond respondents.", + "unresolved_questions": [ + "How many responses were analyzed?", + "How were themes coded and disagreements weighted?", + "Where is the complete version or response-level appendix?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-002", + "candidate_id": "candidate-BATCH-2026-001-007", + "source_classification": "government standards profile", + "canonical_title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile", + "source_version": "July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded", + "research_question": "How can AI RMF functions be applied to generative-AI risks?", + "methodology_summary": "A NIST cross-sectoral profile informed by an open public working group, consultations, public comments, and literature; it is normative rather than an empirical study.", + "major_findings": [ + "The profile organizes suggested actions across governance, mapping, measurement, and management functions.", + "It explicitly warns that pre-deployment tests and benchmark results may not generalize to real-world contexts." + ], + "author_interpretations": [ + "Generative AI may warrant additional oversight, documentation, and management controls." + ], + "important_limitations": [ + "No quantitative sample or outcome evaluation.", + "Public-input synthesis method and denominator are not disclosed." + ], + "applicability_to_executive_roles": [ + "Board and executive teams can use the profile as a control-design reference.", + "CISOs and CTOs can map testing, provenance, and incident controls to deployment workflows." + ], + "geography_limits": "U.S. government publication with cross-sector intent; legal fit must be localized.", + "industry_limits": "Cross-sectoral and deliberately not tailored to one regulated industry.", + "funding_and_sponsor_context": "NIST is both convener and publisher; separate funding and conflicts were not reported.", + "relevance_to_books_and_media": "Provides a stable baseline against which faster-moving practitioner sources can be compared.", + "evidence_strength_assessment": "High as an authoritative normative source; not evidence that controls produce outcomes.", + "unresolved_questions": [ + "Whether a formally versioned post-2024 revision exists corresponding to the PDF metadata modification date." + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-003", + "candidate_id": "candidate-BATCH-2026-001-009", + "source_classification": "standards document; Internet-Draft", + "canonical_title": "AI Agent Authentication and Authorization", + "source_version": "revision 02; expires 2026-12-03", + "research_question": "Which existing standards can establish agent identity, authentication, authorization, delegation, and observability?", + "methodology_summary": "A compositional standards proposal, not an empirical evaluation.", + "major_findings": [ + "The draft models an agent as a workload that requires a stable identifier and cryptographically bound credentials.", + "It proposes preserving user or system delegation context in authorization decisions and audit trails." + ], + "author_interpretations": [ + "Existing standards can supply much of the required stack, while gaps remain for future standardization." + ], + "important_limitations": [ + "Revision 02 is an expiring work in progress.", + "No interoperability tests or deployment outcomes are reported." + ], + "applicability_to_executive_roles": [ + "CTOs can evaluate architecture alignment.", + "CISOs and CIOs can derive identity-lifecycle and logging requirements." + ], + "geography_limits": "No jurisdiction-specific legal analysis.", + "industry_limits": "Protocol-level and not tailored to sector controls.", + "funding_and_sponsor_context": "Funding and conflicts are not reported; all six employer affiliations are disclosed.", + "relevance_to_books_and_media": "Gives technically precise support or qualification for broader practitioner recommendations.", + "evidence_strength_assessment": "High for what revision 02 proposes; no evidence yet that the design is interoperable or effective in deployment.", + "unresolved_questions": [ + "Which provisions will survive later draft revisions or RFC review?", + "How will the proposal be tested across vendors?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-004", + "candidate_id": "candidate-BATCH-2026-001-027", + "source_classification": "nonprofit/company white paper", + "canonical_title": "Identity Management for Agentic AI", + "source_version": "October 2025", + "research_question": "How should identity systems evolve for delegated, asynchronous, cross-domain, and recursive agent activity?", + "methodology_summary": "A multi-author technical white paper synthesizing identity standards and worked use cases; no empirical study design.", + "major_findings": [ + "The authors argue that current mechanisms are strongest for synchronous, single-trust-domain patterns and weaker for cross-domain or recursive delegation.", + "The paper distinguishes agent identity from user identity and calls for lifecycle management and enriched audit trails." + ], + "author_interpretations": [ + "Impersonation should give way to explicit delegated authority.", + "Recursive delegation requires progressive scope attenuation." + ], + "important_limitations": [ + "No implementation benchmark or comparative test.", + "Formal source-selection, funding, conflicts, and review process are not reported." + ], + "applicability_to_executive_roles": [ + "CISOs can translate best-practice bullets into control requirements.", + "CTOs can use the use cases to identify protocol gaps.", + "CIOs can connect agent lifecycle to existing IGA processes." + ], + "geography_limits": "No jurisdiction-specific treatment.", + "industry_limits": "Broad infrastructure focus; sector-specific assurance requirements remain outside scope.", + "funding_and_sponsor_context": "OpenID Foundation is the publisher; funding, sponsor involvement, and conflicts were not reported.", + "relevance_to_books_and_media": "A core technical bridge between identity books, standards drafts, and practitioner media.", + "evidence_strength_assessment": "Substantively useful normative synthesis, but no empirical evidence that the recommendations improve security.", + "unresolved_questions": [ + "Which proposed mechanisms have interoperable implementations?", + "What formal relationship exists between the OpenID PDF and arXiv rendition?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-005", + "candidate_id": "candidate-BATCH-2026-001-045", + "source_classification": "preprint", + "canonical_title": "Identity Management for Agentic AI: The new frontier of authorization, authentication, and security for an AI agent world", + "source_version": "v1 submitted 2025-10-29", + "research_question": "Same core question as the OpenID white paper.", + "methodology_summary": "arXiv v1 rendition of substantially the same 33-page technical synthesis.", + "major_findings": [ + "The same identity, delegation, auditability, and lifecycle claims appear in this rendition." + ], + "author_interpretations": [ + "Recursive delegation should attenuate scope at each hop." + ], + "important_limitations": [ + "Preprint, not peer reviewed.", + "Not independent evidence from the OpenID rendition." + ], + "applicability_to_executive_roles": [ + "Useful for stable academic-style citation, with the same practical audience as the OpenID rendition." + ], + "geography_limits": "No jurisdiction-specific analysis.", + "industry_limits": "Cross-industry technology focus.", + "funding_and_sponsor_context": "Not reported.", + "relevance_to_books_and_media": "Stable preprint identifier for a central framework already represented by the OpenID PDF.", + "evidence_strength_assessment": "Do not double-count with the OpenID rendition.", + "unresolved_questions": [ + "Whether a later peer-reviewed version will supersede v1." + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-006", + "candidate_id": "candidate-BATCH-2026-001-050", + "source_classification": "peer-reviewed conference paper", + "canonical_title": "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents", + "source_version": "v3; updated after a Llama implementation bug fix and travel-suite update", + "research_question": "How reliably can tool-using agents complete tasks and resist prompt injection under controlled adversarial conditions?", + "methodology_summary": "Controlled benchmark of 97 tasks and 629 security cases across four simulated environments, with formal utility/security checks, seven listed model-agent configurations, attacks, defenses, and reported 95% confidence intervals.", + "major_findings": [ + "The benchmark exposes substantial utility and security tradeoffs.", + "Table 5 reports targeted attack success of 57.69% with no defense and 6.84% with tool filtering for the listed GPT-4o setup, with reported intervals." + ], + "author_interpretations": [ + "Stronger model designs and defenses are needed; static benchmark conclusions will age as attacks and defenses evolve." + ], + "important_limitations": [ + "Synthetic environments and versioned APIs limit external validity.", + "Raw numerators and interval-construction method are not reported in the reviewed table.", + "v3 replaced earlier results after a bug fix and travel-suite update." + ], + "applicability_to_executive_roles": [ + "CISOs can use it as evidence that authorization boundaries and tool isolation matter.", + "CTOs should treat rates as benchmark-specific, not production incident probabilities." + ], + "geography_limits": "No human geography; simulated applications.", + "industry_limits": "Four application simulations do not cover all enterprise domains.", + "funding_and_sponsor_context": "Individual support from armasuisse and SNSF is disclosed; no institution explicitly funded benchmark creation; industry affiliations are disclosed without a conflict statement.", + "relevance_to_books_and_media": "Supplies empirical context for normative recommendations about least privilege and tool isolation.", + "evidence_strength_assessment": "Strong internal benchmark evidence, moderate external validity.", + "unresolved_questions": [ + "How do results transfer to production agents and contemporary models?", + "How were confidence intervals constructed?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-007", + "candidate_id": "candidate-BATCH-2026-001-054", + "source_classification": "conference paper; acceptance marker verified in version, venue record not independently retrieved", + "canonical_title": "Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents", + "source_version": "v4; paper and arXiv page state accepted at ICLR 2025", + "research_question": "How do diverse attack and defense methods affect LLM agents across operational stages?", + "methodology_summary": "Synthetic benchmark spanning 10 scenarios, 400 tasks, more than 400 tools, 27 attack/defense methods, 13 LLM backbones, and seven metrics.", + "major_findings": [ + "The authors report a highest average attack success rate of 84.30% for a benchmark configuration.", + "Current defenses vary and often trade security against task performance." + ], + "author_interpretations": [ + "Agent security evaluation should cover the full operational pipeline rather than prompt injection alone." + ], + "important_limitations": [ + "No production population or human sample.", + "The headline average does not disclose a raw numerator or confidence interval.", + "Funding and conflicts are not reported." + ], + "applicability_to_executive_roles": [ + "CISOs can use the attack taxonomy to structure testing.", + "CTOs should avoid converting headline benchmark rates into incident forecasts." + ], + "geography_limits": "No human geography.", + "industry_limits": "Synthetic scenarios approximate industries but do not sample deployed systems.", + "funding_and_sponsor_context": "Not reported.", + "relevance_to_books_and_media": "Broadens the empirical attack surface beyond the narrower identity narratives in books and practitioner sources.", + "evidence_strength_assessment": "Useful comparative benchmark with strong breadth, but bounded external validity and incomplete uncertainty reporting.", + "unresolved_questions": [ + "How stable are rankings across model updates?", + "What exact cells and weights produce the 84.30% headline average?", + "Can venue acceptance be independently verified when OpenReview access is available?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-008", + "candidate_id": "candidate-BATCH-2026-001-058", + "source_classification": "preprint", + "canonical_title": "Teams of LLM Agents can Exploit Zero-Day Vulnerabilities", + "source_version": "v2 submitted 2025-03-30", + "research_question": "Can multi-agent planning improve autonomous exploitation of unknown real-world vulnerabilities?", + "methodology_summary": "HPTSA tested on 14 purposively selected, sandboxed, reproducible open-source web vulnerabilities, with manual trace verification and pass@1/pass@5 metrics.", + "major_findings": [ + "The authors report 42% pass@5 and 18% pass@1 for HPTSA with GPT-4 on the 14-vulnerability benchmark.", + "Open-source scanners and listed open-source models report 0% on this benchmark." + ], + "author_interpretations": [ + "Task-specific agents and hierarchical planning improve exploratory cyber capability." + ], + "important_limitations": [ + "Only 14 selected web vulnerabilities.", + "No confidence intervals.", + "Code and prompts are withheld, limiting replication.", + "Results are capability demonstrations, not estimates of incident prevalence." + ], + "applicability_to_executive_roles": [ + "CISOs can treat the work as capability evidence for threat modeling.", + "CTOs should not extrapolate rates to all vulnerabilities or current model versions." + ], + "geography_limits": "No human geography.", + "industry_limits": "Open-source web applications only.", + "funding_and_sponsor_context": "Funding and conflicts not reported; OpenAI's request to keep agents confidential and responsible disclosure are stated.", + "relevance_to_books_and_media": "Empirically qualifies broad warnings about autonomous cyber capability.", + "evidence_strength_assessment": "Moderate capability evidence with narrow sample and limited replicability.", + "unresolved_questions": [ + "Would results replicate on a broader blinded vulnerability sample?", + "How sensitive are outcomes to model and tool versions?" + ], + "review_status": "machine_checked_human_review_pending" + }, + { + "source_id": "source-BATCH-2026-005-009", + "candidate_id": "candidate-BATCH-2026-001-085", + "source_classification": "company white paper and policy analysis", + "canonical_title": "Practices for Governing Agentic AI Systems", + "source_version": "2023; PDF metadata created 2023-12-18", + "research_question": "Which lifecycle practices and accountability structures could reduce harms from agentic AI?", + "methodology_summary": "Conceptual policy analysis organized around lifecycle actors, seven direct governance practices, open questions, and indirect societal impacts.", + "major_findings": [ + "The authors propose evaluation, action-space constraints, default behaviors, legibility, monitoring, attributability, and interruptibility as building blocks.", + "They argue that at least one human legal entity should remain accountable for uncompensated direct harm." + ], + "author_interpretations": [ + "Agent attribution can support accountability in high-stakes interactions.", + "Interruptibility should extend to spawned subagents." + ], + "important_limitations": [ + "No empirical outcome evaluation.", + "Company-published and funding/conflicts are not reported.", + "Many recommendations are explicitly preliminary." + ], + "applicability_to_executive_roles": [ + "Boards and general counsel can use it to assign lifecycle accountability.", + "CISOs and CTOs can translate attribution and interruptibility into control questions." + ], + "geography_limits": "Global discussion without jurisdiction-specific legal analysis.", + "industry_limits": "Cross-industry and scenario-based.", + "funding_and_sponsor_context": "Official OpenAI-hosted white paper; separate funding, sponsor role, and conflicts not reported.", + "relevance_to_books_and_media": "An early governance frame that can be tested against later standards and benchmark evidence.", + "evidence_strength_assessment": "Influential normative analysis, not empirical proof.", + "unresolved_questions": [ + "Which practices measurably reduce harm?", + "How should privacy tradeoffs constrain attribution and monitoring?" + ], + "review_status": "machine_checked_human_review_pending" + } +] diff --git a/staging/batches/BATCH-2026-005/source-proposals.json b/staging/batches/BATCH-2026-005/source-proposals.json new file mode 100644 index 0000000..1ac4623 --- /dev/null +++ b/staging/batches/BATCH-2026-005/source-proposals.json @@ -0,0 +1,1189 @@ +[ + { + "source_id": "source-BATCH-2026-005-001", + "canonical_title": "Summary Analysis of Responses to the Request for Information Regarding Security Considerations for AI Agents", + "alternate_titles": [], + "source_type": "research_report", + "series_or_parent_source": "NIST AI 800-5", + "publisher": "National Institute of Standards and Technology", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-jared-riggs", + "person-BATCH-2026-005-maia-hamin", + "person-BATCH-2026-005-neil-perry", + "person-BATCH-2026-005-benjamin-edelman", + "person-BATCH-2026-005-peter-cihon" + ], + "institutional_author": "NIST Center for AI Standards and Innovation", + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2026-05-18", + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "United States", + "global issues discussed" + ], + "study_geography": [ + "United States", + "global issues discussed" + ], + "original_url": "https://www.nist.gov/publications/summary-analysis-responses-request-information-regarding-security-considerations-ai", + "canonical_url": "https://www.nist.gov/publications/summary-analysis-responses-request-information-regarding-security-considerations-ai", + "archived_url": null, + "embed_url": null, + "doi": null, + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Official NIST publication page and its analytical abstract reviewed; no full report file was exposed by the canonical page.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-005; contributes government report evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "low_for_accessible_version", + "independence": "government_publisher", + "bibliographic_stability": "high", + "evidence_strength": "moderate_for_describing_commenter_themes_not_population_prevalence", + "unresolved": "Full report body and response-level method were not available from the canonical page." + }, + "methodology_quality": { + "design": "qualitative synthesis of responses to a U.S. government request for information", + "transparency": "low_for_accessible_version", + "human_review_required": true + }, + "study_design": "qualitative synthesis of responses to a U.S. government request for information", + "sample": { + "size": "not reported on the canonical publication page", + "sampling_method": "self-selected RFI respondents; response recruitment and inclusion process not reported on the canonical page" + }, + "population": "organizations and individuals responding to the CAISI RFI; composition not reported", + "date_range": { + "fieldwork": "not reported", + "version": "official HTML publication page accessed 2026-08-15" + }, + "funding": "U.S. government publication; separate research funding not reported", + "sponsor": "National Institute of Standards and Technology", + "peer_review_status": "not applicable; government report", + "findings": [ + "The authors report broad agreement among commenters that agent security creates adoption barriers and requires adaptation of established cybersecurity practice." + ], + "limitations": [ + "No disclosed denominator, response composition, or qualitative coding method on the accessible page.", + "RFI respondents are self-selected and cannot be treated as a representative population." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_canonical_page_as_of_2026-08-15", + "content_hash": "32c07aeeb8f49694dd407080eab757c905dc4395871b52edd46a113208064a03", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.nist.gov/publications/summary-analysis-responses-request-information-regarding-security-considerations-ai", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "official HTML publication page accessed 2026-08-15", + "content_hash": "32c07aeeb8f49694dd407080eab757c905dc4395871b52edd46a113208064a03", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-002", + "canonical_title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile", + "alternate_titles": [], + "source_type": "public_policy_document", + "series_or_parent_source": "NIST AI 600-1", + "publisher": "National Institute of Standards and Technology", + "channel": null, + "speaker_ids": [], + "author_ids": [], + "institutional_author": "National Institute of Standards and Technology", + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": null, + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "United States", + "cross-sectoral global applicability" + ], + "study_geography": [ + "United States", + "cross-sectoral global applicability" + ], + "original_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", + "canonical_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", + "archived_url": null, + "embed_url": null, + "doi": "10.6028/NIST.AI.600-1", + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete official 64-page PDF reviewed; governance action tables and Appendix A limitation text visually inspected.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-007; contributes government standards profile evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "moderate_for_consensus_process", + "independence": "government_publisher", + "bibliographic_stability": "high_doi_resolved", + "evidence_strength": "strong_normative_reference_not_empirical_outcome_evidence", + "unresolved": "PDF metadata shows a 2025 modification date without a visible new edition statement." + }, + "methodology_quality": { + "design": "non-empirical cross-sectoral risk-management profile informed by multistakeholder public working-group feedback and public comments", + "transparency": "moderate_for_consensus_process", + "human_review_required": true + }, + "study_design": "non-empirical cross-sectoral risk-management profile informed by multistakeholder public working-group feedback and public comments", + "sample": { + "size": "not applicable for a normative profile; contributor and commenter counts not reported", + "sampling_method": "open multistakeholder process; selection details not reported" + }, + "population": "organizations designing, developing, deploying, or using generative AI", + "date_range": { + "fieldwork": "not reported", + "version": "July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded" + }, + "funding": "U.S. Department of Commerce/NIST publication; separate research funding not reported", + "sponsor": "National Institute of Standards and Technology", + "peer_review_status": "NIST Editorial Review Board; not a peer-reviewed academic study", + "findings": [ + "The profile organizes suggested actions across governance, mapping, measurement, and management functions.", + "It explicitly warns that pre-deployment tests and benchmark results may not generalize to real-world contexts." + ], + "limitations": [ + "No quantitative sample or outcome evaluation.", + "Public-input synthesis method and denominator are not disclosed." + ], + "correction_ids": [], + "retraction_status": "none_identified_in_pdf_or_official_doi_resolution_as_of_2026-08-15", + "content_hash": "6e73620ab6b64e90ef2c04bf0e0d6246185a2f4b1b13cab0df494496cff89b6a", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded", + "content_hash": "6e73620ab6b64e90ef2c04bf0e0d6246185a2f4b1b13cab0df494496cff89b6a", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-003", + "canonical_title": "AI Agent Authentication and Authorization", + "alternate_titles": [], + "source_type": "working_paper", + "series_or_parent_source": "draft-klrc-aiagent-auth-02", + "publisher": "Internet Engineering Task Force", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-pieter-kasselman", + "person-BATCH-2026-005-jeff-lombardo", + "person-BATCH-2026-005-yaron-rosomakho", + "person-BATCH-2026-005-brian-campbell", + "person-BATCH-2026-005-nick-steele", + "person-BATCH-2026-005-aaron-parecki" + ], + "institutional_author": "IETF Internet-Draft authors", + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2026-06-01", + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "global standards context" + ], + "study_geography": [ + "global standards context" + ], + "original_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "canonical_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "archived_url": null, + "embed_url": null, + "doi": null, + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete official IETF HTML rendition reviewed with stable section and paragraph locators.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-009; contributes standards document; Internet-Draft evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "not_applicable_non_empirical", + "independence": "multi_vendor_author_group", + "bibliographic_stability": "medium_due_to_expiring_draft", + "evidence_strength": "strong_for_current_proposal_weak_for_interoperability_outcomes", + "unresolved": "Future revisions may change normative language or identifier choices." + }, + "methodology_quality": { + "design": "non-empirical technical standards proposal", + "transparency": "not_applicable_non_empirical", + "human_review_required": true + }, + "study_design": "non-empirical technical standards proposal", + "sample": { + "size": "not applicable", + "sampling_method": "not applicable" + }, + "population": "AI-agent workloads and the tools, services, models, systems, and users that interact with them", + "date_range": { + "fieldwork": "not applicable", + "version": "revision 02; expires 2026-12-03" + }, + "funding": "not reported", + "sponsor": "not reported", + "peer_review_status": "working document; not an RFC and not peer reviewed", + "findings": [ + "The draft models an agent as a workload that requires a stable identifier and cryptographically bound credentials.", + "It proposes preserving user or system delegation context in authorization decisions and audit trails." + ], + "limitations": [ + "Revision 02 is an expiring work in progress.", + "No interoperability tests or deployment outcomes are reported." + ], + "correction_ids": [], + "retraction_status": "active_work_in_progress_not_retracted_as_of_2026-08-15", + "content_hash": "e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://www.ietf.org/archive/id/draft-klrc-aiagent-auth-02.html", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "revision 02; expires 2026-12-03", + "content_hash": "e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-004", + "canonical_title": "Identity Management for Agentic AI", + "alternate_titles": [], + "source_type": "research_report", + "series_or_parent_source": "official OpenID Foundation PDF", + "publisher": "OpenID Foundation", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-tobin-south", + "person-BATCH-2026-005-subramanya-nagabhushanaradhya", + "person-BATCH-2026-005-ayesha-dissanayaka", + "person-BATCH-2026-005-sarah-cecchetti", + "person-BATCH-2026-005-george-fletcher", + "person-BATCH-2026-005-victor-lu", + "person-BATCH-2026-005-aldo-pietropaolo", + "person-BATCH-2026-005-dean-h-saxe", + "person-BATCH-2026-005-jeff-lombardo", + "person-BATCH-2026-005-abhishek-shivalingaiah", + "person-BATCH-2026-005-stan-bounev", + "person-BATCH-2026-005-alex-keisner", + "person-BATCH-2026-005-andor-kesselman", + "person-BATCH-2026-005-zack-proser", + "person-BATCH-2026-005-ginny-fahs", + "person-BATCH-2026-005-andrew-bunyea", + "person-BATCH-2026-005-ben-moskowitz", + "person-BATCH-2026-005-atul-tulshibagwale", + "person-BATCH-2026-005-dazza-greenwood", + "person-BATCH-2026-005-jiaxin-pei", + "person-BATCH-2026-005-alex-pentland" + ], + "institutional_author": "OpenID Foundation", + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": null, + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "global technology standards context" + ], + "study_geography": [ + "global technology standards context" + ], + "original_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "canonical_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "archived_url": null, + "embed_url": null, + "doi": null, + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete official 33-page PDF reviewed; identity, delegation, auditability, and use-case sections inspected visually.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-027; contributes nonprofit/company white paper evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_white_paper_but_source_selection_not_reported", + "independence": "standards_community_publisher", + "bibliographic_stability": "high_official_pdf", + "evidence_strength": "moderate_normative_synthesis_not_empirical", + "unresolved": "Substantially the same intellectual work appears as arXiv:2510.25819v1." + }, + "methodology_quality": { + "design": "non-empirical technical white paper and standards synthesis", + "transparency": "appropriate_for_white_paper_but_source_selection_not_reported", + "human_review_required": true + }, + "study_design": "non-empirical technical white paper and standards synthesis", + "sample": { + "size": "not applicable", + "sampling_method": "not applicable" + }, + "population": "AI-agent systems interacting with protected resources and other agents", + "date_range": { + "fieldwork": "not applicable", + "version": "October 2025" + }, + "funding": "not reported", + "sponsor": "OpenID Foundation publisher; separate sponsorship not reported", + "peer_review_status": "not reported; not treated as peer reviewed", + "findings": [ + "The authors argue that current mechanisms are strongest for synchronous, single-trust-domain patterns and weaker for cross-domain or recursive delegation.", + "The paper distinguishes agent identity from user identity and calls for lifecycle management and enriched audit trails." + ], + "limitations": [ + "No implementation benchmark or comparative test.", + "Formal source-selection, funding, conflicts, and review process are not reported." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_official_pdf_or_publisher_url_as_of_2026-08-15", + "content_hash": "e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "October 2025", + "content_hash": "e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-005", + "canonical_title": "Identity Management for Agentic AI: The new frontier of authorization, authentication, and security for an AI agent world", + "alternate_titles": [], + "source_type": "academic_paper", + "series_or_parent_source": "arXiv:2510.25819v1", + "publisher": "arXiv", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-tobin-south", + "person-BATCH-2026-005-subramanya-nagabhushanaradhya", + "person-BATCH-2026-005-ayesha-dissanayaka", + "person-BATCH-2026-005-sarah-cecchetti", + "person-BATCH-2026-005-george-fletcher", + "person-BATCH-2026-005-victor-lu", + "person-BATCH-2026-005-aldo-pietropaolo", + "person-BATCH-2026-005-dean-h-saxe", + "person-BATCH-2026-005-jeff-lombardo", + "person-BATCH-2026-005-abhishek-shivalingaiah", + "person-BATCH-2026-005-stan-bounev", + "person-BATCH-2026-005-alex-keisner", + "person-BATCH-2026-005-andor-kesselman", + "person-BATCH-2026-005-zack-proser", + "person-BATCH-2026-005-ginny-fahs", + "person-BATCH-2026-005-andrew-bunyea", + "person-BATCH-2026-005-ben-moskowitz", + "person-BATCH-2026-005-atul-tulshibagwale", + "person-BATCH-2026-005-dazza-greenwood", + "person-BATCH-2026-005-jiaxin-pei", + "person-BATCH-2026-005-alex-pentland" + ], + "institutional_author": null, + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2025-10-29", + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "global technology standards context" + ], + "study_geography": [ + "global technology standards context" + ], + "original_url": "https://arxiv.org/abs/2510.25819", + "canonical_url": "https://arxiv.org/abs/2510.25819", + "archived_url": null, + "embed_url": null, + "doi": "10.48550/arXiv.2510.25819", + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete arXiv v1 PDF reviewed and compared with the OpenID rendition; the content is substantially the same intellectual work with rendition-level differences.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-045; contributes preprint evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_preprint_synthesis", + "independence": "multi_author_preprint", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_normative_synthesis_not_empirical", + "unresolved": "Treat as a related rendition, not an independent evidentiary replication of the OpenID report." + }, + "methodology_quality": { + "design": "non-empirical technical preprint and standards synthesis", + "transparency": "appropriate_for_preprint_synthesis", + "human_review_required": true + }, + "study_design": "non-empirical technical preprint and standards synthesis", + "sample": { + "size": "not applicable", + "sampling_method": "not applicable" + }, + "population": "AI-agent systems", + "date_range": { + "fieldwork": "not applicable", + "version": "v1 submitted 2025-10-29" + }, + "funding": "not reported", + "sponsor": "not reported", + "peer_review_status": "not peer reviewed; preprint", + "findings": [ + "The same identity, delegation, auditability, and lifecycle claims appear in this rendition." + ], + "limitations": [ + "Preprint, not peer reviewed.", + "Not independent evidence from the OpenID rendition." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "content_hash": "b019c9420565afe6a3f81fcd8e9b3375f546003cecab93ecfa25d965b2f68bb2", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2510.25819", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "v1 submitted 2025-10-29", + "content_hash": "b019c9420565afe6a3f81fcd8e9b3375f546003cecab93ecfa25d965b2f68bb2", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-006", + "canonical_title": "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents", + "alternate_titles": [], + "source_type": "academic_paper", + "series_or_parent_source": "arXiv:2406.13352v3; NeurIPS 2024 proceedings record", + "publisher": "NeurIPS 2024 Datasets and Benchmarks Track", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-edoardo-debenedetti", + "person-BATCH-2026-005-jie-zhang", + "person-BATCH-2026-005-mislav-balunovic", + "person-BATCH-2026-005-luca-beurer-kellner", + "person-BATCH-2026-005-marc-fischer", + "person-BATCH-2026-005-florian-tramer" + ], + "institutional_author": null, + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2024-06-19", + "updated_at": "2024-11-24", + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "synthetic environments; no human geography" + ], + "study_geography": [ + "synthetic environments; no human geography" + ], + "original_url": "https://arxiv.org/abs/2406.13352", + "canonical_url": "https://arxiv.org/abs/2406.13352", + "archived_url": null, + "embed_url": null, + "doi": "10.48550/arXiv.2406.13352", + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete 26-page v3 conference paper reviewed; benchmark composition, figures, Tables 1 and 3-5, confidence intervals, limitations, and funding inspected visually.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-050; contributes peer-reviewed conference paper evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "high", + "independence": "academic_with_disclosed_industry_affiliations", + "bibliographic_stability": "high_arxiv_and_neurips_records", + "evidence_strength": "strong_for_benchmark_conditions_not_real_world_incidence", + "unresolved": "Confidence-interval construction and raw row numerators are not stated in the reviewed tables." + }, + "methodology_quality": { + "design": "controlled benchmark experiments in stateful synthetic tool environments", + "transparency": "high", + "human_review_required": true + }, + "study_design": "controlled benchmark experiments in stateful synthetic tool environments", + "sample": { + "size": "97 user tasks; 27 injection tasks; 629 security test cases; seven listed agent/model configurations in Table 3", + "sampling_method": "researcher-curated realistic tasks and synthetic data; compatible task cross-product" + }, + "population": "LLM agents executing tools over simulated workspace, Slack, travel, and banking environments", + "date_range": { + "fieldwork": "not reported; v3 results reflect the paper's 2024 model/API versions", + "version": "v3; updated after a Llama implementation bug fix and travel-suite update" + }, + "funding": "Edoardo Debenedetti supported by armasuisse Science and Technology; Jie Zhang funded by Swiss National Science Foundation grant 214838; benchmark funding statement says no institution explicitly funded benchmark creation", + "sponsor": "no benchmark sponsor reported", + "peer_review_status": "verified conference publication in official NeurIPS 2024 proceedings", + "findings": [ + "The benchmark exposes substantial utility and security tradeoffs.", + "Table 5 reports targeted attack success of 57.69% with no defense and 6.84% with tool filtering for the listed GPT-4o setup, with reported intervals." + ], + "limitations": [ + "Synthetic environments and versioned APIs limit external validity.", + "Raw numerators and interval-construction method are not reported in the reviewed table.", + "v3 replaced earlier results after a bug fix and travel-suite update." + ], + "correction_ids": [ + "correction-BATCH-2026-005-001" + ], + "retraction_status": "no_retraction_identified;_v3_documents_bug_fix_update", + "content_hash": "349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.13352", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "v3; updated after a Llama implementation bug fix and travel-suite update", + "content_hash": "349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-007", + "canonical_title": "Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents", + "alternate_titles": [], + "source_type": "academic_paper", + "series_or_parent_source": "arXiv:2410.02644v4", + "publisher": "ICLR 2025 / arXiv", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-hanrong-zhang", + "person-BATCH-2026-005-jingyuan-huang", + "person-BATCH-2026-005-kai-mei", + "person-BATCH-2026-005-yifei-yao", + "person-BATCH-2026-005-zhenting-wang", + "person-BATCH-2026-005-chenlu-zhan", + "person-BATCH-2026-005-hongwei-wang", + "person-BATCH-2026-005-yongfeng-zhang" + ], + "institutional_author": null, + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2024-10-03", + "updated_at": "2025-05-30", + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "synthetic benchmark; no human geography" + ], + "study_geography": [ + "synthetic benchmark; no human geography" + ], + "original_url": "https://arxiv.org/abs/2410.02644", + "canonical_url": "https://arxiv.org/abs/2410.02644", + "archived_url": null, + "embed_url": null, + "doi": "10.48550/arXiv.2410.02644", + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete 36-page v4 paper reviewed; attack/defense definitions, scenario tables, metric table, result tables, and reproducibility appendix inspected visually.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-054; contributes conference paper; acceptance marker verified in version, venue record not independently retrieved evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "high_for_benchmark_structure", + "independence": "academic", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_to_strong_for_benchmark_conditions", + "unresolved": "Venue acceptance is author/arXiv-reported; OpenReview verification was unavailable in this run. Headline average lacks uncertainty and a compact denominator statement." + }, + "methodology_quality": { + "design": "multi-scenario controlled security benchmark", + "transparency": "high_for_benchmark_structure", + "human_review_required": true + }, + "study_design": "multi-scenario controlled security benchmark", + "sample": { + "size": "400 tasks; 10 scenarios; 10 agents; more than 400 tools; 27 attack/defense methods; 13 LLM backbones", + "sampling_method": "researcher-constructed scenarios, tasks, tools, and attacks; selection procedure not probabilistic" + }, + "population": "LLM-agent configurations under synthetic attacks and defenses", + "date_range": { + "fieldwork": "not reported; model versions correspond to experiments before v4 dated 2025-05-30", + "version": "v4; paper and arXiv page state accepted at ICLR 2025" + }, + "funding": "not reported", + "sponsor": "not reported", + "peer_review_status": "conference acceptance reported in the paper and arXiv metadata; independent OpenReview record retrieval returned access denied, so peer-review verification remains qualified", + "findings": [ + "The authors report a highest average attack success rate of 84.30% for a benchmark configuration.", + "Current defenses vary and often trade security against task performance." + ], + "limitations": [ + "No production population or human sample.", + "The headline average does not disclose a raw numerator or confidence interval.", + "Funding and conflicts are not reported." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "content_hash": "e20155df01b3a1f6c0a947c4e9e4871957cff2e507939f7ea21a5fe967d84504", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2410.02644", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "v4; paper and arXiv page state accepted at ICLR 2025", + "content_hash": "e20155df01b3a1f6c0a947c4e9e4871957cff2e507939f7ea21a5fe967d84504", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-008", + "canonical_title": "Teams of LLM Agents can Exploit Zero-Day Vulnerabilities", + "alternate_titles": [], + "source_type": "academic_paper", + "series_or_parent_source": "arXiv:2406.01637v2", + "publisher": "arXiv", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-yuxuan-zhu", + "person-BATCH-2026-005-antony-kellermann", + "person-BATCH-2026-005-akul-gupta", + "person-BATCH-2026-005-philip-li", + "person-BATCH-2026-005-richard-fang", + "person-BATCH-2026-005-rohan-bindu", + "person-BATCH-2026-005-daniel-kang" + ], + "institutional_author": null, + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": "2024-06-02", + "updated_at": "2025-03-30", + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "synthetic sandbox; no human geography" + ], + "study_geography": [ + "synthetic sandbox; no human geography" + ], + "original_url": "https://arxiv.org/abs/2406.01637", + "canonical_url": "https://arxiv.org/abs/2406.01637", + "archived_url": null, + "embed_url": null, + "doi": "10.48550/arXiv.2406.01637", + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete 10-page v2 preprint reviewed; Tables 1-2, Figures 2-3, methods, results, and limitation text inspected visually.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-058; contributes preprint evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "moderate", + "independence": "academic_with_platform_coordination_disclosed", + "bibliographic_stability": "high_arxiv_doi_resolved", + "evidence_strength": "moderate_for_selected_sandbox_cases", + "unresolved": "Nonrelease of code/prompts limits replication; funding and conflict disclosures are absent." + }, + "methodology_quality": { + "design": "controlled cybersecurity benchmark experiment in sandboxed reproductions", + "transparency": "moderate", + "human_review_required": true + }, + "study_design": "controlled cybersecurity benchmark experiment in sandboxed reproductions", + "sample": { + "size": "14 vulnerabilities; pass@1 and pass@5 trials; exact total run count not summarized", + "sampling_method": "purposive selection of recent, reproducible web vulnerabilities with clear success triggers and manual exploitability" + }, + "population": "reproducible open-source web vulnerabilities; not all vulnerabilities or deployed systems", + "date_range": { + "fieldwork": "not reported", + "version": "v2 submitted 2025-03-30" + }, + "funding": "not reported", + "sponsor": "not reported; authors state OpenAI requested code and prompts remain confidential", + "peer_review_status": "not peer reviewed; preprint", + "findings": [ + "The authors report 42% pass@5 and 18% pass@1 for HPTSA with GPT-4 on the 14-vulnerability benchmark.", + "Open-source scanners and listed open-source models report 0% on this benchmark." + ], + "limitations": [ + "Only 14 selected web vulnerabilities.", + "No confidence intervals.", + "Code and prompts are withheld, limiting replication.", + "Results are capability demonstrations, not estimates of incident prevalence." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_arxiv_version_history_as_of_2026-08-15", + "content_hash": "f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://arxiv.org/abs/2406.01637", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "v2 submitted 2025-03-30", + "content_hash": "f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + }, + { + "source_id": "source-BATCH-2026-005-009", + "canonical_title": "Practices for Governing Agentic AI Systems", + "alternate_titles": [], + "source_type": "research_report", + "series_or_parent_source": "official OpenAI PDF", + "publisher": "OpenAI", + "channel": null, + "speaker_ids": [], + "author_ids": [ + "person-BATCH-2026-005-yonadav-shavit", + "person-BATCH-2026-005-sandhini-agarwal", + "person-BATCH-2026-005-miles-brundage", + "person-BATCH-2026-005-steven-adler", + "person-BATCH-2026-005-cullen-o-keefe", + "person-BATCH-2026-005-rosie-campbell", + "person-BATCH-2026-005-teddy-lee", + "person-BATCH-2026-005-pamela-mishkin", + "person-BATCH-2026-005-tyna-eloundou", + "person-BATCH-2026-005-alan-hickey", + "person-BATCH-2026-005-katarina-slama", + "person-BATCH-2026-005-lama-ahmad", + "person-BATCH-2026-005-paul-mcmillan", + "person-BATCH-2026-005-alex-beutel", + "person-BATCH-2026-005-alexandre-passos", + "person-BATCH-2026-005-david-g-robinson" + ], + "institutional_author": null, + "organization_references": [], + "recorded_at": null, + "event_date": null, + "published_at": null, + "updated_at": null, + "duration_seconds": null, + "language": "lang-en", + "translated_title": null, + "translation_method": null, + "geography_of_speaker": [], + "geography_of_organization": [], + "geography_discussed": [ + "global policy context" + ], + "study_geography": [ + "global policy context" + ], + "original_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "canonical_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "archived_url": null, + "embed_url": null, + "doi": null, + "canonical_identity_status": "verified", + "repost_status": "original_or_authoritative_rendition", + "original_source_id": null, + "rights_status": "link_and_paraphrase", + "ownership_status": "third_party", + "relationship_to_off": "none_identified; Executive AI Research snapshot contains no matching production records", + "transcript_status": null, + "transcript_source": null, + "transcript_republication_permission": null, + "chapter_markers": [], + "analysis_basis": "Complete official 23-page PDF reviewed; definition, accountability, monitoring, attribution, interruptibility, indirect impacts, and acknowledgements inspected visually.", + "topics": [ + "topic-ai-agents", + "topic-ai-governance", + "topic-cybersecurity", + "topic-non-human-identity" + ], + "executive_roles": [ + "role-ciso", + "role-cio", + "role-cto", + "role-general-counsel", + "role-board-director" + ], + "original_abstract": null, + "inclusion_rationale": "Accepted discovery candidate candidate-BATCH-2026-001-085; contributes company white paper and policy analysis evidence or normative context to governed AI-agent identity.", + "source_quality_dimensions": { + "attribution_strength": "high", + "methodological_transparency": "appropriate_for_policy_analysis", + "independence": "company_published", + "bibliographic_stability": "high_official_pdf", + "evidence_strength": "moderate_normative_policy_analysis_not_empirical", + "unresolved": "Publication day, funding, affiliations, conflicts, and external review status are not reported in the PDF." + }, + "methodology_quality": { + "design": "non-empirical company white paper and policy analysis", + "transparency": "appropriate_for_policy_analysis", + "human_review_required": true + }, + "study_design": "non-empirical company white paper and policy analysis", + "sample": { + "size": "not applicable", + "sampling_method": "not applicable" + }, + "population": "model developers, system deployers, users, and third parties in agentic-system lifecycles", + "date_range": { + "fieldwork": "not applicable", + "version": "2023; PDF metadata created 2023-12-18" + }, + "funding": "not reported", + "sponsor": "OpenAI publisher; separate sponsor involvement not reported", + "peer_review_status": "not reported; not treated as peer reviewed", + "findings": [ + "The authors propose evaluation, action-space constraints, default behaviors, legibility, monitoring, attributability, and interruptibility as building blocks.", + "They argue that at least one human legal entity should remain accountable for uncompensated direct harm." + ], + "limitations": [ + "No empirical outcome evaluation.", + "Company-published and funding/conflicts are not reported.", + "Many recommendations are explicitly preliminary." + ], + "correction_ids": [], + "retraction_status": "none_identified_on_official_pdf_url_as_of_2026-08-15", + "content_hash": "22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c", + "accessed_at": "2026-08-15", + "verification_status": "metadata_content_and_version_machine_verified_human_review_pending", + "source_depth": "deeply_analyzed", + "publication_status": "staging_only", + "workflow_status": "content_reviewed", + "machine_review_status": "machine_checked", + "human_review_status": "pending", + "reviewed_by": null, + "reviewed_at": null, + "provenance": [ + { + "source_url": "https://cdn.openai.com/papers/practices-for-governing-agentic-ai-systems.pdf", + "accessed_at": "2026-08-15", + "retrieval_method": "official source retrieval and machine-assisted review", + "exact_locator": "2023; PDF metadata created 2023-12-18", + "content_hash": "22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c", + "batch_id": "BATCH-2026-005", + "prompt_id": "OEII-EVIDENCE-RESEARCH", + "prompt_version": "2.0", + "notes": "Human review pending." + } + ], + "revision_history": [] + } +] diff --git a/staging/batches/BATCH-2026-005/source-version-review.csv b/staging/batches/BATCH-2026-005/source-version-review.csv new file mode 100644 index 0000000..baff36f --- /dev/null +++ b/staging/batches/BATCH-2026-005/source-version-review.csv @@ -0,0 +1,10 @@ +source_id,permanent_identifier,version_reviewed,content_hash_sha256,version_status,version_caveat,review_status +source-BATCH-2026-005-001,NIST AI 800-5,official HTML publication page accessed 2026-08-15,32c07aeeb8f49694dd407080eab757c905dc4395871b52edd46a113208064a03,exact reviewed version recorded,Full report body and response-level method were not available from the canonical page.,machine_checked_human_review_pending +source-BATCH-2026-005-002,NIST AI 600-1,July 2024; Editorial Review Board approval 2024-07-25; retrieved PDF SHA-256 recorded,6e73620ab6b64e90ef2c04bf0e0d6246185a2f4b1b13cab0df494496cff89b6a,exact reviewed version recorded,PDF metadata shows a 2025 modification date without a visible new edition statement.,machine_checked_human_review_pending +source-BATCH-2026-005-003,draft-klrc-aiagent-auth-02,revision 02; expires 2026-12-03,e34f335c56546a6d91d88d43159c9f80d4dcfbbad020d41c31607fc56848befa,exact reviewed version recorded,Future revisions may change normative language or identifier choices.,machine_checked_human_review_pending +source-BATCH-2026-005-004,official OpenID Foundation PDF,October 2025,e6a0e5909fcb2486568180f69d511ae2af8d20a1bfb3028b39b3d1072846f4f3,exact reviewed version recorded,Substantially the same intellectual work appears as arXiv:2510.25819v1.,machine_checked_human_review_pending +source-BATCH-2026-005-005,arXiv:2510.25819v1,v1 submitted 2025-10-29,b019c9420565afe6a3f81fcd8e9b3375f546003cecab93ecfa25d965b2f68bb2,related rendition retained without independent-evidence counting,"Treat as a related rendition, not an independent evidentiary replication of the OpenID report.",machine_checked_human_review_pending +source-BATCH-2026-005-006,arXiv:2406.13352v3; NeurIPS 2024 proceedings record,v3; updated after a Llama implementation bug fix and travel-suite update,349884fffbf43282591c5accffd57bb651632f38e412fe303114941bdd111b05,current retrieved arXiv v3; earlier versions superseded for extracted statistics,Confidence-interval construction and raw row numerators are not stated in the reviewed tables.,machine_checked_human_review_pending +source-BATCH-2026-005-007,arXiv:2410.02644v4,v4; paper and arXiv page state accepted at ICLR 2025,e20155df01b3a1f6c0a947c4e9e4871957cff2e507939f7ea21a5fe967d84504,exact reviewed version recorded,Venue acceptance is author/arXiv-reported; OpenReview verification was unavailable in this run. Headline average lacks uncertainty and a compact denominator statement.,machine_checked_human_review_pending +source-BATCH-2026-005-008,arXiv:2406.01637v2,v2 submitted 2025-03-30,f4fc06040f0856d99d2009042b4c6fa0c01206da8ca61cb39a94d52b9867cf71,exact reviewed version recorded,Nonrelease of code/prompts limits replication; funding and conflict disclosures are absent.,machine_checked_human_review_pending +source-BATCH-2026-005-009,official OpenAI PDF,2023; PDF metadata created 2023-12-18,22b3a8607ed781a848b82b0bfe8e638b16cd027aac90a19b2aecf236063d0e7c,exact reviewed version recorded,"Publication day, funding, affiliations, conflicts, and external review status are not reported in the PDF.",machine_checked_human_review_pending diff --git a/staging/batches/BATCH-2026-005/statistics-review.csv b/staging/batches/BATCH-2026-005/statistics-review.csv new file mode 100644 index 0000000..70e4784 --- /dev/null +++ b/staging/batches/BATCH-2026-005/statistics-review.csv @@ -0,0 +1,8 @@ +statistic_id,statement_id,source_id,claim,numerator,denominator,sample,population,geography,fieldwork_dates,locator,unit,confidence_interval,methodological_caveat,reported_or_calculated,review_status +stat-BATCH-2026-005-001,statement-BATCH-2026-005-012,source-BATCH-2026-005-006,97 user tasks,not applicable,not applicable,97 curated user tasks,simulated tool-using agent tasks,not applicable,not reported; v3 dated 2024-11-24,"arXiv:2406.13352v3 PDF p. 1, Abstract; Table 1 p. 6",task count,not applicable,Curated synthetic tasks; not a production sample.,directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-002,statement-BATCH-2026-005-012,source-BATCH-2026-005-006,629 security test cases,not applicable,not applicable,629 compatible cross-product security cases,AgentDojo v3 task and injection combinations,not applicable,not reported; v3 dated 2024-11-24,"arXiv:2406.13352v3 PDF p. 1, Abstract",security-case count,not applicable,"Benchmark case count, not independent real-world incidents.",directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-003,statement-BATCH-2026-005-013,source-BATCH-2026-005-006,57.69% targeted attack success with no defense,not reported,629 security cases in full suite; row-specific usable-case denominator not restated,AgentDojo v3 GPT-4o security cases,GPT-4o under listed no-defense benchmark configuration,not applicable,not reported,"arXiv:2406.13352v3 PDF p. 20, Table 5",percent of security cases,95% CI shown as +/-3.9 percentage points,Raw numerator and interval construction not reported; synthetic versioned benchmark.,directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-004,statement-BATCH-2026-005-013,source-BATCH-2026-005-006,6.84% targeted attack success with tool filtering,not reported,629 security cases in full suite; row-specific usable-case denominator not restated,AgentDojo v3 GPT-4o security cases,GPT-4o under listed tool-filter defense,not applicable,not reported,"arXiv:2406.13352v3 PDF p. 20, Table 5",percent of security cases,95% CI shown as +/-2.0 percentage points,Raw numerator and interval construction not reported; defense may also change utility.,directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-005,statement-BATCH-2026-005-016,source-BATCH-2026-005-007,84.30% highest average attack success rate,not reported,not reported as a compact n for the headline average,"ASB v4: 400 tasks, 27 attack/defense methods, 13 LLM backbones",evaluated ASB configurations,not applicable,not reported,"arXiv:2410.02644v4 PDF p. 1, Abstract",percent,not reported,Headline maximum average across benchmark configurations; weighting and exact cell set require table-level reconstruction.,directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-006,statement-BATCH-2026-005-018,source-BATCH-2026-005-008,42% pass@5,not reported,14 selected vulnerabilities,14 sandboxed open-source web vulnerabilities,purposively selected reproducible web vulnerabilities,not applicable,not reported,"arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2",percent of vulnerabilities with at least one success in up to five attempts,not reported,Rounded percentage; small non-random sample; raw success count not stated.,directly reported,machine_checked_human_review_pending +stat-BATCH-2026-005-007,statement-BATCH-2026-005-018,source-BATCH-2026-005-008,18% pass@1,not reported,14 selected vulnerabilities,14 sandboxed open-source web vulnerabilities,purposively selected reproducible web vulnerabilities,not applicable,not reported,"arXiv:2406.01637v2 PDF p. 5, Figure 2 and Section 5.2",percent of vulnerabilities successful on first attempt,not reported,Rounded percentage; small non-random sample; raw success count not stated.,directly reported,machine_checked_human_review_pending diff --git a/staging/batches/BATCH-2026-005/validation-results.md b/staging/batches/BATCH-2026-005/validation-results.md new file mode 100644 index 0000000..4bab800 --- /dev/null +++ b/staging/batches/BATCH-2026-005/validation-results.md @@ -0,0 +1,24 @@ +# Validation results — BATCH-2026-005 + +Completed 2026-08-15. All records remain staging-only and require named human review. + +## Passed + +- Source schema: 9 of 9 source proposals. +- Statement schema: 22 of 22 evidence statements. +- DOI validation: five asserted DOIs resolved by HTTPS GET; no guessed DOI was assigned to NIST AI 800-5. +- Source-version validation: nine exact versions have SHA-256 hashes and stable identifiers or official URLs. +- Duplicate detection: the OpenID PDF and arXiv v1 are retained as related renditions and are blocked from independent-evidence double counting. +- Methodology fields: all 23 required fields are populated for all nine sources; missing information is recorded as `not reported` or explicitly not applicable. +- Sample context: the response-synthesis sample gap is visible; all benchmark samples and populations are scoped. +- Statistics: seven rows include numerator status, denominator, sample, population, geography, dates, locator, unit, interval status, caveat, and reported/calculated status. +- PDF and locator review: all 22 statement locators passed; relevant tables, figures, axes, labels, footnotes, and versioned page numbers were visually checked. +- Funding and conflicts: one complete row per source; only one source contains explicit funding disclosures and no source contains a formal conflict statement. +- Corrections and retractions: AgentDojo v3 bug-fix update and NIST PDF metadata anomaly are flagged; no retractions identified. +- OFF crosswalk: nine explicit unmatched records against pinned Executive AI Research commit `d205a6b2e6f4`; no OFF ownership asserted. +- Rights: no PDFs, full text, tables, figures, or direct quotations staged. +- Analytical abstracts and separate evidence-quality dimensions: complete for all nine sources. + +## Required shell gates + +The repository data/provenance validators, staging-leak search, whitespace check, and targeted project tests are run separately and recorded in the final manifest/PR checks. From 8d0b721f7bf7d753ce3fdd6ce9dd2d05d01ab3ae Mon Sep 17 00:00:00 2001 From: murraylovecode Date: Sat, 15 Aug 2026 07:37:29 -0700 Subject: [PATCH 2/2] Record evidence research draft pull request --- staging/batches/BATCH-2026-005/manifest.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/staging/batches/BATCH-2026-005/manifest.yml b/staging/batches/BATCH-2026-005/manifest.yml index 2464626..16b3769 100644 --- a/staging/batches/BATCH-2026-005/manifest.yml +++ b/staging/batches/BATCH-2026-005/manifest.yml @@ -4,9 +4,10 @@ prompt_version: "2.0" topic: Governed identities for AI agents topic_slug: governed-agent-identities branch: research/evidence-governed-agent-identities-BATCH-2026-005 -pull_request: null +pull_request: https://github.com/OpenFutureForum/executive-intelligence-index/pull/6 base_branch: research/discovery-governed-agent-identities-BATCH-2026-001 execution_date: 2026-08-15 +execution_completed_at: 2026-08-15T13:31:00Z agent_or_researcher: OpenAI Codex; machine-assisted evidence review; human review pending model_disclosure: AI-assisted metadata verification, methodology coding, statistical-context review, PDF visual inspection, and statement drafting; no human approval or publication approval occurred. accepted_evidence_source_candidate_ids: