{
  "benchmark": "state-of-agent-security",
  "edition": "2026-Q3",
  "published": "2026-09-20",
  "dataset": {
    "provenance": "internal dogfooding — Orithos scanning its own demo and test agents; not a customer engagement",
    "label": "internal dogfooding data",
    "period": [
      "2026-05-19",
      "2026-08-15"
    ],
    "agents": 6,
    "scans": 126,
    "scans_completed": 80,
    "scans_failed": 43,
    "scans_aborted": 3,
    "evaluated_probe_outcomes": 2000
  },
  "summary": {
    "probes_blocked_pct": 87.8,
    "real_findings": 243,
    "critical_or_high": 175,
    "critical_or_high_pct": 72.0,
    "findings_at_100pct_attack_success": 74,
    "critical_findings_at_100pct_attack_success": 38,
    "distinct_categories_with_findings": 46,
    "framework_control_tags": 86,
    "framework_families": 19
  },
  "severity_distribution": {
    "critical": 82,
    "high": 93,
    "medium": 41,
    "low": 11,
    "info": 16
  },
  "category_distribution": {
    "SAFETY": 85,
    "tool_abuse": 17,
    "prompt_leakage": 13,
    "policy-violation": 10,
    "instruction_hierarchy_violation": 9,
    "general": 9,
    "mcp_reported": 9,
    "financial_advice_violation": 5,
    "direct_injection": 5,
    "credential_harvesting": 5,
    "membership_inference": 5,
    "prompt_injection": 4,
    "unauthorized_exfiltration": 3,
    "bias_discrimination": 3,
    "legal_advice_violation": 3,
    "output_manipulation": 2,
    "overreliance": 2,
    "privilege_escalation": 2,
    "identity_confusion": 2,
    "jailbreak_roleplay": 2,
    "misinformation": 2,
    "resource_exhaustion": 2,
    "api_attack": 2,
    "social_engineering": 2,
    "SECURITY": 2,
    "memory_poisoning": 2,
    "secrets_exfiltration": 1,
    "supply_chain_injection": 1,
    "auth_bypass": 1,
    "timing_attack": 1,
    "browser_exfiltration": 1,
    "gdpr_violation": 1,
    "PHARMACY_DOSAGE": 1,
    "code_injection": 1,
    "response_manipulation": 1,
    "model_extraction": 1,
    "crypto_attack": 1,
    "data_poisoning": 1,
    "supply_chain": 1,
    "INJECTION": 1,
    "denial_of_service": 1,
    "information_disclosure": 1,
    "indirect_injection": 1,
    "multi_agent": 1,
    "secret_disclosure": 1,
    "jailbreak_hypothetical": 1
  },
  "framework_coverage": {
    "OWASP LLM Top 10": {
      "findings": 184,
      "controls": 17
    },
    "NIST AI RMF": {
      "findings": 149,
      "controls": 10
    },
    "ISO/IEC 42001": {
      "findings": 140,
      "controls": 7
    },
    "EU AI Act": {
      "findings": 131,
      "controls": 14
    },
    "SOC 2": {
      "findings": 98,
      "controls": 3
    },
    "ISO/IEC 27001": {
      "findings": 98,
      "controls": 2
    },
    "MITRE ATLAS": {
      "findings": 27,
      "controls": 10
    },
    "CWE": {
      "findings": 10,
      "controls": 6
    },
    "GDPR": {
      "findings": 10,
      "controls": 3
    },
    "HIPAA": {
      "findings": 9,
      "controls": 1
    },
    "FS AI RMF": {
      "findings": 7,
      "controls": 2
    },
    "CSA MAESTRO": {
      "findings": 6,
      "controls": 3
    },
    "MiFID II": {
      "findings": 5,
      "controls": 1
    },
    "Colorado SB 24-205": {
      "findings": 3,
      "controls": 1
    },
    "ENISA": {
      "findings": 2,
      "controls": 1
    },
    "Illinois AI Amendment": {
      "findings": 2,
      "controls": 1
    },
    "NY Local Law 144": {
      "findings": 2,
      "controls": 1
    },
    "IEEE": {
      "findings": 2,
      "controls": 2
    },
    "UNESCO": {
      "findings": 1,
      "controls": 1
    }
  },
  "evidence_quality": {
    "findings_with_supporting_traces": 243,
    "findings_with_telemetry_metrics": 243,
    "findings_with_evaluator_verdict": 208,
    "avg_evaluator_confidence": 0.8976,
    "findings_with_cvss4": 208,
    "attack_success_rate_avg": 0.6357,
    "attack_success_rate_scored_findings": 143
  },
  "scan_reliability": {
    "failed_scan_error_buckets": {
      "worker/queue lifecycle (restarts, lost jobs, DB mismatch)": 19,
      "agent endpoint misconfiguration (missing/invalid URL, wrong content type)": 10,
      "timeout / stalled scan": 9,
      "platform bug (data marshalling / code error)": 3,
      "other / unspecified": 1,
      "scan quota limit reached": 1
    },
    "avg_completed_scan_minutes": 7.1
  },
  "methodology": {
    "pipeline": "orithos probe catalog (PRB-xxxx families incl. adaptive multi-turn) -> judge cascade (tier1/tier2/tier3) -> evidence pack (trace IDs, telemetry metrics, evaluator verdict, CVSS:4.0)",
    "kv_anonymity": "cross-org publishable cells require >= 5 distinct contributing orgs; single-org cells are never published",
    "quarterly_cadence": "aggregation job runs quarterly (benchmark v1 pipeline); embargo-safe labeling until design-partner cohort graduates"
  },
  "caveats": [
    "Internal dogfooding data: Orithos scanning its own demo and test agents — not customer deployments.",
    "Only one vertical demo persona (Fortis Legal) was scanned to completion; others remain incomplete.",
    "43 of 126 scans (34%) failed and their outcomes are excluded except where noted.",
    "Single org contributes data; cross-org benchmark cells remain unpublished under the k>=5 anonymity rule until the design-partner cohort graduates."
  ],
  "citation": {
    "plain": "Orithos Research. \"The State of Agent Security — Q3 2026\". Orithos, 2026-09-20. <https://orithos.com/research/state-of-agent-security-q3-2026>",
    "data_url": "https://orithos.com/benchmarks/2026-q3.json"
  },
  "license": "CC BY 4.0"
}
