{
 "name": "YFarmX Model Security Capability",
 "homepage": "https://yfarmx.com/tools/model-security-capability/",
 "updated": "2026-09-19",
 "source": "Vendor system cards and launch tables, the RealVuln benchmark dashboard (Kolega, version 3.1.0, generated 11 September 2026), the Cybench leaderboard CSV (read 11 September 2026), Anthropic's refusals-and-fallback documentation and alignment assessment, and the published licences and injected prompts of the open-weight labs. Every score carries its harness, the task count or corpus, the safeguard state it was measured under and its provenance.",
 "license": "CC BY 4.0",
 "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
 "attribution": "YFarmX, https://yfarmx.com",
 "method": "A score row is comparable only with the three fields beside it: the harness, the corpus (or task count) and the safeguard state. Vendor scaffolds run roughly 15 to 30 points above a shared harness, so a vendor-card row and an independent row are never set side by side as one column. RealVuln micro figures score the repositories the model ran; strict figures score all 140. Refresh: re-pull reports/dashboard.json monthly and the leaderboard CSV on each frontier release.",
 "records": [
  {
   "id": "muse-spark",
   "name": "Muse Spark",
   "lab": "Meta",
   "slug": "/ai/llms/muse-spark/",
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Meta's control is contractual: the acceptable use policy and deployer-side Llama Guard, Prompt Guard and Code Shield",
   "summary": "Muse Spark scores 65.4% end-to-end on Cybench on all 40 tasks, the highest full-task-set card figure on the leaderboard read 11 September 2026.",
   "scores": [
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 65.4,
     "unit": "percent",
     "corpus": "40 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/meta-llama/llama-models/main/models/llama4/USE_POLICY.md",
    "gate": "Acceptable use policy bars creating malicious code; no verification tier; deployer-side filters shipped as Purple Llama",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Cybench leaderboard CSV",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/muse-spark//"
  },
  {
   "id": "claude-opus-4-5",
   "name": "Claude Opus 4.5",
   "lab": "Anthropic",
   "slug": null,
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "As run by Anthropic for its system card",
   "summary": "Claude Opus 4.5 scores 82% end-to-end on Cybench on the 39 of 40 tasks Anthropic ran, per the benchmark's own leaderboard read on 11 September 2026.",
   "scores": [
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 82,
     "unit": "percent",
     "corpus": "39 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Safeguard state per the system card run",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "Cybench leaderboard CSV",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability/claude-opus-4-5/"
  },
  {
   "id": "claude-opus-4-6",
   "name": "Claude Opus 4.6",
   "lab": "Anthropic",
   "slug": null,
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "As run by Anthropic for its system card",
   "summary": "Claude Opus 4.6 scores 93% end-to-end on Cybench on the 37 of 40 tasks Anthropic ran, per the benchmark's own leaderboard read on 11 September 2026.",
   "scores": [
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 93,
     "unit": "percent",
     "corpus": "37 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Safeguard state per the system card run",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "Cybench leaderboard CSV",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability/claude-opus-4-6/"
  },
  {
   "id": "claude-opus-4-7",
   "name": "Claude Opus 4.7",
   "lab": "Anthropic",
   "slug": null,
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "As run by Anthropic for its system card",
   "summary": "Claude Opus 4.7 scores 96% end-to-end on Cybench on the 35 of 40 tasks Anthropic ran, per the benchmark's own leaderboard read on 11 September 2026.",
   "scores": [
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 96,
     "unit": "percent",
     "corpus": "35 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Safeguard state per the system card run",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "Cybench leaderboard CSV",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability/claude-opus-4-7/"
  },
  {
   "id": "claude-mythos-preview",
   "name": "Claude Mythos Preview",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-mythos-5/",
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "As run by Anthropic for its system card",
   "summary": "Claude Mythos Preview scores 100% end-to-end on Cybench on the 35 of 40 tasks Anthropic ran, per the benchmark's own leaderboard read on 11 September 2026. The task count is the honest way to state it: saturated on the subset the vendor ran.",
   "scores": [
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 100,
     "unit": "percent",
     "corpus": "35 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Safeguard state per the system card run",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "Cybench leaderboard CSV",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-mythos-5//"
  },
  {
   "id": "gemma-4-31b",
   "name": "Gemma 4 31B",
   "lab": "Google",
   "slug": "/ai/llms/gemma-4/",
   "released": null,
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "Open weights under the Gemma terms; deployer-side filtering",
   "summary": "The highest precision on the RealVuln board, 90.12%, at 23.50% recall and F3 25.4 on the Python subset, run on local hardware and billed at zero. A first-pass screen that reports few findings and is usually right about them.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 25.4,
     "unit": "score",
     "corpus": "66 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "agentic harness (gemma4-31b-agentic-v1)",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "independent",
     "refusal": "open weights, no classifier",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 11.8; precision 90.12%, recall 23.5%; $0 for the run; $None per 100 lines; 303.8s wall clock a run. not billed; 303.8 seconds a repository, about 5.57 hours of single-stream wall clock over 133,782 lines",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": null,
    "gate": "Deployer-side filtering; no published refusal contract",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gemma-4//"
  },
  {
   "id": "qwen3-8-max",
   "name": "Qwen3.8-Max",
   "lab": "Alibaba",
   "slug": "/ai/llms/qwen3-8-max/",
   "released": "2026-08-03",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "Bespoke qwen3.8-max licence; Qwen3.8-27B is Apache-2.0 with no field-of-use restriction",
   "summary": "A search-derived CVE campaign puts it at 26 of 32 recent CVEs, 81.25% pass@3, for $821.35 over three runs. Its smaller sibling Qwen3.8-27B ships under Apache-2.0 and had 150 refusal-removed rebuilds within a week of release.",
   "scores": [
    {
     "benchmark": "Recent-CVE reproduction campaign",
     "version": null,
     "metric": "pass@3",
     "value": 81.25,
     "unit": "percent",
     "corpus": "32 recent CVEs",
     "harness": "Dourassov, three runs",
     "runDate": "2026-09",
     "costUsd": 821.35,
     "provenance": "independent",
     "refusal": "open weights",
     "url": "https://yfarmx.com/ai/llms/qwen3-8-max/",
     "note": "26 of 32; search-derived",
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/QwenLM/Qwen3.8/main/README.md",
    "gate": "No field-of-use restriction on the Apache-2.0 sibling; the Max licence is bespoke",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "confidence": "SINGLE",
   "links": [],
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/qwen3-8-max//"
  },
  {
   "id": "cyberkimi",
   "name": "CyberKimi",
   "lab": "Adverserial AI",
   "slug": null,
   "released": null,
   "access": "waitlist",
   "openWeights": false,
   "safeguardTier": "Refusal layer ablated by design, cyber post-training added",
   "summary": "Kimi K3 with the refusal layer removed and cyber post-training added. On the same V8 bug, harness and prompt as stock Kimi K3 it scored 8 of 16 capabilities unassisted and 10 of 16 with a methodology pack against the stock model's 4, the cleanest published control on how much willingness moves a security score.",
   "scores": [
    {
     "benchmark": "ExploitBench",
     "version": "bench-v8",
     "metric": "capabilities reached on CVE-2024-6100",
     "value": "8 of 16",
     "unit": "count",
     "corpus": "one V8 type-confusion bug, 400-turn episodes",
     "harness": "Adverserial AI campaign, 8 and 9 August 2026, unassisted",
     "runDate": "2026-08-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "refusal layer ablated",
     "url": "https://raw.githubusercontent.com/lordx64/cyberkimi-benchmarks/main/ExploitBench/CVE-2024-6100.md",
     "note": "10 of 16 with the methodology pack",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "first pass",
     "value": 65.6,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "Adverserial AI campaign",
     "runDate": "2026-08-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "refusal layer ablated",
     "url": "https://raw.githubusercontent.com/lordx64/cyberkimi-benchmarks/main/CyberGym/README.md",
     "note": "86.7% as a union over passes",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://github.com/lordx64/cyberkimi-benchmarks",
    "gate": "Ablated by design; hosted only, behind a waitlist",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "CyberKimi benchmarks and evidence",
     "url": "https://github.com/lordx64/cyberkimi-benchmarks"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability/cyberkimi/"
  },
  {
   "id": "kimi-k3",
   "name": "Kimi K3",
   "lab": "Moonshot AI",
   "slug": "/ai/llms/kimi-k3/",
   "released": "2026-07-17",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "Bespoke Moonshot licence, silent on cyber use; no moderation flag on OpenRouter",
   "summary": "The value pick on RealVuln: F3 59.4 on the Python subset at 73.02% precision for $21.33, $0.32 a repository. DeepSeek's table gives it 80.0 on CyberGym, and on ExploitBench against CVE-2024-6100 the stock model scored 4 of 16 capabilities. It is the model credited with a working Redis exploit, \"the first llm that is capable and willing to write an exploit\".",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 59.4,
     "unit": "score",
     "corpus": "66 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "agentic harness (kimi-k3-agentic-v1), prompt sha256:3481f1432c23",
     "runDate": "2026-09-11",
     "costUsd": 0.3232,
     "provenance": "independent",
     "refusal": "open weights, no classifier",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 28.6; precision 73.02%, recall 58.2%; $21.33 for the run; $0.0159 per 100 lines; 380.1s wall clock a run",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "score",
     "value": 80,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "DeepSeek comparison table",
     "runDate": "2026-09-10",
     "costUsd": null,
     "provenance": "third-party-vendor",
     "refusal": "open weights",
     "url": "https://yfarmx.com/ai/llms/kimi-k3/",
     "note": null,
     "confidence": "SINGLE"
    },
    {
     "benchmark": "ExploitBench",
     "version": "bench-v8",
     "metric": "capabilities reached on CVE-2024-6100",
     "value": "4 of 16",
     "unit": "count",
     "corpus": "one V8 type-confusion bug, 400-turn episodes",
     "harness": "Adverserial AI campaign, 8 and 9 August 2026, hosted stock control",
     "runDate": "2026-08-09",
     "costUsd": null,
     "provenance": "independent",
     "refusal": "stock hosted model, refusal layer intact",
     "url": "https://raw.githubusercontent.com/lordx64/cyberkimi-benchmarks/main/ExploitBench/CVE-2024-6100.md",
     "note": "the control against CyberKimi on the same bug, harness and prompt",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/MoonshotAI/Kimi-K3/main/LICENSE",
    "gate": "Licence silent on cyber use; Chaofan Shou called it capable and willing to write an exploit",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "Moonshot AI, Kimi K3 README",
     "url": "https://raw.githubusercontent.com/MoonshotAI/Kimi-K3/main/README.md"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/kimi-k3//"
  },
  {
   "id": "deepseek-v4-pro",
   "name": "DeepSeek V4 Pro",
   "lab": "DeepSeek",
   "slug": "/ai/llms/deepseek-v4-pro/",
   "released": "2026-08-13",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "DeepSeek licence lists twelve use restrictions; cyber activity is absent from the list",
   "summary": "RealVuln scores it F3 38.0 across all 140 repositories at 60.91% precision over three runs; a search-derived CVE campaign puts it at 87.5% pass@3 at 65.6% precision on 32 recent CVEs.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 38,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "agentic harness (deepseek-v4-pro-agentic-v1), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": 0.0815,
     "provenance": "independent",
     "refusal": "open weights, no classifier",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 38.0; precision 60.91%, recall 36.43%; $15.08 for the run; $0.002 per 100 lines; 3 runs; 285.4s wall clock a run",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "Recent-CVE reproduction campaign",
     "version": null,
     "metric": "pass@3",
     "value": 87.5,
     "unit": "percent",
     "corpus": "32 recent CVEs",
     "harness": "Dourassov, three runs",
     "runDate": "2026-09",
     "costUsd": null,
     "provenance": "independent",
     "refusal": "open weights",
     "url": "https://yfarmx.com/ai/llms/deepseek-v4-pro/",
     "note": "65.6% precision; search-derived",
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/deepseek-ai/DeepSeek-V3/main/LICENSE-MODEL",
    "gate": "Licence use restrictions name nothing on intrusion or exploitation",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/deepseek-v4-pro//"
  },
  {
   "id": "deepseek-v4-flash",
   "name": "DeepSeek V4 Flash",
   "lab": "DeepSeek",
   "slug": "/ai/llms/deepseek-v4/",
   "released": "2026-07-31",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "DeepSeek licence lists twelve use restrictions; cyber activity is absent from the list",
   "summary": "The cheapest measured pass on RealVuln: all 140 repositories at $0.0005 per 100 lines, F3 41.8 at 56.31% precision, $3.83 in total over three trials. A 120,000-line project costs about $0.62.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 41.8,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "agentic harness (deepseek-v4-flash-agentic-v1), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": 0.0204,
     "provenance": "independent",
     "refusal": "open weights, no classifier",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 41.8; precision 56.31%, recall 40.67%; $3.83 for the run; $0.0005 per 100 lines; 3 runs; 231.8s wall clock a run. 188 successful runs over 140 repositories, so cost per run carries about 1.34 passes a repository",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/deepseek-ai/DeepSeek-V3/main/LICENSE-MODEL",
    "gate": "Licence use restrictions name nothing on intrusion or exploitation",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/deepseek-v4//"
  },
  {
   "id": "deepseek-v4-1-flash",
   "name": "DeepSeek V4.1 Flash",
   "lab": "DeepSeek",
   "slug": "/ai/llms/deepseek-v4-1-flash/",
   "released": "2026-09-10",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "DeepSeek licence lists twelve use restrictions; cyber activity is absent from the list",
   "summary": "Released 10 September 2026 under MIT. DeepSeek's own table gives 88.1 on CyberGym; RealVuln has it at F3 50.9 across all 140 repositories at 97.8 seconds a run, the fastest full-corpus wall clock on the board, billed at zero because it ran on local hardware.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 50.9,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "agentic harness (deepseek-v4.1-flash-agentic-v1), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "independent",
     "refusal": "open weights, no classifier",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 50.9; precision 52.34%, recall 50.72%; $0 for the run; $None per 100 lines; 97.8s wall clock a run. not billed: the run carried no per-token bill and its real cost is GPU time",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "score",
     "value": 88.1,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "DeepSeek launch table",
     "runDate": "2026-09-10",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "open weights; no classifier",
     "url": "https://yfarmx.com/ai/llms/deepseek-v4-1-flash/",
     "note": null,
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/deepseek-ai/DeepSeek-V3/main/LICENSE-MODEL",
    "gate": "Attachment A names military use, minors, disinformation and automated decision-making among twelve restrictions and nothing on intrusion or exploitation; the alignment sits at model level and survives private deployment",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/deepseek-v4-1-flash//"
  },
  {
   "id": "glm-5-3",
   "name": "GLM-5.3",
   "lab": "Z.ai",
   "slug": "/ai/llms/glm-5-3/",
   "released": "2026-08-14",
   "access": "open-weights",
   "openWeights": true,
   "safeguardTier": "Weights licence carries no cyber restriction; hosted terms prohibit generating malicious code",
   "summary": "Z.ai's own table gives 84.5 on CyberGym and 54.4 on ExploitBench; RealVuln has it at F3 56.9 across 136 of 140 repositories at $0.0148 per 100 lines, with four validation failures. It is the default model in the Strix pentesting agent's quickstart.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 56.9,
     "unit": "score",
     "corpus": "136 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "agentic harness (glm-5.3-agentic-v1), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": 0.7774,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 54.6; precision 44.89%, recall 58.7%; $105.73 for the run; $0.0148 per 100 lines; 518.7s wall clock a run. four validation_failed exits out of 140",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "score",
     "value": 84.5,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "Z.ai launch table, 14 August 2026",
     "runDate": "2026-08-14",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "open weights; no classifier",
     "url": "https://github.com/zai-org/GLM-5",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "ExploitBench",
     "version": null,
     "metric": "score",
     "value": 54.4,
     "unit": "score",
     "corpus": "V8 exploitation ladder as run by Z.ai",
     "harness": "Z.ai launch table, 14 August 2026",
     "runDate": "2026-08-14",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "open weights; no classifier",
     "url": "https://github.com/zai-org/GLM-5",
     "note": "Z.ai's ExploitBench scaffold is one of three the record conflates under the name",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/BroWang1/ai-archive/main/snapshots/zai-org/GLM-5.3/aca966e4e02791568aa6a4ced368624b3d897f42/LICENSE",
    "gate": "Licence over the weights carries no use restriction on cyber work; the hosted API's terms are separate",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "Strix README, which names GLM-5.3 first",
     "url": "https://raw.githubusercontent.com/usestrix/strix/main/README.md"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/glm-5-3//"
  },
  {
   "id": "grok-4-6",
   "name": "Grok 4.6",
   "lab": "xAI",
   "slug": "/ai/llms/grok-4-6/",
   "released": "2026-08-12",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Published injected safety prompt: intent test, disallowed list, written-consent carve-out for official red teams",
   "summary": "On 28 August 2026 the Shannon agent ran Grok 4.6 through an authorised pentest of Photoview 2.4.0: 10 reported findings, 10 true positives, no false positives, for $35.07 in 5 hours 26 minutes, with the SARIF output opened by this desk. xAI is the one lab that publishes the text it injects in front of the model.",
   "scores": [
    {
     "benchmark": "Photoview 2.4.0 authorised pentest",
     "version": null,
     "metric": "true positives of findings reported",
     "value": "10 of 10",
     "unit": "count",
     "corpus": "Photoview 2.4.0, one application",
     "harness": "Shannon agent (Keygraph), 28 August 2026, SARIF opened",
     "runDate": "2026-08-28",
     "costUsd": 35.07,
     "provenance": "independent",
     "refusal": "xAI injected safety prompt as deployed",
     "url": "https://raw.githubusercontent.com/KeygraphHQ/shannon/main/benchmark/photoview-grok-4-6.sarif",
     "note": "a single-target run, carried for its cost and its zero false positives rather than as a benchmark",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "Cybench",
     "version": null,
     "metric": "unguided, solved",
     "value": 43,
     "unit": "percent",
     "corpus": "40 of 40 tasks",
     "harness": "vendor system card run",
     "runDate": "2026-09-11",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "as run by the vendor; safeguard state per the card",
     "url": "https://raw.githubusercontent.com/cybench/cybench.github.io/main/data/leaderboard.csv",
     "note": "task count is what the vendor ran, so rows are comparable only with it beside them",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/xai-org/grok-prompts/main/grok_4_safety_prompt.txt",
    "gate": "Do not answer queries that show clear intent to engage in disallowed activities; high-level answers without actionable detail to general questions such as \"how to hack a website?\"; \"unlawfully\" is the operative word",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "xAI, Grok 4 safety prompt",
     "url": "https://raw.githubusercontent.com/xai-org/grok-prompts/main/grok_4_safety_prompt.txt"
    }
   ],
   "details": "The Cybench row is the Grok 4 card figure on all 40 tasks; Grok 4.1 Thinking sits at 39% and Grok 4 Fast at 30% on the same leaderboard.",
   "confidence": "CONFIRMED",
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/grok-4-6//"
  },
  {
   "id": "gemini-3-5-flash",
   "name": "Gemini 3.5 Flash",
   "lab": "Google",
   "slug": "/ai/llms/gemini-3-5-family/",
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Classifiers plus in-model protections; malicious accounts disabled",
   "summary": "The precision-shaped row on RealVuln: 89.77% precision at 33.26% recall, F3 35.5, on 64 of 140 repositories over three runs, with one timeout and one validation failure. It reports few findings and is usually right about them.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 35.5,
     "unit": "score",
     "corpus": "64 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "agentic harness (gemini-3.5-flash-agentic-v1), prompt sha256:3481f1432c23",
     "runDate": "2026-09-11",
     "costUsd": 0.5513,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 16.2; precision 89.77%, recall 33.26%; $81.05 for the run; $0.027 per 100 lines; 3 runs; 175.4s wall clock a run",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://cloud.google.com/blog/topics/threat-intelligence/ai-vulnerability-exploitation-initial-access",
    "gate": "Classifier layer alongside in-model training; the Pro tier of the same generation still refuses code-vulnerability analysis on retest",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-08-16",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gemini-3-5-family//"
  },
  {
   "id": "gemini-3-5-flash-cyber",
   "name": "Gemini 3.5 Flash Cyber",
   "lab": "Google",
   "slug": "/ai/llms/gemini-3-5-family/",
   "released": "2026-07-21",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Limited-access pilot through CodeMender for governments and trusted partners",
   "summary": "The first Flash Cyber model, credited by Google with 55 unique confirmed V8 issues on Big Sleep; its CyberGym figure is contested between 83.2% and 77.5% across Google's own materials.",
   "scores": [
    {
     "benchmark": "Big Sleep on V8",
     "version": null,
     "metric": "unique confirmed issues",
     "value": 55,
     "unit": "count",
     "corpus": "Chromium V8",
     "harness": "Google Big Sleep harness",
     "runDate": "2026-07-21",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "cyber-tuned model inside Google's own agent",
     "url": "https://yfarmx.com/ai/llms/gemini-3-5-family/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "pass@1",
     "value": "83.2 or 77.5",
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "Google launch materials, 21 July 2026",
     "runDate": "2026-07-21",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "cyber-tuned model",
     "url": "https://yfarmx.com/ai/llms/gemini-3-5-family/",
     "note": "two figures in Google's own materials",
     "confidence": "CONTESTED"
    }
   ],
   "refusal": {
    "policyUrl": "https://cloud.google.com/blog/topics/threat-intelligence/ai-vulnerability-exploitation-initial-access",
    "gate": "Limited-access pilot; Google mitigates model abuse by disabling malicious accounts",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "confidence": "CONFIRMED",
   "links": [],
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gemini-3-5-family//"
  },
  {
   "id": "gemini-3-8-flash-cyber",
   "name": "Gemini 3.8 Flash Cyber",
   "lab": "Google",
   "slug": "/ai/llms/gemini-3-8-flash/",
   "released": "2026-09-02",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Fairwind Programme: trusted government authorities, critical infrastructure operators and software maintainers",
   "summary": "The model inside CodeMender since 2 September 2026, reachable only through Fairwind. Google's own table gives 86.2% pass@1 on CyberGym, 47.2% on CWE-Bench at roughly $3.60 a rollout, and a 6.0% attack success rate on Gray Swan injection.",
   "scores": [
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "pass@1",
     "value": 86.2,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "Google comparison table, 2 September 2026",
     "runDate": "2026-09-02",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "cyber-tuned model for vetted defenders",
     "url": "https://yfarmx.com/ai/llms/gemini-3-8-flash/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CWE-Bench",
     "version": null,
     "metric": "pass@1",
     "value": 47.2,
     "unit": "percent",
     "corpus": "CWE-Bench",
     "harness": "Google comparison table, 2 September 2026",
     "runDate": "2026-09-02",
     "costUsd": 3.6,
     "provenance": "vendor-card",
     "refusal": "cyber-tuned model for vetted defenders",
     "url": "https://yfarmx.com/ai/llms/gemini-3-8-flash/",
     "note": "cost is per rollout, per Google",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "Gray Swan injection",
     "version": null,
     "metric": "attack success rate",
     "value": 6,
     "unit": "percent",
     "corpus": "Gray Swan prompt-injection arena",
     "harness": "Google comparison table, 2 September 2026",
     "runDate": "2026-09-02",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "lower is better",
     "url": "https://yfarmx.com/ai/llms/gemini-3-8-flash/",
     "note": null,
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://cloud.google.com/blog/topics/threat-intelligence/ai-vulnerability-exploitation-initial-access",
    "gate": "Access by programme: Fairwind, restricted to trusted government authorities, critical infrastructure operators and software maintainers",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "Google Cloud threat intelligence, AI in vulnerability exploitation",
     "url": "https://cloud.google.com/blog/topics/threat-intelligence/ai-vulnerability-exploitation-initial-access"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gemini-3-8-flash//"
  },
  {
   "id": "gpt-5-5",
   "name": "GPT-5.5",
   "lab": "OpenAI",
   "slug": "/ai/llms/gpt-5-5-spud/",
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Classifier refusals; GPT-5.5-Cyber is the permissive variant under Trusted Access",
   "summary": "RealVuln scores GPT-5.5 at F3 56.7 on the Python subset at 72.62% precision over three runs, and Google's comparison table quotes GPT-5.5-Cyber at 85.6% pass@1 on CyberGym.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 56.7,
     "unit": "score",
     "corpus": "66 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "agentic harness (gpt-5.5-agentic-v1)",
     "runDate": "2026-09-11",
     "costUsd": 0.7354,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 27.2; precision 72.62%, recall 55.36%; $144.14 for the run; $0.036 per 100 lines; 3 runs; 180.1s wall clock a run",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CyberGym",
     "version": null,
     "metric": "pass@1",
     "value": 85.6,
     "unit": "percent",
     "corpus": "CyberGym",
     "harness": "quoted in Google's comparison table, 2 September 2026",
     "runDate": "2026-09-02",
     "costUsd": null,
     "provenance": "third-party-vendor",
     "refusal": "GPT-5.5-Cyber variant, safeguards reduced",
     "url": "https://yfarmx.com/ai/llms/gemini-3-8-flash/",
     "note": null,
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/openai/model_spec/main/model_spec.md",
    "gate": "Model Spec: high-risk activities including hacking prohibited unless explicitly authorised by applicable instructions",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gpt-5-5-spud//"
  },
  {
   "id": "gpt-daybreak-red",
   "name": "GPT Daybreak Red",
   "lab": "OpenAI",
   "slug": null,
   "released": "2026-08-10",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Trusted Access for Cyber, offensive tier: separate approval, stronger monitoring, hardware security keys from 1 September 2026",
   "summary": "The only model in the field sold for \"proof-of-concept exploit development, exploit-chain validation, penetration testing, and red teaming\", at $12.50 and $75 per million tokens with a 400,000-token window. No public benchmark row exists for it.",
   "scores": [],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/openai/codex/main/codex-rs/tui/src/daybreak.rs",
    "gate": "Refusals lifted for approved organisations under separate provisioning",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "OpenAI python client, model union naming gpt-daybreak-red-latest",
     "url": "https://raw.githubusercontent.com/openai/openai-python/main/src/openai/types/shared/all_models.py"
    }
   ],
   "confidence": "SINGLE",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability/gpt-daybreak-red/"
  },
  {
   "id": "gpt-daybreak-blue",
   "name": "GPT Daybreak Blue",
   "lab": "OpenAI",
   "slug": "/ai/llms/gpt-5-6/",
   "released": "2026-08-10",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Trusted Access for Cyber, defensive tier: identity verification, legal attestation, account monitoring",
   "summary": "The highest-scoring model row on RealVuln: F3 79.5 across all 140 repositories at 63.98% precision and 81.75% recall, $3.91 a repository, in 347.2 seconds a run. It is GPT-5.6 Sol's weights served to verified defenders under the Daybreak programme, and the 4.8-point gap to Sol on the same board is consistent with the permitted tier spending fewer turns declining.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 79.5,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "Codex CLI (gpt-daybreak-blue-codex-cli), prompt sha256:45a1200d61e6 (tsjs-v1)",
     "runDate": "2026-09-11",
     "costUsd": 3.9112,
     "provenance": "independent",
     "refusal": "Daybreak Blue tier: defensive tasks on mainline weights",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 79.5; precision 63.98%, recall 81.75%; $547.56 for the run; $0.0739 per 100 lines; 347.2s wall clock a run; reasoning effort high",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/openai/codex/main/codex-rs/tui/src/daybreak.rs",
    "gate": "Defensive tasks on mainline weights for verified organisations; Blue approval does not carry into Red",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "OpenAI python client, model union naming gpt-daybreak-blue-latest",
     "url": "https://raw.githubusercontent.com/openai/openai-python/main/src/openai/types/shared/all_models.py"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gpt-5-6//"
  },
  {
   "id": "gpt-5-6-sol",
   "name": "GPT-5.6 Sol",
   "lab": "OpenAI",
   "slug": "/ai/llms/gpt-5-6/",
   "released": "2026-07-09",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Classifier refusals; a refused turn retries once on the configured Daybreak model",
   "summary": "$4 and $20 per million tokens. RealVuln scores it F3 74.7 across all 140 repositories at 59.33% precision and 76.92% recall, 4.8 points behind Daybreak Blue on identical weights, and OpenAI's own table gives 78.5% on ExploitBench. In Codex a refused turn retries once on the Daybreak model without changing the session's stored model.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 74.7,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "Codex CLI (gpt-5.6-sol-codex-cli), prompt sha256:45a1200d61e6 (tsjs-v1)",
     "runDate": "2026-09-11",
     "costUsd": 4.0024,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 74.7; precision 59.33%, recall 76.92%; $560.34 for the run; $0.0756 per 100 lines; 568.6s wall clock a run; reasoning effort high",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "ExploitBench",
     "version": null,
     "metric": "score",
     "value": 78.5,
     "unit": "percent",
     "corpus": "ExploitBench as run by OpenAI",
     "harness": "OpenAI comparison table",
     "runDate": "2026-09-03",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguard state unstated in the summaries read",
     "url": "https://yfarmx.com/ai/llms/gpt-6-astra/",
     "note": null,
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/openai/codex/main/codex-rs/tui/src/daybreak.rs",
    "gate": "Refused turns retry once on the configured Daybreak model; only a refused turn routes there",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "OpenAI Codex, Daybreak eligibility module",
     "url": "https://raw.githubusercontent.com/openai/codex/main/codex-rs/tui/src/daybreak.rs"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gpt-5-6//"
  },
  {
   "id": "gpt-6-astra",
   "name": "GPT-6 Astra",
   "lab": "OpenAI",
   "slug": "/ai/llms/gpt-6-astra/",
   "released": "2026-09-03",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Classifier refusals; enterprise cyber access off by default at launch; Daybreak unavailable for Astra",
   "summary": "OpenAI's newest and most expensive model scores 100% on ExploitBench in its own system card, with a contamination caveat reported on the same set, and refuses 91.5% of disallowed cyber requests. On RealVuln it reads all 140 repositories at F3 52.1 for $1,040.62, $7.43 a repository, behind models costing a twentieth as much.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 52.1,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "Codex CLI (gpt-6-astra-codex-cli), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": 7.433,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 52.1; precision 44.73%, recall 53.09%; $1040.62 for the run; $0.1404 per 100 lines; 574.7s wall clock a run; reasoning effort high",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "ExploitBench",
     "version": null,
     "metric": "score",
     "value": 100,
     "unit": "percent",
     "corpus": "ExploitBench as run by OpenAI, described as 41 vulnerabilities",
     "harness": "OpenAI system card, 3 September 2026",
     "runDate": "2026-09-03",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "production safeguards state unstated in the summaries read",
     "url": "https://yfarmx.com/ai/llms/gpt-6-astra/",
     "note": "the same card reportedly flags historical-vulnerability contamination on the set; two V8 zero-days found mid-evaluation are described as disclosed to maintainers",
     "confidence": "SINGLE"
    },
    {
     "benchmark": "ExploitGym",
     "version": null,
     "metric": "honeypot overreach",
     "value": 0,
     "unit": "percent",
     "corpus": "ExploitGym honeypot scenario",
     "harness": "OpenAI system card, 3 September 2026",
     "runDate": "2026-09-03",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards on; GPT-5.6 Sol without production safeguards is quoted at 48.2%",
     "url": "https://yfarmx.com/ai/llms/gpt-6-astra/",
     "note": null,
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://raw.githubusercontent.com/openai/model_spec/main/model_spec.md",
    "gate": "Refused 91.5% of disallowed cyber requests against 59% for GPT-5.6 Sol (card summary); the Codex client says Daybreak is unavailable for Astra",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "SINGLE"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "OpenAI Model Spec",
     "url": "https://raw.githubusercontent.com/openai/model_spec/main/model_spec.md"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/gpt-6-astra//"
  },
  {
   "id": "claude-sonnet-4-6",
   "name": "Claude Sonnet 4.6",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-sonnet-4-6/",
   "released": null,
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Real-time cyber safeguards on Sonnet-class models",
   "summary": "Carried for one row: Kolega's own Sonnet 4.6 adaptation on RealVuln scored F3 47.9 on the Python subset at 66.89% precision, and at $0.2173 per 100 lines it is the most expensive row on the board per line.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 47.9,
     "unit": "score",
     "corpus": "66 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "Kolega adaptation on Claude Code (kolega-ca-cc-sonnet)",
     "runDate": "2026-09-11",
     "costUsd": 4.4042,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 22.9; precision 66.89%, recall 46.42%; $290.68 for the run; $0.2173 per 100 lines; 396.6s wall clock a run",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Real-time cyber safeguards on Opus and Sonnet",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-sonnet-4-6//"
  },
  {
   "id": "claude-sonnet-5",
   "name": "Claude Sonnet 5",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-sonnet-5/",
   "released": "2026-06-30",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "First Sonnet with real-time cyber safeguards",
   "summary": "$2 and $10 per million tokens with a 1M window. RealVuln scored it F3 43.3 across all 140 repositories at 52.65% precision and 42.48% recall for $130.67, and its classifier blocks 3.2% of defensive discovery requests and flags 0.52% of benign traffic.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 43.3,
     "unit": "score",
     "corpus": "140 of 140 repositories pinned by commit SHA (full corpus, TypeScript and JavaScript included)",
     "harness": "Claude Code agentic harness (claude-sonnet-5-cc-agentic-v1), prompt sha256:45a1200d61e6",
     "runDate": "2026-09-11",
     "costUsd": 0.9333,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 43.3; precision 52.65%, recall 42.48%; $130.67 for the run; $0.0176 per 100 lines; 230.0s wall clock a run; reasoning effort high",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Real-time cyber safeguards; the Cyber Verification Programme lifts the dual-use tier",
    "blockedPct": 3.2,
    "flaggedPct": 0.52,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "Anthropic, Claude Sonnet 5 overview",
     "url": "https://platform.claude.com/docs/en/models/sonnet-5/overview"
    }
   ],
   "details": "the four Anthropic blocked and flagged rates rest on two independent readings of the Fable 5.1 and Mythos 5.1 system card, which this desk has not opened.",
   "confidence": "CONFIRMED",
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-sonnet-5//"
  },
  {
   "id": "claude-opus-4-8",
   "name": "Claude Opus 4.8",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-opus-4-8/",
   "released": "2026-05-28",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Real-time cyber safeguards; the fallback target for flagged cyber requests",
   "summary": "The legacy model keeps a working role: every worked example on Anthropic's refusals page routes a cyber refusal to claude-opus-4-8, and Claude Code's in-product message reads \"Switched to Opus 4.8\". On 11 September 2026 a research agent in this desk's own pack was stopped on Opus 4.8 with Details: [cyber] while working from a published, patched Redis advisory.",
   "scores": [],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Real-time cyber safeguards; the documented default fallback for a cyber refusal on Fable 5, Fable 5.1 and Opus 5; the Cyber Verification Programme lifts the dual-use tier",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Anthropic, refusals and fallback",
     "url": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-opus-4-8//"
  },
  {
   "id": "claude-opus-5",
   "name": "Claude Opus 5",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-opus-5/",
   "released": "2026-07-24",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Refusal classifiers on; source-code discovery permitted, exploitation classified; fallback to Opus 4.8",
   "summary": "The one model in this tracker with a fully auditable independent run: RealVuln scored it F3 67.7 at 61.95% precision and 68.40% recall on 66 Python repositories for $90.68, $1.37 a repository, on 30 July 2026 through Claude Code with edits disabled. Anthropic's own chart has it identifying at 79.4% and solving 4 exploitation challenges, and its classifier blocks 13.9% of defensive discovery.",
   "scores": [
    {
     "benchmark": "RealVuln",
     "version": "3.1.0",
     "metric": "F3 (micro)",
     "value": 67.7,
     "unit": "score",
     "corpus": "66 of 140 repositories pinned by commit SHA (Python subset)",
     "harness": "Claude Code CLI 2.1.220, headless, Edit, Write and NotebookEdit disabled, concurrency 5",
     "runDate": "2026-07-30",
     "costUsd": 1.3739,
     "provenance": "independent",
     "refusal": "production safeguards as deployed on the API",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json",
     "note": "strict F3 33.1; precision 61.95%, recall 68.4%; $90.68 for the run; $0.0678 per 100 lines; 208.2s wall clock a run. 145 true positives against 2 false positives on critical findings; quote the micro figure and say it was scored on the Python subset",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "OSS-Fuzz",
     "version": null,
     "metric": "vulnerability identification",
     "value": 79.4,
     "unit": "percent",
     "corpus": "OSS-Fuzz set as run by Anthropic",
     "harness": "Anthropic launch chart, 24 July 2026",
     "runDate": "2026-07-24",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards on",
     "url": "https://yfarmx.com/ai/llms/claude-opus-5/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "OSS-Fuzz",
     "version": null,
     "metric": "exploitation challenges solved",
     "value": 4,
     "unit": "count",
     "corpus": "OSS-Fuzz exploitation set as run by Anthropic",
     "harness": "Anthropic launch chart, 24 July 2026",
     "runDate": "2026-07-24",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards on; Anthropic says the distance to Mythos 5 is the safeguard blocking the exploit step",
     "url": "https://yfarmx.com/ai/llms/claude-opus-5/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "Anthropic alignment assessment",
     "version": null,
     "metric": "runs with at least one severely harmful action",
     "value": 31,
     "unit": "percent",
     "corpus": "150 capture-the-flag replication runs",
     "harness": "Anthropic internal replication, cyber safeguards off",
     "runDate": "2026-09-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off by design of the test",
     "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
     "note": null,
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Source-code discovery permitted at every access level; binary scanning, penetration testing and exploit generation blocked by default; fallback to Claude Opus 4.8",
    "blockedPct": 13.9,
    "flaggedPct": 0.61,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "RealVuln dashboard",
     "url": "https://raw.githubusercontent.com/kolega-ai/Real-Vuln-Benchmark/main/reports/dashboard.json"
    },
    {
     "label": "RealVuln Opus 5 run manifest",
     "url": "https://github.com/kolega-ai/Real-Vuln-Benchmark/blob/main/llm-bench/run-manifests/claude-opus-5-cc-agentic-v1-python-v2.json"
    },
    {
     "label": "Anthropic, Claude Opus 5 overview",
     "url": "https://platform.claude.com/docs/en/models/opus-5/overview"
    }
   ],
   "details": "The strict column, which counts every ground-truth finding in all 140 repositories including the 74 TypeScript and JavaScript ones the run never saw, puts Opus 5 at F3 33.1 and recall 31.44%. the four Anthropic blocked and flagged rates rest on two independent readings of the Fable 5.1 and Mythos 5.1 system card, which this desk has not opened.",
   "confidence": "CONFIRMED",
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-opus-5//"
  },
  {
   "id": "claude-mythos-5",
   "name": "Claude Mythos 5",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-mythos-5/",
   "released": "2026-06-09",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Same weights as Fable 5 without the classifiers; Project Glasswing, by invitation",
   "summary": "The gated cyber tier of the June 2026 generation. On Anthropic's OSS-Fuzz chart it identifies vulnerabilities at 80.0% and solves 13 exploitation challenges against Claude Opus 5's 79.4% and 4, the largest published capability gap on that task inside one family. In the September 2026 alignment assessment it took at least one severely harmful action in 82% of 150 replication runs with safeguards off.",
   "scores": [
    {
     "benchmark": "OSS-Fuzz",
     "version": null,
     "metric": "vulnerability identification",
     "value": 80,
     "unit": "percent",
     "corpus": "OSS-Fuzz set as run by Anthropic",
     "harness": "Anthropic launch chart, 24 July 2026",
     "runDate": "2026-07-24",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off (Mythos tier)",
     "url": "https://yfarmx.com/ai/llms/claude-opus-5/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "OSS-Fuzz",
     "version": null,
     "metric": "exploitation challenges solved",
     "value": 13,
     "unit": "count",
     "corpus": "OSS-Fuzz exploitation set as run by Anthropic",
     "harness": "Anthropic launch chart, 24 July 2026",
     "runDate": "2026-07-24",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off (Mythos tier)",
     "url": "https://yfarmx.com/ai/llms/claude-opus-5/",
     "note": null,
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "Anthropic alignment assessment",
     "version": null,
     "metric": "runs with at least one severely harmful action",
     "value": 82,
     "unit": "percent",
     "corpus": "150 capture-the-flag replication runs",
     "harness": "Anthropic internal replication, cyber safeguards off",
     "runDate": "2026-09-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off by design of the test",
     "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
     "note": "the models in the four replicated incidents ran without the cyber safeguards that ship with released models",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/models/mythos-5/overview",
    "gate": "Reduced safeguards, organisation-vetted access through Project Glasswing",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Anthropic, Claude Mythos 5 overview",
     "url": "https://platform.claude.com/docs/en/models/mythos-5/overview"
    },
    {
     "label": "Anthropic, alignment assessment",
     "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-mythos-5//"
  },
  {
   "id": "claude-fable-5",
   "name": "Claude Fable 5",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-fable-5/",
   "released": "2026-06-09",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Refusal classifiers on; at launch blocked 90.0% of defensive discovery",
   "summary": "The June 2026 flagship whose launch classifier blocked 90.0% of defensive vulnerability discovery and flagged 15.0% of benign defensive traffic, the baseline the Fable 5.1 figures are measured against. Two rival labs' tables score it: 78.0 on ExploitBench (Z.ai) and 47.8% pass@1 on CWE-Bench (Google).",
   "scores": [
    {
     "benchmark": "ExploitBench",
     "version": null,
     "metric": "score",
     "value": 78,
     "unit": "score",
     "corpus": "V8 exploitation ladder as run by Z.ai",
     "harness": "Z.ai comparison table, 14 August 2026",
     "runDate": "2026-08-14",
     "costUsd": null,
     "provenance": "third-party-vendor",
     "refusal": "as run by Z.ai; safeguard state unstated",
     "url": "https://yfarmx.com/ai/llms/glm-5-3/",
     "note": "ExploitBench appears under three scaffolds in the record; demand harness, bug subset and seed count beside any percentage",
     "confidence": "CONFIRMED"
    },
    {
     "benchmark": "CWE-Bench",
     "version": null,
     "metric": "pass@1",
     "value": 47.8,
     "unit": "percent",
     "corpus": "CWE-Bench",
     "harness": "Google comparison table, 2 September 2026",
     "runDate": "2026-09-02",
     "costUsd": null,
     "provenance": "third-party-vendor",
     "refusal": "as run by Google; safeguard state unstated",
     "url": "https://yfarmx.com/ai/llms/gemini-3-8-flash/",
     "note": "Google prices the same row at roughly $10 a rollout against $3.60 for its own Flash Cyber",
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Refusal classifier with stop_reason refusal; category cyber",
    "blockedPct": 90,
    "flaggedPct": 15,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Anthropic, refusals and fallback",
     "url": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback"
    }
   ],
   "details": "the four Anthropic blocked and flagged rates rest on two independent readings of the Fable 5.1 and Mythos 5.1 system card, which this desk has not opened.",
   "confidence": "CONFIRMED",
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-fable-5//"
  },
  {
   "id": "claude-mythos-5-1",
   "name": "Claude Mythos 5.1",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-mythos-5/",
   "released": "2026-09-01",
   "access": "gated",
   "openWeights": false,
   "safeguardTier": "Same weights as Fable 5.1 without the classifiers; Project Glasswing, by invitation",
   "summary": "Offered separately, by invitation only, as part of Project Glasswing, sharing Fable 5.1's specifications and pricing. Its Terminal-Bench 4.0 score of 60.9% against Fable 5.1's 55.8% is the published price of the safety classifier, per the system card summary.",
   "scores": [
    {
     "benchmark": "Terminal-Bench",
     "version": "4.0",
     "metric": "solved",
     "value": 60.9,
     "unit": "percent",
     "corpus": "Terminal-Bench 4.0 task set",
     "harness": "Anthropic system card",
     "runDate": "2026-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off",
     "url": "https://platform.claude.com/docs/en/models/mythos-5-1/overview",
     "note": "a coding score, carried because the card ties the gap to Fable 5.1 to the safeguard",
     "confidence": "SINGLE"
    },
    {
     "benchmark": "Anthropic alignment assessment",
     "version": null,
     "metric": "runs with at least one severely harmful action",
     "value": 33,
     "unit": "percent",
     "corpus": "150 capture-the-flag replication runs",
     "harness": "Anthropic internal replication, cyber safeguards off",
     "runDate": "2026-09-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards off by design of the test",
     "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
     "note": null,
     "confidence": "CONFIRMED"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/models/mythos-5-1/overview",
    "gate": "Reduced safeguards, organisation-vetted access through Project Glasswing",
    "blockedPct": null,
    "flaggedPct": null,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Anthropic, Claude Mythos 5.1 overview",
     "url": "https://platform.claude.com/docs/en/models/mythos-5-1/overview"
    },
    {
     "label": "Anthropic, alignment assessment",
     "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents"
    }
   ],
   "confidence": "CONFIRMED",
   "details": null,
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-mythos-5//"
  },
  {
   "id": "claude-fable-5-1",
   "name": "Claude Fable 5.1",
   "lab": "Anthropic",
   "slug": "/ai/llms/claude-fable-5-1/",
   "released": "2026-09-01",
   "access": "api",
   "openWeights": false,
   "safeguardTier": "Refusal classifiers on; cyber work as deployed falls back to Opus 4.8",
   "summary": "Anthropic's flagship as of 1 September 2026, $10 and $50 per million tokens with a 1M window. Source-code vulnerability discovery is permitted; the classifier blocks 7.0% of defensive discovery requests and flags 1.03% of benign defensive traffic, and a flagged cyber request retries on Claude Opus 4.8 through server-side fallback.",
   "scores": [
    {
     "benchmark": "Terminal-Bench",
     "version": "4.0",
     "metric": "solved",
     "value": 55.8,
     "unit": "percent",
     "corpus": "Terminal-Bench 4.0 task set",
     "harness": "Anthropic system card",
     "runDate": "2026-09",
     "costUsd": null,
     "provenance": "vendor-card",
     "refusal": "safeguards on; the card summary attributes the 5.1-point gap to Mythos 5.1 to tasks where the cyber safeguards intervened",
     "url": "https://yfarmx.com/ai/llms/claude-fable-5-1/",
     "note": "a coding score, carried because the card ties it to the safeguard",
     "confidence": "SINGLE"
    }
   ],
   "refusal": {
    "policyUrl": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback",
    "gate": "Source-code vulnerability discovery permitted; exploitation classified; server-side fallback to Claude Opus 4.8 on a cyber refusal",
    "blockedPct": 7,
    "flaggedPct": 1.03,
    "observedDate": "2026-09-11",
    "confidence": "CONFIRMED"
   },
   "links": [
    {
     "label": "Anthropic, refusals and fallback",
     "url": "https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback"
    },
    {
     "label": "Anthropic, prompting Claude Fable 5.1",
     "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5-1"
    }
   ],
   "details": "the four Anthropic blocked and flagged rates rest on two independent readings of the Fable 5.1 and Mythos 5.1 system card, which this desk has not opened. Anthropic's own prompting guide states the direction: \"finding vulnerabilities in source code is permitted\".",
   "confidence": "CONFIRMED",
   "date": "2026-09-11",
   "url": "https://yfarmx.com/tools/model-security-capability//ai/llms/claude-fable-5-1//"
  }
 ]
}