{ "rejected_claims": [ { "filename": "metr-time-horizon-benchmark-saturating-at-governance-relevant-capability-levels.md", "issues": [ "missing_attribution_extractor" ] }, { "filename": "ai-capability-evaluation-scaffold-sensitivity-introduces-cross-model-comparison-uncertainty.md", "issues": [ "missing_attribution_extractor" ] } ], "validation_stats": { "total": 2, "kept": 0, "fixed": 3, "rejected": 2, "fixes_applied": [ "metr-time-horizon-benchmark-saturating-at-governance-relevant-capability-levels.md:set_created:2026-03-24", "metr-time-horizon-benchmark-saturating-at-governance-relevant-capability-levels.md:stripped_wiki_link:verification degrades faster than capability grows", "ai-capability-evaluation-scaffold-sensitivity-introduces-cross-model-comparison-uncertainty.md:set_created:2026-03-24" ], "rejections": [ "metr-time-horizon-benchmark-saturating-at-governance-relevant-capability-levels.md:missing_attribution_extractor", "ai-capability-evaluation-scaffold-sensitivity-introduces-cross-model-comparison-uncertainty.md:missing_attribution_extractor" ] }, "model": "anthropic/claude-sonnet-4.5", "date": "2026-03-24" }