{ "rejected_claims": [ { "filename": "benchmark-based-capability-metrics-overstate-autonomous-performance-through-automated-scoring-gap.md", "issues": [ "missing_attribution_extractor" ] }, { "filename": "ai-tools-reduced-experienced-developer-productivity-19-percent-in-rct-conditions.md", "issues": [ "missing_attribution_extractor" ] } ], "validation_stats": { "total": 2, "kept": 0, "fixed": 3, "rejected": 2, "fixes_applied": [ "benchmark-based-capability-metrics-overstate-autonomous-performance-through-automated-scoring-gap.md:set_created:2026-03-24", "benchmark-based-capability-metrics-overstate-autonomous-performance-through-automated-scoring-gap.md:stripped_wiki_link:verification degrades faster than capability grows", "ai-tools-reduced-experienced-developer-productivity-19-percent-in-rct-conditions.md:set_created:2026-03-24" ], "rejections": [ "benchmark-based-capability-metrics-overstate-autonomous-performance-through-automated-scoring-gap.md:missing_attribution_extractor", "ai-tools-reduced-experienced-developer-productivity-19-percent-in-rct-conditions.md:missing_attribution_extractor" ] }, "model": "anthropic/claude-sonnet-4.5", "date": "2026-03-24" }