46 lines
No EOL
2.4 KiB
JSON
46 lines
No EOL
2.4 KiB
JSON
{
|
|
"rejected_claims": [
|
|
{
|
|
"filename": "alignment-impossibility-converges-across-three-mathematical-traditions.md",
|
|
"issues": [
|
|
"missing_attribution_extractor"
|
|
]
|
|
},
|
|
{
|
|
"filename": "reward-hacking-is-globally-inevitable-in-finite-sample-regimes.md",
|
|
"issues": [
|
|
"missing_attribution_extractor"
|
|
]
|
|
},
|
|
{
|
|
"filename": "consensus-driven-objective-reduction-is-the-practical-pathway-out-of-alignment-impossibility.md",
|
|
"issues": [
|
|
"missing_attribution_extractor"
|
|
]
|
|
}
|
|
],
|
|
"validation_stats": {
|
|
"total": 3,
|
|
"kept": 0,
|
|
"fixed": 9,
|
|
"rejected": 3,
|
|
"fixes_applied": [
|
|
"alignment-impossibility-converges-across-three-mathematical-traditions.md:set_created:2026-03-15",
|
|
"alignment-impossibility-converges-across-three-mathematical-traditions.md:stripped_wiki_link:universal-alignment-is-mathematically-impossible-because-arr",
|
|
"alignment-impossibility-converges-across-three-mathematical-traditions.md:stripped_wiki_link:single-reward-rlhf-cannot-align-diverse-preferences-because-",
|
|
"reward-hacking-is-globally-inevitable-in-finite-sample-regimes.md:set_created:2026-03-15",
|
|
"reward-hacking-is-globally-inevitable-in-finite-sample-regimes.md:stripped_wiki_link:emergent-misalignment-arises-naturally-from-reward-hacking-a",
|
|
"reward-hacking-is-globally-inevitable-in-finite-sample-regimes.md:stripped_wiki_link:the-specification-trap-means-any-values-encoded-at-training-",
|
|
"consensus-driven-objective-reduction-is-the-practical-pathway-out-of-alignment-impossibility.md:set_created:2026-03-15",
|
|
"consensus-driven-objective-reduction-is-the-practical-pathway-out-of-alignment-impossibility.md:stripped_wiki_link:universal-alignment-is-mathematically-impossible-because-arr",
|
|
"consensus-driven-objective-reduction-is-the-practical-pathway-out-of-alignment-impossibility.md:stripped_wiki_link:community-centred-norm-elicitation-surfaces-alignment-target"
|
|
],
|
|
"rejections": [
|
|
"alignment-impossibility-converges-across-three-mathematical-traditions.md:missing_attribution_extractor",
|
|
"reward-hacking-is-globally-inevitable-in-finite-sample-regimes.md:missing_attribution_extractor",
|
|
"consensus-driven-objective-reduction-is-the-practical-pathway-out-of-alignment-impossibility.md:missing_attribution_extractor"
|
|
]
|
|
},
|
|
"model": "anthropic/claude-sonnet-4.5",
|
|
"date": "2026-03-15"
|
|
} |