[{"applicability":"hypothesized","applicability_caveats":["Estimated carryover only. Reaching 85% would not establish researcher substitution."],"attributed_interpretation":"Anthropic expects a research-staff substitute to score at least 85% on the prior version and estimates carryover to 2.1.","benchmark_version_id":"cobench-2-1","id":"co85","kind":"estimated_necessary_capability","logical_role":"estimated_necessary_condition","logical_role_notes":"Necessary does not mean sufficient. No researcher-replacement percentage is calculated.","metric_id":"score","qualitative_condition":null,"source_refs":[{"locator":"§2.3.4.1, printed pp.36–37; PDF indices 35–36; Figure 2.3.4.1.A","source_id":"src-opus55"}],"valid_from":null,"valid_to":null,"value":85},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"Source prints 72.38%; represented as 0.7238 on its 0–1 reward scale. Best human solution, not average human ability.","benchmark_version_id":"nanogpt-2026","id":"nanogpt-human","kind":"human_baseline","logical_role":"measured_baseline","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"score","qualitative_condition":null,"source_refs":[{"locator":"§10.1.3.3","source_id":"src-astra"}],"valid_from":null,"valid_to":null,"value":0.7238},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"DeepMind uses 90% as a conservative CCL rule-out boundary, not proof that a system reaching it automates research.","benchmark_version_id":"grb-2026-08","id":"grb-policy","kind":"policy_threshold","logical_role":"policy_trigger","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"pass1","qualitative_condition":null,"source_refs":[{"locator":"Printed pp.36–37; PDF indices 35–36","source_id":"src-grb"}],"valid_from":null,"valid_to":null,"value":90},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"The original repository algorithm maps to 0.1; this is not a human-researcher threshold.","benchmark_version_id":"ai4ai-v1","id":"ai4ai-anchor","kind":"normalization_anchor","logical_role":"unknown","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"normalized","qualitative_condition":null,"source_refs":[{"locator":"§2.5","source_id":"src-ai4ai"}],"valid_from":null,"valid_to":null,"value":0.1},{"applicability":"established","applicability_caveats":["Qualitative source assessment only; do not infer a numeric cutoff or distance to the threshold."],"attributed_interpretation":"OpenAI reports 78.05% as below its indicative High capability threshold. A numeric cutoff has not been verified in this record. This is not an RSI threshold.","benchmark_version_id":"openai-research-debugging-2026","id":"openai-debugging-high-reference","kind":"other","logical_role":"unknown","logical_role_notes":"A capability reference does not establish recursive self-improvement.","metric_id":"score","qualitative_condition":"Indicative High capability threshold","source_refs":[{"locator":"§10.1.3.1, statement immediately following the evaluation description","source_id":"src-astra"}],"valid_from":null,"valid_to":null,"value":null}]
