[{"benchmark_id":"cobench","id":"cobench-prior-unspecified","methodology_notes":"Historical values restated in the Opus 5.5 report; original version equivalence is not established.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Reported score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":null,"source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}],"successor_id":"cobench-2-1","task_count":null,"task_population":null,"version_label":"Prior version (label unspecified)"},{"benchmark_id":"cobench","id":"cobench-2-1","methodology_notes":"One attempt per problem; environment drift within the reported comparison.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Reported score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":"cobench-prior-unspecified","release_date":{"precision":"day","value":"2026-09-22"},"source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}],"successor_id":null,"task_count":500,"task_population":null,"version_label":"2.1"},{"benchmark_id":"openai-research-debugging","id":"openai-research-debugging-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Success rate","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":41,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"kernelgen-1p","id":"kernelgen-1p-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Performance metric (exact scale unextracted)","unit":"not_reported","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"nanogpt","id":"nanogpt-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Normalized reward","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"posttrainbench-lite","id":"posttrainbench-lite-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Normalized reward","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":12,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"grb","id":"grb-2026-08","methodology_notes":"Task set includes known buggy tasks.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"pass1","name":"Average pass@1","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"month","value":"2026-08"},"source_refs":[{"locator":"GRB Internal Benchmark, printed p.37, PDF index 36","source_id":"src-grb"}],"successor_id":null,"task_count":74,"task_population":null,"version_label":"August 2026 report snapshot"},{"benchmark_id":"paperbench","id":"paperbench-v1","methodology_notes":"Full PaperBench, not Code-Dev.","metrics":[{"aggregation":"Mean weighted rubric completion across 20 papers and three runs per paper","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"replication","name":"Mean replication score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-04-02"},"source_refs":[{"locator":"§§2–3 and §§5.2–5.3; Tables 4–5","source_id":"src-paperbench"}],"successor_id":null,"task_count":20,"task_population":null,"version_label":"Original paper v1"},{"benchmark_id":"mle-bench","id":"mle-2024","methodology_notes":"Mean any-medal fraction per attempt; not pass@k.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"medal","name":"Any medal rate","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2024-10-09"},"source_refs":[{"locator":"§§2–3; Table 2","source_id":"src-mle"}],"successor_id":"mle-revised-2026","task_count":75,"task_population":null,"version_label":"Original paper v1 (2024)"},{"benchmark_id":"mle-bench","id":"mle-revised-2026","methodology_notes":"Replaces tasks and scores percentile rank against a generated reference distribution; up to three leaderboard submissions.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"percentile","name":"Reference-distribution percentile","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":"mle-2024","release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":72,"task_population":null,"version_label":"Revised — Astra card snapshot"},{"benchmark_id":"re-bench","id":"rebench-original","methodology_notes":"No chart heights digitized; retain source qualitative comparison.","metrics":[{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Reported assessment","unit":"qualitative","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2024-11-22"},"source_refs":[{"locator":"Environment descriptions; Results; What do we mean by time budget?","source_id":"src-rebench"}],"successor_id":null,"task_count":7,"task_population":null,"version_label":"Original November 2024 report"},{"benchmark_id":"re-bench","id":"rebench-gemini3","methodology_notes":"Two internet-requiring tasks omitted; 16 attempts × 2 hours; 24 runs used in bootstrap.","metrics":[{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Reported assessment","unit":"qualitative","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"month","value":"2025-11"},"source_refs":[{"locator":"Machine Learning R&D, printed pp.13–15, PDF indices 13–15","source_id":"src-gemini3"}],"successor_id":null,"task_count":5,"task_population":null,"version_label":"Google five-task subset, November 2025"},{"benchmark_id":"time-horizons","id":"th-1-0","methodology_notes":"Appendix estimates published together on January 29; evaluation dates not supplied.","metrics":[{"aggregation":"Source fitted 50% success horizon against human duration","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"p50","name":"50% task-completion horizon","unit":"minutes","valid_range":null}],"predecessor_id":null,"release_date":null,"source_refs":[{"locator":"Task-suite changes; Appendix: Changes to Model Horizon Estimates","source_id":"src-th11"}],"successor_id":"th-1-1","task_count":170,"task_population":null,"version_label":"TH1 (historical estimates)"},{"benchmark_id":"time-horizons","id":"th-1-1","methodology_notes":"Appendix estimates published together on January 29; evaluation dates not supplied.","metrics":[{"aggregation":"Source fitted 50% success horizon against human duration","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"p50","name":"50% task-completion horizon","unit":"minutes","valid_range":null}],"predecessor_id":"th-1-0","release_date":{"precision":"day","value":"2026-01-29"},"source_refs":[{"locator":"Task-suite changes; Appendix: Changes to Model Horizon Estimates","source_id":"src-th11"}],"successor_id":null,"task_count":228,"task_population":null,"version_label":"TH1.1"},{"benchmark_id":"ai4ai-bench","id":"ai4ai-v1","methodology_notes":"Ten algorithm families; fixed hidden evaluator retrains submissions.","metrics":[{"aggregation":"Mean task-normalized score across the system’s tested effort levels","direction":"higher","interpretation":"0.1 is repository baseline; 1 is task optimum.","metric_id":"normalized","name":"Mean normalized score","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-08-20"},"source_refs":[{"locator":"§§2–3; Figure 2 and Table 2","source_id":"src-ai4ai"}],"successor_id":null,"task_count":10,"task_population":null,"version_label":"Paper v1"},{"benchmark_id":"anthropic-rd-automation","id":"automation-aug26","methodology_notes":"378 leaf categories; work basket built from July records.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"ai_leads","name":"Work rated AI leads","unit":"percent","valid_range":{"max":100,"min":0}},{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"autonomous","name":"Work rated fully autonomous","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-17"},"source_refs":[{"locator":"§1 Measuring AI-led AI R&D; Appendix: Measuring AI-led R&D","source_id":"src-automation"}],"successor_id":null,"task_count":378,"task_population":null,"version_label":"August 2026 snapshot"},{"benchmark_id":"developer-productivity","id":"productivity-early25","methodology_notes":"16 developers; issue-level randomization.","metrics":[{"aggregation":"Source-reported aggregate","direction":"lower","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"time_change","name":"Change in completion time","unit":"percent","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-07-10"},"source_refs":[{"locator":"Methodology; Core Result","source_id":"src-productivity"}],"successor_id":"productivity-late25","task_count":246,"task_population":null,"version_label":"Early-2025 randomized trial"},{"benchmark_id":"developer-productivity","id":"productivity-late25","methodology_notes":"Selection effects prevent a reliable current effect estimate.","metrics":[{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Interpretability assessment","unit":"qualitative","valid_range":null}],"predecessor_id":"productivity-early25","release_date":{"precision":"day","value":"2026-02-24"},"source_refs":[{"locator":"Introduction; Wider adoption of AI has made it more difficult to measure task-level productivity","source_id":"src-productivity-update"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Late-2025 study update"},{"benchmark_id":"alphaevolve-training","id":"alphaevolve-2025","methodology_notes":"Two different denominators remain separate metrics.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"training_reduction","name":"Gemini training-time reduction","unit":"percent","valid_range":{"max":100,"min":0}},{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"kernel_speedup","name":"Kernel speedup","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-05-14"},"source_refs":[{"locator":"Designing better algorithms; Enhancing AI training and inference","source_id":"src-alphaevolve"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"May 2025 report"}]
