[{"aliases":[],"coverage_notes":"CoBench and operational automation included. Internal tasks are not public; lab disclosure is not independent replication.","id":"anthropic","name":"Anthropic","official_url":"https://www.anthropic.com/","organization_type":"frontier_lab","slug":"anthropic","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-opus55"}]},{"aliases":[],"coverage_notes":"Current AI self-improvement suite plus original MLE-bench and PaperBench. Graph-only scores withheld.","id":"openai","name":"OpenAI","official_url":"https://openai.com/","organization_type":"frontier_lab","slug":"openai","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-astra"}]},{"aliases":[],"coverage_notes":"GRB, a qualitative RE-Bench assessment and AlphaEvolve. Different internal metrics cannot rank labs.","id":"deepmind","name":"Google DeepMind","official_url":"https://deepmind.google/","organization_type":"frontier_lab","slug":"deepmind","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-grb"}]},{"aliases":[],"coverage_notes":"Independent evaluations, task horizons and productivity studies. Selected historical snapshots, not a live leaderboard.","id":"metr","name":"METR","official_url":"https://metr.org/","organization_type":"independent_evaluator","slug":"metr","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-rebench"}]},{"aliases":[],"coverage_notes":"Original benchmark-author report; no independent replication included.","id":"einsia","name":"AI4AI-Bench authors","official_url":"https://lab.einsia.ai/ai4ai/","organization_type":"benchmark_creator","slug":"einsia","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-ai4ai"}]},{"aliases":[],"coverage_notes":"Historical v1 study; later revisions require separate reconciliation.","id":"dgm-authors","name":"Darwin Gödel Machine authors","official_url":"https://arxiv.org/abs/2505.22954","organization_type":"academic_group","slug":"dgm-authors","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-dgm"}]},{"aliases":[],"coverage_notes":"Only the Llama 3.1 evaluation reported by MLE-bench authors is included.","id":"meta","name":"Meta","official_url":"https://ai.meta.com/","organization_type":"frontier_lab","slug":"meta","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-mle"}]},{"aliases":[],"coverage_notes":"Only Kimi K3 in the original AI4AI-Bench study is included.","id":"moonshot","name":"Moonshot AI","official_url":"https://www.moonshot.ai/","organization_type":"frontier_lab","slug":"moonshot","source_refs":[{"locator":"Publisher / authorship and relevant research sections","source_id":"src-ai4ai"}]},{"aliases":[],"coverage_notes":"Source-authored research, with classification separated from independent empirical verification.","id":"duan-authors","name":"Duan et al.","official_url":"https://arxiv.org/abs/2609.11873v1","organization_type":"academic_group","slug":"duan-authors","source_refs":[{"locator":"Title and authorship","source_id":"src-duan-v1"}]},{"aliases":[],"coverage_notes":"Source-authored research, with classification separated from independent empirical verification.","id":"weco","name":"Weco AI","official_url":"https://www.weco.ai/","organization_type":"other","slug":"weco","source_refs":[{"locator":"Title and authorship","source_id":"src-aide2"}]}]
