[{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_rd_capability"],"description":"Diagnose historical internal R&D failures from infrastructure snapshots.","framework_relevance":[{"evidence_aspect":"capability","framework_id":"duan-2026","level_codes":[],"rationale":"AI R&D task performance informs capability. This result alone does not establish persistent self-improvement, mechanism inheritance or L5.","source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}]}],"id":"cobench","lifecycle_status":"unknown","limitations":["Environment changed between versions; scores are not comparable across versions.","Private historical infrastructure and model-graded root-cause rubrics limit external replication."],"measurement_type":"benchmark","name":"CoBench","owner_id":"anthropic","related_family_ids":[],"rsi_relevance":"Measures a component of research engineering.","slug":"cobench","source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_rd_capability"],"description":"Resolve real bugs in internal research experiments.","framework_relevance":[{"evidence_aspect":"capability","framework_id":"duan-2026","level_codes":[],"rationale":"AI R&D task performance informs capability. This result alone does not establish persistent self-improvement, mechanism inheritance or L5.","source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}]}],"id":"openai-research-debugging","lifecycle_status":"unknown","limitations":["Internal suite; no cross-lab percentage comparison.","Not a demonstration of recursive improvement."],"measurement_type":"benchmark","name":"Internal Research Debugging Evaluation","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Targets a specific AI-development task.","slug":"openai-research-debugging","source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_system_improvement"],"description":"Optimize correct kernels for OpenAI first-party hardware.","framework_relevance":[],"id":"kernelgen-1p","lifecycle_status":"unknown","limitations":["Internal suite; no cross-lab percentage comparison.","Not a demonstration of recursive improvement."],"measurement_type":"benchmark","name":"KernelGen 1P","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Targets a specific AI-development task.","slug":"kernelgen-1p","source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_system_improvement"],"description":"Reduce small-model training time to a target validation objective.","framework_relevance":[],"id":"nanogpt","lifecycle_status":"unknown","limitations":["Internal suite; no cross-lab percentage comparison.","Not a demonstration of recursive improvement."],"measurement_type":"benchmark","name":"NanoGPT","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Targets a specific AI-development task.","slug":"nanogpt","source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_system_improvement"],"description":"Improve a pretrained model within a five-hour GPU budget.","framework_relevance":[],"id":"posttrainbench-lite","lifecycle_status":"unknown","limitations":["Internal suite; no cross-lab percentage comparison.","Not a demonstration of recursive improvement."],"measurement_type":"benchmark","name":"PostTrainBench Lite","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Targets a specific AI-development task.","slug":"posttrainbench-lite","source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_rd_capability"],"description":"End-to-end research engineering in an internal coding scaffold.","framework_relevance":[],"id":"grb","lifecycle_status":"unknown","limitations":["Approximately 20% of tasks had bugs that could cause false negatives.","Older result is a preliminary estimate."],"measurement_type":"benchmark","name":"GRB internal research engineering","owner_id":"deepmind","related_family_ids":[],"rsi_relevance":"Tests task chaining relevant to research automation.","slug":"grb","source_refs":[{"locator":"GRB Internal Benchmark, printed p.37, PDF index 36","source_id":"src-grb"}]},{"aliases":[],"availability":{"public_code":"yes","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"yes"},"categories":["ai_rd_capability"],"description":"Replicate research papers against author-developed rubrics.","framework_relevance":[],"id":"paperbench","lifecycle_status":"unknown","limitations":["Rubric credit is not the fraction of papers fully replicated.","Scaffold changes can reverse model ordering."],"measurement_type":"benchmark","name":"PaperBench","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Measures implementation and experimental replication.","slug":"paperbench","source_refs":[{"locator":"§§2–3 and §§5.2–5.3; Tables 4–5","source_id":"src-paperbench"}]},{"aliases":[],"availability":{"public_code":"yes","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"yes"},"categories":["ai_rd_capability"],"description":"ML competition engineering under bounded compute.","framework_relevance":[],"id":"mle-bench","lifecycle_status":"unknown","limitations":["Revised 2026 tasks and percentile scoring are incompatible with 2024 medal rates.","Public competition history creates contamination risk."],"measurement_type":"benchmark","name":"MLE-bench","owner_id":"openai","related_family_ids":[],"rsi_relevance":"Measures practical training and data-science work.","slug":"mle-bench","source_refs":[{"locator":"§§2–3; Table 2","source_id":"src-mle"}]},{"aliases":[],"availability":{"public_code":"yes","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"yes"},"categories":["ai_rd_capability"],"description":"Research engineering environments with expert baselines.","framework_relevance":[],"id":"re-bench","lifecycle_status":"unknown","limitations":["Seven selected environments do not cover the full research process.","Best-of-k allocations and subsets must be kept distinct."],"measurement_type":"benchmark","name":"RE-Bench","owner_id":"metr","related_family_ids":[],"rsi_relevance":"Compares agent and expert work under explicit resource budgets.","slug":"re-bench","source_refs":[{"locator":"Environment descriptions; Results; What do we mean by time budget?","source_id":"src-rebench"}]},{"aliases":[],"availability":{"public_code":"yes","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"yes"},"categories":["research_autonomy"],"description":"Human-task duration associated with 50% agent success.","framework_relevance":[],"id":"time-horizons","lifecycle_status":"unknown","limitations":["Human task duration is not uninterrupted AI runtime.","Task distribution and scaffold revisions change estimates."],"measurement_type":"benchmark","name":"Task-completion time horizons","owner_id":"metr","related_family_ids":[],"rsi_relevance":"Supporting measure of research and software-task autonomy.","slug":"time-horizons","source_refs":[{"locator":"Task-suite changes; Appendix: Changes to Model Horizon Estimates","source_id":"src-th11"}]},{"aliases":[],"availability":{"public_code":"yes","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"yes"},"categories":["ai_system_improvement"],"description":"Rewrite training algorithms in frozen research repositories.","framework_relevance":[{"evidence_aspect":"capability","framework_id":"duan-2026","level_codes":[],"rationale":"AI R&D task performance informs capability. This result alone does not establish persistent self-improvement, mechanism inheritance or L5.","source_refs":[{"locator":"§§2–3; Figure 2 and Table 2","source_id":"src-ai4ai"}]},{"evidence_aspect":"autonomy","framework_id":"duan-2026","level_codes":["L2"],"rationale":"Agents choose training-algorithm interventions within human-defined tasks and evaluation. This informs the strategy-selection component of L2; the benchmark alone does not establish a persistent autonomous improvement loop.","source_refs":[{"locator":"§§2–3; Figure 2 and Table 2","source_id":"src-ai4ai"}]}],"id":"ai4ai-bench","lifecycle_status":"unknown","limitations":["System means average different effort grids.","No multi-generation optimizer improvement is demonstrated by these task scores."],"measurement_type":"benchmark","name":"AI4AI-Bench","owner_id":"einsia","related_family_ids":[],"rsi_relevance":"Tests algorithm-design ability relevant to AI improving AI.","slug":"ai4ai-bench","source_refs":[{"locator":"§§2–3; Figure 2 and Table 2","source_id":"src-ai4ai"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["observed_rd_automation"],"description":"Model-rated automation across a fixed basket of R&D work.","framework_relevance":[],"id":"anthropic-rd-automation","lifecycle_status":"unknown","limitations":["Model judgment and approximate person-time weights.","Automation share is not a causal productivity estimate."],"measurement_type":"operational_metric","name":"Anthropic R&D Automation Index","owner_id":"anthropic","related_family_ids":[],"rsi_relevance":"Describes actual reported organizational use.","slug":"anthropic-rd-automation","source_refs":[{"locator":"§1 Measuring AI-led AI R&D; Appendix: Measuring AI-led R&D","source_id":"src-automation"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["observed_rd_automation"],"description":"Randomized access to AI tools on real repository issues.","framework_relevance":[],"id":"developer-productivity","lifecycle_status":"unknown","limitations":["Experienced developers in familiar open-source repositories; not representative of all work.","Later study reports selection bias, so no unqualified trend."],"measurement_type":"operational_metric","name":"Experienced developer productivity","owner_id":"metr","related_family_ids":[],"rsi_relevance":"Operational counterpoint to benchmark performance.","slug":"developer-productivity","source_refs":[{"locator":"Methodology; Core Result","source_id":"src-productivity"}]},{"aliases":[],"availability":{"public_code":"unknown","public_description":"yes","public_evaluation_service":"unknown","public_results":"yes","public_tasks":"no"},"categories":["ai_system_improvement","observed_rd_automation"],"description":"Reported deployment of AI-discovered training kernels.","framework_relevance":[],"id":"alphaevolve-training","lifecycle_status":"unknown","limitations":["Reported by developer; experimental compute and full production denominator undisclosed.","A fixed optimizer improving code does not establish recursive acceleration."],"measurement_type":"operational_metric","name":"AlphaEvolve training optimization","owner_id":"deepmind","related_family_ids":[],"rsi_relevance":"Connects algorithm search to model-training efficiency.","slug":"alphaevolve-training","source_refs":[{"locator":"Designing better algorithms; Enhancing AI training and inference","source_id":"src-alphaevolve"}]}]
