[{"benchmark_id":"cobench","id":"cobench-prior-unspecified","methodology_notes":"Historical values restated in the Opus 5.5 report; original version equivalence is not established.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Reported score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":null,"source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}],"successor_id":"cobench-2-1","task_count":null,"task_population":null,"version_label":"Prior version (label unspecified)"},{"benchmark_id":"cobench","id":"cobench-2-1","methodology_notes":"One attempt per problem; environment drift within the reported comparison.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Reported score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":"cobench-prior-unspecified","release_date":{"precision":"day","value":"2026-09-22"},"source_refs":[{"locator":"§2.3.4.1, Figure 2.3.4.1.A; printed pp.36–37, PDF indices 35–36","source_id":"src-opus55"}],"successor_id":null,"task_count":500,"task_population":null,"version_label":"2.1"},{"benchmark_id":"openai-research-debugging","id":"openai-research-debugging-2026","methodology_notes":"Disclosure snapshot. The source describes 41 research bugs and six alignment-auditing tasks, but does not explicitly identify the plotted mean’s scored denominator.","metrics":[{"aggregation":"Source-reported mean rubric reward; Figure 84 scored population not explicitly mapped to disclosed task components","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"score","name":"Mean rubric reward","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":null,"task_population":"Source describes 41 real research bugs plus six alignment-auditing tasks; exact scored population for Appendix Figure 84 is not specified.","version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"kernelgen-1p","id":"kernelgen-1p-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Mean rubric reward, as labelled in Figure 85","direction":"higher","interpretation":"Mean score on the kernel-optimization rubric, displayed as a percentage. It is not a percentage reduction in runtime or a task-success rate.","metric_id":"score","name":"Mean kernel optimization reward","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"nanogpt","id":"nanogpt-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Mean reward at each output-token budget","direction":"higher","interpretation":"Reduction in training time to the target objective, divided by baseline training time and limited to 0–1; 0 means no improvement and 1 would mean zero training time.","metric_id":"score","name":"Fraction of baseline training time saved","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"posttrainbench-lite","id":"posttrainbench-lite-2026","methodology_notes":"Version identity is this disclosure snapshot, not an invented upstream version.","metrics":[{"aggregation":"Mean normalized reward across 12 tasks, at each output-token budget","direction":"higher","interpretation":"Improvement over the base model divided by the gap to an instruction-tuned reference, limited to 0–1. This is not raw downstream accuracy.","metric_id":"score","name":"Gain toward instruction-tuned reference performance","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":12,"task_population":null,"version_label":"Astra card snapshot, September 2026"},{"benchmark_id":"grb","id":"grb-2026-08","methodology_notes":"Task set includes known buggy tasks.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"pass1","name":"Average pass@1","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"month","value":"2026-08"},"source_refs":[{"locator":"GRB Internal Benchmark, printed p.37, PDF index 36","source_id":"src-grb"}],"successor_id":null,"task_count":74,"task_population":null,"version_label":"August 2026 report snapshot"},{"benchmark_id":"paperbench","id":"paperbench-v1","methodology_notes":"Full PaperBench, not Code-Dev.","metrics":[{"aggregation":"Mean weighted rubric completion across 20 papers and three runs per paper","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"replication","name":"Mean replication score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-04-02"},"source_refs":[{"locator":"§§2–3 and §§5.2–5.3; Tables 4–5","source_id":"src-paperbench"}],"successor_id":null,"task_count":20,"task_population":null,"version_label":"Original paper v1"},{"benchmark_id":"mle-bench","id":"mle-2024","methodology_notes":"Mean any-medal fraction per attempt; not pass@k.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"medal","name":"Any medal rate","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2024-10-09"},"source_refs":[{"locator":"§§2–3; Table 2","source_id":"src-mle"}],"successor_id":"mle-revised-2026","task_count":75,"task_population":null,"version_label":"Original paper v1 (2024)"},{"benchmark_id":"mle-bench","id":"mle-revised-2026","methodology_notes":"Replaces tasks and scores percentile rank against a generated reference distribution; up to three leaderboard submissions.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"Percentile rank against a generated distribution of strong solutions; not the percentage of competitions earning a medal.","metric_id":"percentile","name":"Mean percentile against reference solutions","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":"mle-2024","release_date":{"precision":"day","value":"2026-09-03"},"source_refs":[{"locator":"§10.1.3 and §§10.1.3.1–10.1.3.5; Table 18","source_id":"src-astra"}],"successor_id":null,"task_count":72,"task_population":null,"version_label":"Revised — Astra card snapshot"},{"benchmark_id":"re-bench","id":"rebench-original","methodology_notes":"Original seven-environment benchmark. Scores normalize the provided starting solution to 0 and the task reference solution to 1; values can exceed 1. Chart extraction limitations do not make the benchmark qualitative.","metrics":[{"aggregation":"Mean across seven tasks for the specified budget and run-allocation rule","direction":"higher","interpretation":"0 is the starting solution and 1 the task reference solution. Below-starting scores are floored at 0. Values may exceed 1; this is not a universal average-human score.","metric_id":"normalized","name":"Mean normalized task score","unit":"normalized_score","valid_range":{"max":null,"min":0}},{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Reported assessment","unit":"qualitative","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2024-11-22"},"source_refs":[{"locator":"Environment descriptions; Results; What do we mean by time budget?","source_id":"src-rebench"}],"successor_id":null,"task_count":7,"task_population":null,"version_label":"Original November 2024 report"},{"benchmark_id":"re-bench","id":"rebench-gemini3","methodology_notes":"Two internet-requiring tasks omitted; 16 attempts × 2 hours; 24 runs used in bootstrap.","metrics":[{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Reported assessment","unit":"qualitative","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"month","value":"2025-11"},"source_refs":[{"locator":"Machine Learning R&D, printed pp.13–15, PDF indices 13–15","source_id":"src-gemini3"}],"successor_id":null,"task_count":5,"task_population":null,"version_label":"Google five-task subset, November 2025"},{"benchmark_id":"time-horizons","id":"th-1-0","methodology_notes":"Appendix estimates published together on January 29; evaluation dates not supplied.","metrics":[{"aggregation":"Source fitted 50% success horizon against human duration","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"p50","name":"50% task-completion horizon","unit":"minutes","valid_range":null}],"predecessor_id":null,"release_date":null,"source_refs":[{"locator":"Task-suite changes; Appendix: Changes to Model Horizon Estimates","source_id":"src-th11"}],"successor_id":"th-1-1","task_count":170,"task_population":null,"version_label":"TH1 (historical estimates)"},{"benchmark_id":"time-horizons","id":"th-1-1","methodology_notes":"Appendix estimates published together on January 29; evaluation dates not supplied.","metrics":[{"aggregation":"Source fitted 50% success horizon against human duration","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"p50","name":"50% task-completion horizon","unit":"minutes","valid_range":null}],"predecessor_id":"th-1-0","release_date":{"precision":"day","value":"2026-01-29"},"source_refs":[{"locator":"Task-suite changes; Appendix: Changes to Model Horizon Estimates","source_id":"src-th11"}],"successor_id":null,"task_count":228,"task_population":null,"version_label":"TH1.1"},{"benchmark_id":"ai4ai-bench","id":"ai4ai-v1","methodology_notes":"Ten algorithm families; fixed hidden evaluator retrains submissions.","metrics":[{"aggregation":"Mean task-normalized score across the system’s tested effort levels","direction":"higher","interpretation":"0.1 is repository baseline; 1 is task optimum.","metric_id":"normalized","name":"Mean normalized score","unit":"normalized_score","valid_range":{"max":1,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-08-20"},"source_refs":[{"locator":"§§2–3; Figure 2 and Table 2","source_id":"src-ai4ai"}],"successor_id":null,"task_count":10,"task_population":null,"version_label":"Paper v1"},{"benchmark_id":"anthropic-rd-automation","id":"automation-aug26","methodology_notes":"378 leaf categories; work basket built from July records.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"ai_leads","name":"Work rated AI leads","unit":"percent","valid_range":{"max":100,"min":0}},{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"autonomous","name":"Work rated fully autonomous","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-09-17"},"source_refs":[{"locator":"§1 Measuring AI-led AI R&D; Appendix: Measuring AI-led R&D","source_id":"src-automation"}],"successor_id":null,"task_count":378,"task_population":null,"version_label":"August 2026 snapshot"},{"benchmark_id":"developer-productivity","id":"productivity-early25","methodology_notes":"16 developers; issue-level randomization.","metrics":[{"aggregation":"Source-reported aggregate","direction":"lower","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"time_change","name":"Change in completion time","unit":"percent","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-07-10"},"source_refs":[{"locator":"Methodology; Core Result","source_id":"src-productivity"}],"successor_id":"productivity-late25","task_count":246,"task_population":null,"version_label":"Early-2025 randomized trial"},{"benchmark_id":"developer-productivity","id":"productivity-late25","methodology_notes":"Selection effects prevent a reliable current effect estimate.","metrics":[{"aggregation":"Source-reported aggregate","direction":"neither","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"assessment","name":"Interpretability assessment","unit":"qualitative","valid_range":null},{"aggregation":"Raw cohort-specific time-change estimate","direction":"lower","interpretation":"Negative means faster completion. Not a reliable population effect because of selection bias.","metric_id":"time_change","name":"AI-assisted task-time change","unit":"percent","valid_range":null}],"predecessor_id":"productivity-early25","release_date":{"precision":"day","value":"2026-02-24"},"source_refs":[{"locator":"Introduction; Wider adoption of AI has made it more difficult to measure task-level productivity","source_id":"src-productivity-update"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Late-2025 study update"},{"benchmark_id":"alphaevolve-training","id":"alphaevolve-2025","methodology_notes":"Two different denominators remain separate metrics.","metrics":[{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"training_reduction","name":"Gemini training-time reduction","unit":"percent","valid_range":{"max":100,"min":0}},{"aggregation":"Source-reported aggregate","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"kernel_speedup","name":"Kernel speedup","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-05-14"},"source_refs":[{"locator":"Designing better algorithms; Enhancing AI training and inference","source_id":"src-alphaevolve"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"May 2025 report"},{"benchmark_id":"re-bench","id":"rebench-google-feb2026","methodology_notes":"Disclosure snapshot of Google’s reported human-normalised average. The model card does not repeat the task subset, budget or normalization formula; these are not assumed to match the original METR or November 2025 Google protocols.","metrics":[{"aggregation":"Source-reported average; exact task population and weighting not restated","direction":"higher","interpretation":"Google reports a human-normalised average. A score above 1 is possible; it is not a percentage of tasks solved or evidence of RSI.","metric_id":"human_normalized","name":"Human-normalised average score","unit":"normalized_score","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-02-19"},"source_refs":[{"locator":"Printed page 9, Machine Learning R&D row (Deep Think mode): human-normalised average score","source_id":"src-gemini31-card"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"Google model-card assessment, February 2026"},{"benchmark_id":"paperbench","id":"paperbench-code-dev-v1","methodology_notes":"Code Development rubric nodes only; execution and result-match requirements omitted. Not comparable as full PaperBench replication.","metrics":[{"aggregation":"Mean score over Code Development rubric nodes across the 20 papers; execution and result-match nodes omitted","direction":"higher","interpretation":"A Code-Dev-only score; not a full paper-replication percentage or percentage of RSI achieved.","metric_id":"replication","name":"Mean Code Development score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-04-02"},"source_refs":[{"locator":"§2.6, §5.3 Table 6, PaperBench Code-Dev","source_id":"src-paperbench"}],"successor_id":null,"task_count":20,"task_population":null,"version_label":"Original paper v1, Code-Dev variant"},{"benchmark_id":"time-horizons","id":"th-1-1-sol-2026","methodology_notes":"METR's June 2026 TH1.1 software-task assessment uses the ReAct harness family; exact revision/configuration and task count are not stated. Cheating attempts scored as failures. Approximate 50%-horizon estimate, which METR does not consider robust. Source hours retained.","metrics":[{"aggregation":"Source fitted 50% success horizon against human duration","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"p50","name":"50% task-completion horizon","unit":"hours","valid_range":null}],"predecessor_id":null,"release_date":{"precision":"day","value":"2026-06-26"},"source_refs":[{"locator":"50%-Time Horizon paragraph, cheating attempts scored as failures","source_id":"src-metr-gpt56-sol"}],"successor_id":null,"task_count":null,"task_population":null,"version_label":"TH1.1, GPT-5.6 Sol June 2026 assessment"},{"benchmark_id":"paperbench","id":"paperbench-human-subset-v1","methodology_notes":"Agent and human use the full PaperBench rubric on the same three-paper subset, but are not time-matched: o1 IterativeAgent 36-hour run versus human best of three after 48 tracked work hours (including unattended experiments) in part-time arrangements. Not full 20-paper or Code-Dev score.","metrics":[{"aggregation":"Mean full-rubric replication score on the three-paper human comparison subset","direction":"higher","interpretation":"A task-specific measurement; not a percentage of RSI achieved.","metric_id":"replication","name":"Three-paper subset replication score","unit":"percent","valid_range":{"max":100,"min":0}}],"predecessor_id":null,"release_date":{"precision":"day","value":"2025-04-02"},"source_refs":[{"locator":"Introduction; §5.4 Human Baseline Performance; Figure 3 and caption","source_id":"src-paperbench"}],"successor_id":null,"task_count":3,"task_population":"Three of four human-baselined papers; excludes test-time-model-adaptation, whose human attempt ended at 24 hours.","version_label":"Original paper v1, three-paper human comparison subset"}]
