[{"event_date":{"precision":"day","value":"2026-09-22"},"id":"indexed-opus55-card","linked_record_ids":["opus55-card","cobench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-opus55"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Claude Opus 5.5 System Card","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-03"},"id":"indexed-astra-card","linked_record_ids":["astra-card","openai-research-debugging","kernelgen-1p","nanogpt","posttrainbench-lite","mle-bench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-astra"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: GPT-6 Astra System Card","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"month","value":"2026-08"},"id":"indexed-gemini37-report","linked_record_ids":["gemini37-report","grb"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-grb"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Gemini 3.7 Flash Frontier Safety Framework Report","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-04-02"},"id":"indexed-paperbench-paper","linked_record_ids":["paperbench-paper","paperbench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-paperbench"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: PaperBench: Evaluating AI’s Ability to Replicate AI Research","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2024-10-09"},"id":"indexed-mle-paper","linked_record_ids":["mle-paper","mle-bench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-mle"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2024-11-22"},"id":"indexed-rebench-report","linked_record_ids":["rebench-report","re-bench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-rebench"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Evaluating frontier AI R&D capabilities of language model agents against human experts","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-01-29"},"id":"indexed-time-horizon-revision","linked_record_ids":["time-horizon-revision","time-horizons"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-th11"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Time Horizon 1.1","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-07-10"},"id":"indexed-productivity-trial","linked_record_ids":["productivity-trial","developer-productivity"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-productivity"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-02-24"},"id":"indexed-productivity-redesign","linked_record_ids":["productivity-redesign","developer-productivity"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-productivity-update"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: We are Changing our Developer Productivity Experiment Design","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-06-05"},"id":"indexed-reward-hacking","linked_record_ids":["reward-hacking","re-bench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-hacking"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Recent Frontier Models Are Reward Hacking","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-08-20"},"id":"indexed-ai4ai-paper","linked_record_ids":["ai4ai-paper","ai4ai-bench"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-ai4ai"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: AI4AI-Bench: Benchmarking LLM Agents in Algorithmic Design for Recursive Self-Improvement","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-17"},"id":"indexed-automation-measurements","linked_record_ids":["automation-measurements","anthropic-rd-automation"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-automation"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Measurements for understanding the pace of AI development inside frontier labs","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-05-29"},"id":"indexed-dgm-paper","linked_record_ids":["dgm-paper"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-dgm"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: Darwin Gödel Machine: Open-Ended Evolution of Self-Improving Agents","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-05-14"},"id":"indexed-alphaevolve-report","linked_record_ids":["alphaevolve-report","alphaevolve-training"],"linked_source_refs":[{"locator":"Original publication; linked record contains exact result locators","source_id":"src-alphaevolve"}],"prior_id":null,"replacement_id":null,"summary":"Indexed: AlphaEvolve: A Gemini-powered coding agent for designing advanced algorithms","tracker_added_date":{"precision":"day","value":"2026-09-25"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-10"},"id":"update-autonomy-framework","linked_record_ids":["duan-paper","duan-sep2026"],"linked_source_refs":[{"locator":"Supplied v1, §§3.1–3.7; Figure 9 p.36; §§4.1–4.4","source_id":"src-duan-v1"}],"prior_id":null,"replacement_id":null,"summary":"Added the source-attributed B0–L5 framework and historical domain assessment.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-dgm-model-attribution","linked_record_ids":["dgm","dgm-study"],"linked_source_refs":[{"locator":"v1 §4.1 Experiment Setup","source_id":"src-dgm"}],"prior_id":null,"replacement_id":null,"summary":"Corrected DGM model attribution: Claude 3.5 Sonnet (New) handles self-modification and SWE-bench evaluation; o3-mini handles Polyglot evaluation. Reported scores are unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-kernelgen-1p-astra","linked_record_ids":["kernelgen-1p-astra","kernelgen-appendix-astra"],"linked_source_refs":[{"locator":"Appendix A.8.1.3.2, Figure 85 (PDF printed page 152); exact bar labels; section 10.1.3.2","source_id":"src-astra"}],"prior_id":"kernelgen-1p-astra","replacement_id":"kernelgen-appendix-astra","summary":"Transcribed KernelGen Figure 85; the previous missing-result entry was a tracker omission, not an absent published score.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-mle-revised-astra","linked_record_ids":["mle-revised-astra","mle-revised-max-astra"],"linked_source_refs":[{"locator":"Section 10.1.3.5, Figure 56, printed page 105; GPT-6 Astra (max)","source_id":"src-astra"}],"prior_id":"mle-revised-astra","replacement_id":"mle-revised-max-astra","summary":"Transcribed the explicitly labelled Revised MLE-bench score; preserved the separate historical medal-rate series.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-nanogpt-astra","linked_record_ids":["nanogpt-astra","nanogpt-curve-astra"],"linked_source_refs":[{"locator":"Section 10.1.3.3, Figure 54; GPT-6 Astra series","source_id":"src-astra"}],"prior_id":"nanogpt-astra","replacement_id":"nanogpt-curve-astra","summary":"Replaced an incorrect missing-result label with the source’s published budget-curve evidence. Exact point values remain unavailable in the checked source.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-posttrainbench-lite-astra","linked_record_ids":["posttrainbench-lite-astra","posttrainbench-lite-curve-astra"],"linked_source_refs":[{"locator":"Section 10.1.3.4, Figure 55; GPT-6 Astra series","source_id":"src-astra"}],"prior_id":"posttrainbench-lite-astra","replacement_id":"posttrainbench-lite-curve-astra","summary":"Replaced an incorrect missing-result label with the source’s published budget-curve evidence. Exact point values remain unavailable in the checked source.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-rebench-metric","linked_record_ids":["rebench-original"],"linked_source_refs":[{"locator":"Environment descriptions; Results; What do we mean by time budget?","source_id":"src-rebench"}],"prior_id":null,"replacement_id":null,"summary":"Corrected the tracker’s original RE-Bench metric definition to include normalized numerical scores; no upstream benchmark revision is implied.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-02-19"},"id":"backfill-gemini31-rebench","linked_record_ids":["gemini31-rebench-score","gemini3-rebench-comparator-score"],"linked_source_refs":[{"locator":"Printed page 9, Machine Learning R&D row (Deep Think mode): human-normalised average score","source_id":"src-gemini31-card"}],"prior_id":null,"replacement_id":null,"summary":"Added Google’s February 2026 numerical RE-Bench assessment.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-code-dev-metadata","linked_record_ids":["paperbench-code-dev-v1","pb-code-dev-iterative"],"linked_source_refs":[{"locator":"§2.6, §5.3 Table 6","source_id":"src-paperbench"}],"prior_id":null,"replacement_id":null,"summary":"Corrected the local Code-Dev metric label and removed a copied run count not stated for Table 6; its score is unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-th-sol-metadata","linked_record_ids":["th-1-1-sol-2026","th-1-1-sol-p","th-1-1-sol-gpt56"],"linked_source_refs":[{"locator":"Summary, TH1.1 paragraphs","source_id":"src-metr-gpt56-sol"}],"prior_id":null,"replacement_id":null,"summary":"Removed January task-count and scaffold details not stated for the June assessment; retained METR’s non-robustness warning.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-ai4ai-medium-subject","linked_record_ids":["a4-opus-medium","a4-opus-medium-mean"],"linked_source_refs":[{"locator":"§3.2, prose preceding Figure 2","source_id":"src-ai4ai"}],"prior_id":null,"replacement_id":null,"summary":"Assigned the medium-effort configuration its own system identity instead of the all-effort aggregate identity; score unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-03"},"id":"backfill-debugging-figure84","linked_record_ids":["debug-gpt6-sol-card","debug-gpt56-sol-card","debug-gpt6-luna-card","debug-gpt56-luna-card"],"linked_source_refs":[{"locator":"Appendix Figure 84","source_id":"src-astra"}],"prior_id":null,"replacement_id":null,"summary":"Indexed four exact comparator labels from the Astra-card debugging figure; this is historical evidence, not a new model release.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-04-02"},"id":"backfill-paperbench-original-tables","linked_record_ids":["pb-deepseek-result","pb-o1-code-dev-result"],"linked_source_refs":[{"locator":"§5.2 Table 4; §5.3 Table 6","source_id":"src-paperbench"}],"prior_id":null,"replacement_id":null,"summary":"Indexed omitted original PaperBench and distinct Code-Dev results from the 2025 paper.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2024-10-09"},"id":"backfill-mle-original-scaffolds","linked_record_ids":["mle-4o-mlab-result","mle-4o-openhands-result"],"linked_source_refs":[{"locator":"§3.1 Table 2","source_id":"src-mle"}],"prior_id":null,"replacement_id":null,"summary":"Indexed original MLE-bench GPT-4o scaffold comparisons from the 2024 paper.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-01-29"},"id":"backfill-th11-comparison-table","linked_record_ids":["th-1-0-th-codexmax","th-1-0-th-sonnet45","th-1-0-th-opus41","th-1-0-th-grok4","th-1-0-th-sonnet4","th-1-0-th-gpt4-1106","th-1-0-th-gpt4-0314","th-1-1-th-gpt4-1106","th-1-1-th-gpt4-0314"],"linked_source_refs":[{"locator":"Appendix: Changes to Model Horizon Estimates table","source_id":"src-th11"}],"prior_id":null,"replacement_id":null,"summary":"Completed the printed TH1/TH1.1 comparison-table rows omitted from the tracker’s historical snapshot.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-06-26"},"id":"backfill-th11-gpt56-sol","linked_record_ids":["th-1-1-sol-gpt56"],"linked_source_refs":[{"locator":"Summary, TH1.1 paragraphs","source_id":"src-metr-gpt56-sol"}],"prior_id":null,"replacement_id":null,"summary":"Indexed METR’s conditional, non-robust GPT-5.6 Sol time-horizon estimate with its uncertainty.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-02-24"},"id":"backfill-productivity-late25-cohorts","linked_record_ids":["productivity-returning","productivity-new"],"linked_source_refs":[{"locator":"Introduction, raw results paragraph","source_id":"src-productivity-update"}],"prior_id":null,"replacement_id":null,"summary":"Indexed late-2025 raw cohort estimates, preserving METR’s selection-bias warning.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-08-20"},"id":"backfill-ai4ai-opus-medium","linked_record_ids":["a4-opus-medium-mean"],"linked_source_refs":[{"locator":"§3.2, prose preceding Figure 2","source_id":"src-ai4ai"}],"prior_id":null,"replacement_id":null,"summary":"Indexed the paper’s separate Claude Opus 5 medium-effort result, distinct from its across-effort mean.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2025-04-02"},"id":"backfill-paperbench-human-subset","linked_record_ids":["paperbench-human-subset-v1","pb-o1-human-subset-result","paperbench-human-subset-best3"],"linked_source_refs":[{"locator":"Introduction; §5.4 Human Baseline Performance; Figure 3 and caption","source_id":"src-paperbench"}],"prior_id":null,"replacement_id":null,"summary":"Indexed the paper’s three-paper agent result and human best-of-three reference with different time accounting; no upstream benchmark change.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"newly_indexed_historical_evidence"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-metr-react-family","linked_record_ids":["th-1-1-sol-p","th-gpt56-sol","th-1-1-sol-gpt56"],"linked_source_refs":[{"locator":"Summary, paragraph identifying ReAct agent harness","source_id":"src-metr-gpt56-sol"}],"prior_id":null,"replacement_id":null,"summary":"Qualified the earlier metadata correction: METR discloses the ReAct harness family but not its exact revision, configuration or task count; estimate unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-debugging-population","linked_record_ids":["openai-research-debugging-2026","openai-research-debugging-protocol"],"linked_source_refs":[{"locator":"§10.1.3.1; Appendix Figure 84","source_id":"src-astra"}],"prior_id":null,"replacement_id":null,"summary":"Preserved the disclosed 41-bug and six-alignment-task components while removing an unsupported assertion that Figure 84 averaged exactly 41 tasks; scores unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"},{"event_date":{"precision":"day","value":"2026-09-26"},"id":"correct-standard-error-level","linked_record_ids":["pb-o3-result","pb-4o-result","pb-gemini-result","pb-o1-result","pb-claude-result","pb-o3-iter-result","pb-claude-iter-result","pb-o1-iter-result","pb-o1-iter36-result","mle-o1-result","mle-4o-result","mle-llama-result","mle-claude-result","pb-deepseek-result","pb-o1-code-dev-result","mle-4o-mlab-result","mle-4o-openhands-result"],"linked_source_refs":[{"locator":"Tables 4–6; reported standard errors","source_id":"src-paperbench"},{"locator":"Table 2; reported standard errors","source_id":"src-mle"}],"prior_id":null,"replacement_id":null,"summary":"Cleared a metadata field that repeated the standard-error amplitude as if it were a confidence level; printed values and reported standard errors are unchanged.","tracker_added_date":{"precision":"day","value":"2026-09-26"},"type":"corrected_result"}]
