[{"applicability":"hypothesized","applicability_caveats":["Estimated carryover only. Reaching 85% would not establish researcher substitution."],"attributed_interpretation":"Anthropic expects a research-staff substitute to score at least 85% on the prior version and estimates carryover to 2.1.","benchmark_version_id":"cobench-2-1","id":"co85","kind":"estimated_necessary_capability","logical_role":"estimated_necessary_condition","logical_role_notes":"Necessary does not mean sufficient. No researcher-replacement percentage is calculated.","metric_id":"score","qualitative_condition":null,"source_refs":[{"locator":"§2.3.4.1, printed pp.36–37; PDF indices 35–36; Figure 2.3.4.1.A","source_id":"src-opus55"}],"valid_from":null,"valid_to":null,"value":85},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"Source prints 72.38%; represented as 0.7238 on its 0–1 reward scale. Best human solution, not average human ability.","benchmark_version_id":"nanogpt-2026","id":"nanogpt-human","kind":"human_baseline","logical_role":"measured_baseline","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"score","qualitative_condition":null,"source_refs":[{"locator":"§10.1.3.3","source_id":"src-astra"}],"valid_from":null,"valid_to":null,"value":0.7238},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"DeepMind uses 90% as a conservative CCL rule-out boundary, not proof that a system reaching it automates research.","benchmark_version_id":"grb-2026-08","id":"grb-policy","kind":"policy_threshold","logical_role":"policy_trigger","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"pass1","qualitative_condition":null,"source_refs":[{"locator":"Printed pp.36–37; PDF indices 35–36","source_id":"src-grb"}],"valid_from":null,"valid_to":null,"value":90},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"The original repository algorithm maps to 0.1; this is not a human-researcher threshold.","benchmark_version_id":"ai4ai-v1","id":"ai4ai-anchor","kind":"normalization_anchor","logical_role":"unknown","logical_role_notes":"Applies only to the specified metric and version.","metric_id":"normalized","qualitative_condition":null,"source_refs":[{"locator":"§2.5","source_id":"src-ai4ai"}],"valid_from":null,"valid_to":null,"value":0.1},{"applicability":"established","applicability_caveats":["Qualitative source assessment only; do not infer a numeric cutoff or distance to the threshold."],"attributed_interpretation":"OpenAI reports 78.05% as below its indicative High capability threshold. A numeric cutoff has not been verified in this record. This is not an RSI threshold.","benchmark_version_id":"openai-research-debugging-2026","id":"openai-debugging-high-reference","kind":"other","logical_role":"unknown","logical_role_notes":"A capability reference does not establish recursive self-improvement.","metric_id":"score","qualitative_condition":"Indicative High capability threshold","source_refs":[{"locator":"§10.1.3.1, statement immediately following the evaluation description","source_id":"src-astra"}],"valid_from":null,"valid_to":null,"value":null},{"applicability":"established","applicability_caveats":[],"attributed_interpretation":"1 represents the instruction-tuned reference model’s performance for each task.","benchmark_version_id":"posttrainbench-lite-2026","id":"posttrain-instruction-reference","kind":"normalization_anchor","logical_role":"unknown","logical_role_notes":"A scoring anchor, not a human baseline or a threshold for RSI.","metric_id":"score","qualitative_condition":null,"source_refs":[{"locator":"Section 10.1.3.4, reward definition","source_id":"src-astra"}],"valid_from":null,"valid_to":null,"value":1},{"applicability":"established","applicability_caveats":["The card does not restate the averaging or reference construction. This is not the average human researcher."],"attributed_interpretation":"Human reference on Google’s normalized scale.","benchmark_version_id":"rebench-google-feb2026","id":"rebench-google-human-anchor","kind":"normalization_anchor","logical_role":"unknown","logical_role_notes":"Normalization reference only.","metric_id":"human_normalized","qualitative_condition":null,"source_refs":[{"locator":"Printed page 9, Machine Learning R&D row (Deep Think mode): human-normalised average score","source_id":"src-gemini31-card"}],"valid_from":null,"valid_to":null,"value":1},{"applicability":"established","applicability_caveats":["Applies only to the three-paper human comparison subset, not full PaperBench or Code-Dev.","Human best-of-three after 48 tracked work hours (including unattended experiments) versus o1 extended 36-hour IterativeAgent run; selection and time accounting differ."],"attributed_interpretation":"ML PhD participants, best of three independent attempts per paper, achieved 41.4% after 48 tracked work hours (including unattended experiments) on the three-paper comparison subset. Participants worked part-time and could use AI assistants; not a single unaided human or a matched 36-hour trial.","benchmark_version_id":"paperbench-human-subset-v1","id":"paperbench-human-subset-best3","kind":"human_baseline","logical_role":"measured_baseline","logical_role_notes":"Measured subset reference, not an RSI threshold or an across-benchmark human baseline.","metric_id":"replication","qualitative_condition":null,"source_refs":[{"locator":"Introduction; §5.4 Human Baseline Performance; Figure 3 and caption","source_id":"src-paperbench"}],"valid_from":null,"valid_to":null,"value":41.4}]
