[{"acceleration_summary":"Not established by this survey’s reviewed evidence under comparable resources.","as_of":{"precision":"day","value":"2026-09-10"},"assessment_ids":["duan-science-frontier","duan-science-emerging","duan-embodied_intelligence-frontier","duan-embodied_intelligence-emerging","duan-software_engineering-frontier","duan-software_engineering-emerging","duan-healthcare-frontier","duan-healthcare-emerging"],"assessor":"Duan et al.","broad_frontier":"Lower and intermediate levels have broad evidence; L3–L4 are more domain-dependent.","domain_frontiers":[{"domain":"science","emerging_level_codes":["L3"],"evidence_quality":"Survey-author synthesis; preprint; heterogeneous source studies. Representative examples outside indexed records are not independently primary-audited here.","external_controls":["Scientific objectives","Validation, reproducibility and safety constraints"],"representative_system_ids":[],"source_refs":[{"locator":"§4.1.2–4.1.4; Figure 9 p.36","source_id":"src-duan-v1"}],"strongest_level_codes":["L2"],"summary":"Strong L2; early L3. Survey examples include CASCADE and CORAL; these examples are survey coverage, not independently audited system records."},{"domain":"embodied_intelligence","emerging_level_codes":["L4"],"evidence_quality":"Survey-author synthesis; preprint; heterogeneous source studies. Representative examples outside indexed records are not independently primary-audited here.","external_controls":["Environment interfaces","Evaluators","Physical safety boundaries"],"representative_system_ids":[],"source_refs":[{"locator":"§4.2.1–4.2.4; Figure 9 p.36","source_id":"src-duan-v1"}],"strongest_level_codes":["L2","L3"],"summary":"L2–L3 main frontier; simulated L4 evidence. Survey examples include Voyager, EnvGen and POET; physical-world integration remains limited."},{"domain":"software_engineering","emerging_level_codes":["L3","L5"],"evidence_quality":"Survey-author synthesis; preprint; heterogeneous source studies. Representative examples outside indexed records are not independently primary-audited here.","external_controls":["Objectives and specifications","Benchmark tests","Archive and parent-selection machinery"],"representative_system_ids":["dgm"],"source_refs":[{"locator":"§4.3; Figure 9 p.36","source_id":"src-duan-v1"}],"strongest_level_codes":["L2"],"summary":"Mature L2; emerging L3; bounded structural L5. The survey cites Self-Harness, learner-conditioned SWE systems, SICA and DGM; L4 deployment adaptation remains largely absent."},{"domain":"healthcare","emerging_level_codes":["L3","L4"],"evidence_quality":"Survey-author synthesis; preprint; heterogeneous source studies. Representative examples outside indexed records are not independently primary-audited here.","external_controls":["Patient populations and task distribution","Clinical evaluation","Deployment and safety gates"],"representative_system_ids":[],"source_refs":[{"locator":"§4.4.2–4.4.4; Figure 9 p.36","source_id":"src-duan-v1"}],"strongest_level_codes":["L2"],"summary":"Mature L2; limited L3-like evidence; simulated L4. Survey examples include EvoClinician and EvoPatient; this is not evidence of autonomous clinical deployment."}],"effective_l5_summary":"Some bounded meta-improvement and transfer; reliable multi-generation accumulation remains unresolved.","extraction_review":{"notes":"Checked against original source; not human review or independent experimental replication.","reviewed_at":"2026-09-26T05:09:03Z","reviewer":"Codex","status":"agent_checked"},"framework_id":"duan-2026","id":"duan-sep2026","label":"Duan et al. (Sep. 2026) assessment","limitations":["Historical supplied v1 assessment, not a live independent verdict.","No single global stage. Domains and system boundaries differ.","The survey’s representative examples are not all independently audited here."],"publication_status":"published","snapshot_type":"source_author_assessment","source_refs":[{"locator":"Supplied v1, §§3.1–3.7; Figure 9 p.36; §§4.1–4.4","source_id":"src-duan-v1"}],"structural_l5_summary":"Bounded prototypes and emerging research or industrial systems.","supersedes_id":null},{"acceleration_summary":"Not established in this evidence set; not a claim that no other evidence exists.","as_of":{"precision":"day","value":"2026-09-26"},"assessment_ids":["tracker-aide2-l5"],"assessor":"RSI Tracker (Codex evidence synthesis)","broad_frontier":"Coverage is insufficient for an independent broad or global stage assignment.","domain_frontiers":[{"domain":"ai_rd_frontier_model_development","emerging_level_codes":[],"evidence_quality":"Selected company-authored preprint; checked extraction, no independent replication.","external_controls":["Protected evaluation","Task families","Model substrate","Cost and release authority"],"representative_system_ids":["aide2"],"source_refs":[{"locator":"v1 §§2–3.6, especially ignition test §3.6","source_id":"src-aide2"}],"strongest_level_codes":["L5"],"summary":"Bounded structural L5 in AIDE²; no representative domain-wide frontier assigned. Capability-only lab benchmarks do not establish autonomous inheritance."}],"effective_l5_summary":"Partial evidence of better research harnesses; outer-improver advantage inconclusive.","extraction_review":{"notes":"Checked against original source; not human review or independent experimental replication.","reviewed_at":"2026-09-26T05:09:03Z","reviewer":"Codex","status":"agent_checked"},"framework_id":"duan-2026","id":"tracker-ai-rd-sep2026","label":"RSI Tracker: selected AI R&D evidence (Sep. 2026)","limitations":["Selective evidence set, not an exhaustive literature review.","A categorical L5 mechanism is not a scalar capability score."],"publication_status":"published","snapshot_type":"tracker_evidence_synthesis","source_refs":[{"locator":"v1 §§2–3.6, especially ignition test §3.6","source_id":"src-aide2"}],"structural_l5_summary":"Bounded mechanism reuse in the selected AIDE² experiment.","supersedes_id":null},{"acceleration_summary":"Sustained recursive acceleration is not established in these selected studies.","as_of":{"precision":"day","value":"2026-09-26"},"assessment_ids":[],"assessor":"RSI Tracker (Codex evidence synthesis)","broad_frontier":"Robust full-loop RSI is not established in the reviewed evidence.","domain_frontiers":[],"effective_l5_summary":"Evidence that future improvement becomes better remains partial or inconclusive.","evidence_overview":{"claims":[{"aspect":"capability","brief":"Research and engineering gains appear in selected evaluations.","recursive_evidence_ids":["rec-aide2"],"source_refs":[{"locator":"§§2–3.6","source_id":"src-aide2"}],"status_label":"Measured task gains","summary":"Selected evaluations measure research and engineering tasks. AIDE² reports improved research harness performance and transfer to held-out benchmarks. Task performance alone does not show autonomous control of an improvement loop.","title":"AI research capability"},{"aspect":"autonomy","brief":"People still set key goals, evaluations and resource limits.","recursive_evidence_ids":["rec-dgm","rec-aide2"],"source_refs":[{"locator":"Supplied v1, §§3.1–3.7; Figure 9 p.36; §§4.1–4.4","source_id":"src-duan-v1"},{"locator":"v1 §§3–4","source_id":"src-dgm"},{"locator":"§§2–3.6","source_id":"src-aide2"}],"status_label":"Partial control","summary":"Control depends on the system and domain. The Duan survey distinguishes intervention choice, learning experience, deployment adaptation and future-improver revision. Human-set objectives, evaluation, resources and release decisions remain external controls in the selected experiments.","title":"Control of the improvement loop"},{"aspect":"structural_inheritance","brief":"Modified mechanisms are retained and reused in DGM and AIDE².","recursive_evidence_ids":["rec-dgm","rec-aide2"],"source_refs":[{"locator":"v1 §§3–4","source_id":"src-dgm"},{"locator":"§§2–3.6","source_id":"src-aide2"}],"status_label":"Bounded demonstrations","summary":"DGM and AIDE² provide bounded examples of modified improvement mechanisms being retained and used again. DGM changes agent scaffolds over fixed foundation models. AIDE² separates inner harness search from outer-improver revision. These are distinct experimental systems.","title":"Structural inheritance"},{"aspect":"effective_meta_improvement","brief":"Better task performance has not settled whether the improver improves.","recursive_evidence_ids":["rec-dgm","rec-aide2"],"source_refs":[{"locator":"v1 §§3–4","source_id":"src-dgm"},{"locator":"§§2–3.6","source_id":"src-aide2"}],"status_label":"Inconclusive","summary":"An improved task solver does not automatically make a better improver. AIDE² reports harness gains, but its evolved outer-improver comparison is inconclusive. The reviewed DGM evidence does not settle robust meta-improvement under a defensible matched comparison.","title":"Better future improvement"},{"aspect":"sustained_improvement","brief":"Reliable multi-generation gains and acceleration remain unproven here.","recursive_evidence_ids":["rec-dgm","rec-aide2"],"source_refs":[{"locator":"Supplied v1, §§3.1–3.7; Figure 9 p.36; §§4.1–4.4","source_id":"src-duan-v1"},{"locator":"v1 §§3–4","source_id":"src-dgm"},{"locator":"§§2–3.6","source_id":"src-aide2"}],"status_label":"Not established","summary":"Search iterations, accepted updates and inherited generations are different quantities. These selected studies do not establish reliable, sustained improvement of the improver or acceleration under comparable resources. This is a limit of the reviewed evidence, not an exhaustive absence claim.","title":"Sustained gains and acceleration"}],"homepage":{"context":"People still set the goals, tests and resource limits in these experiments.","questions":[{"answer":"Yes, in limited experiments.","explanation":"Examples improve an AI agent’s code or research tools.","question":"Can AI make an AI system better?"},{"answer":"There are limited examples.","explanation":"Some modified agents and tools are reused to produce further changes.","question":"Can the improved system help make the next version?"},{"answer":"Not established in the reviewed studies.","explanation":"Better performance in one experiment is not proof that a system can keep finding better ways to improve itself.","question":"Does it reliably get better at improving itself?"}],"takeaway":"AI can improve parts of AI systems. Reliable, ongoing self-improvement is still unproven."},"qualifier":"Robust full-loop RSI is not established in the reviewed evidence.","scope":"Selected public evidence: Duan et al. supplied v1, DGM v1 and AIDE² v1. Each study retains its own system boundary and evaluation conditions.","status":"Bounded recursive demonstrations"},"extraction_review":{"notes":"Synthesis checked against existing source-linked mechanism records and the preserved Duan v1 snapshot; no new source retrieval or replication.","reviewed_at":"2026-09-26T15:41:20Z","reviewer":"Codex","status":"agent_checked"},"framework_id":"duan-2026","id":"tracker-overview-2026-09-26","label":"RSI Tracker: selected evidence overview","limitations":["Selected public evidence, not an exhaustive or continuously monitored literature review.","Different systems and benchmarks cannot be assembled into evidence of a single integrated RSI loop.","Source extraction has been agent-checked; no independent experimental replication or human review is claimed.","This qualitative summary is not a percentage of completion, probability of RSI or time-to-RSI forecast."],"publication_status":"published","snapshot_type":"tracker_evidence_synthesis","source_refs":[{"locator":"Supplied v1, §§3.1–3.7; Figure 9 p.36; §§4.1–4.4","source_id":"src-duan-v1"},{"locator":"v1 §§3–4","source_id":"src-dgm"},{"locator":"§§2–3.6","source_id":"src-aide2"}],"structural_l5_summary":"Bounded recursive demonstrations in distinct experimental systems.","supersedes_id":null}]
