{
  "$schema": "https://vydex.pages.dev/schemas/vydex-dataset/2.0.0.json",
  "dataset_name": "VyDex",
  "dataset_schema_version": "2.0.0",
  "release_id": "01a03558-8c51-75ee-8ea1-aab808a32d30",
  "generated_at": "2026-08-24T19:56:30.675Z",
  "scope": "latest_entry_versions",
  "entry_count": 9,
  "methodology_versions": [
    "2.0.0"
  ],
  "entries": [
    {
      "id": "019fb336-18b1-7652-9af7-fdbe971db4f0",
      "entry_schema_version": "2.0.0",
      "slug": "artificial-neuron-biological-voltage-energy",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/artificial-neuron-biological-voltage-energy/",
      "revision_id": "01a0348c-fc6f-7026-904f-c7e3e9e1c162",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:10.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and clarified the biological-range voltage and energy result.",
      "title": "Artificial neuron repeatedly fires at biological-range voltage and energy",
      "summary": "Researchers built a memristor–RC artificial neuron that repeatedly produced voltage spikes up to about 120 mV. In its 4.7 nF configuration, the study calculated about 37 pJ per spike, while a conditioned cardiomyocyte signal path showed that the circuit could distinguish two cellular firing-rate states in real time.",
      "reality_check": "The voltage and energy comparison is to biological ranges cited by the study, not proof that the device is a living neuron. The 37 pJ result applies to one capacitor configuration, and the cardiomyocyte demonstration used filtering, amplification, pulse conversion, and attenuation rather than direct, amplification-free neural communication.",
      "context": "The Nature Communications study combined a volatile silver memristor containing *Geobacter sulfurreducens* protein nanowires with a resistor–capacitor circuit. Across 1,000 voltage sweeps, the memristor switched at approximately 60 ± 3 mV and 1.76 ± 0.06 nA. With 120 mV input pulses, the circuit produced repeated output spikes reaching approximately 120 mV, returned close to zero between spikes, and drove a downstream artificial neuron in a cascade test.\n\nWith a 4.7 nF capacitor, the authors calculated approximately 37 pJ per spike and compared it with a cited biological-neuron range of approximately 0.3–100 pJ. They estimated approximately 3.5 pJ at 0.5 nF and 0.2 pJ at 50 pF, so the complete set of reported configurations was not wholly inside that cited range. The paper’s energy calculation included resistor and capacitor currents and was checked against direct current measurement in a separate 33 nF configuration.\n\nFor the living-cell demonstration, signals from cardiomyocyte tissue were filtered, amplified, converted into 5 V pulses, and attenuated to standardized 120 mV, 70 ms inputs. Untreated tissue produced approximately 0.4 Hz signals and left the artificial neuron silent; norepinephrine-treated tissue produced approximately 0.6 Hz signals and triggered firing. This was a one-way, bench-top proof of concept, not a living-neuron replacement, implant, scalable network, or independent replication.",
      "takeaway": "This work supports a meaningful component-level result: one artificial-neuron circuit repeatedly fired at a biological-range voltage while its 4.7 nF configuration used a calculated biological-range spiking energy. It narrows an electrical mismatch that has complicated cascading and bioelectronic integration.\n\nIt does not demonstrate a deployable artificial neuron or direct neural interface. The cell experiment used cardiomyocytes and conventional analog conditioning, and no independent group’s reproduction of the complete voltage-and-energy result was found in this review.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because a peer-reviewed primary paper, official source data, and transparent peer review directly support the measured circuit behavior and reported energy calculation. The conclusion remains bounded by author-controlled firstness and the absence of independent reproduction.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong because the result has peer-reviewed primary evidence, released measurement data, and documented reviewer scrutiny. It is not Very Strong because the complete result has not been independently reproduced.",
      "evidence_strength_score": 3,
      "domains": [
        "biology",
        "physical_sciences",
        "hardware"
      ],
      "primary_topic_trail": {
        "id": "019fb336-18b5-76e4-90c9-f8aa8dbaae4f",
        "name": "Brain-Inspired Hardware Approaching Biological Function",
        "slug": "brain-inspired-hardware-biological-function",
        "description": "Tracks progress in hardware that reproduces or interoperates with biological neural signaling and information processing.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/brain-inspired-hardware-biological-function/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-07-30",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "peer_reviewed_paper",
        "developer_vendor_claim",
        "technical_artifact",
        "audit"
      ],
      "sources": [
        {
          "citation_id": "fu-2025-nature-communications",
          "title": "Constructing artificial neurons with functional parameters comprehensively matching biological values",
          "publisher_or_domain": "Nature Communications",
          "url": "https://www.nature.com/articles/s41467-025-63640-7",
          "evidence_types": [
            "peer_reviewed_paper",
            "developer_vendor_claim"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Publication metadata; device construction; voltage, current, and repeated-spiking measurements; energy calculation; cascade test; chemical experiments; cardiomyocyte setup and results; methods; and disclosed limitations. Because the authors report a result about the device they developed, the paper is also a Developer / Vendor Claim."
        },
        {
          "citation_id": "fu-2025-source-data",
          "title": "Source data for Constructing artificial neurons with functional parameters comprehensively matching biological values",
          "publisher_or_domain": "Nature Communications",
          "url": "https://static-content.springer.com/esm/art%3A10.1038%2Fs41467-025-63640-7/MediaObjects/41467_2025_63640_MOESM3_ESM.xlsx",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Official plotted measurement series for memristor switching, repeated output spikes, capacitor-dependent firing, chemical sensing, energy-validation traces, and the cardiomyocyte signal path."
        },
        {
          "citation_id": "zhao-2025-diffusive-memristor-neuron",
          "title": "A spiking artificial neuron based on one diffusive memristor, one transistor and one resistor",
          "publisher_or_domain": "Nature Electronics",
          "url": "https://www.nature.com/articles/s41928-025-01488-x",
          "evidence_types": [
            "peer_reviewed_paper",
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Post-publication context showing a different compact artificial-neuron design with picojoule-per-spike energy and cascade behavior. It is not an independent reproduction of the UMass biological-amplitude device."
        },
        {
          "citation_id": "sarkar-2022-organic-artificial-neuron",
          "title": "An organic artificial spiking neuron for in situ neuromorphic sensing and biointerfacing",
          "publisher_or_domain": "Nature Electronics",
          "url": "https://www.nature.com/articles/s41928-022-00859-y",
          "evidence_types": [
            "peer_reviewed_paper",
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Prior operation in liquid, chemical responsiveness, real-time biological interfacing, and the earlier artificial neuron's higher bias and spiking-energy requirements. Its publisher correction did not affect these cited results."
        },
        {
          "citation_id": "fu-2020-bio-voltage-memristors",
          "title": "Bioinspired bio-voltage memristors",
          "publisher_or_domain": "Nature Communications",
          "url": "https://www.nature.com/articles/s41467-020-15759-y",
          "evidence_types": [
            "peer_reviewed_paper",
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous frontier established by protein-nanowire memristors operating at approximately 40–100 mV and artificial-neuron functions driven by biological-amplitude inputs."
        },
        {
          "citation_id": "fu-2025-transparent-peer-review",
          "title": "Transparent peer review file for Constructing artificial neurons with functional parameters comprehensively matching biological values",
          "publisher_or_domain": "Nature Communications",
          "url": "https://static-content.springer.com/esm/art%3A10.1038%2Fs41467-025-63640-7/MediaObjects/41467_2025_63640_MOESM2_ESM.pdf",
          "evidence_types": [
            "audit"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Independent reviewer scrutiny of the energy accounting, comparison against lower-energy CMOS designs, chemical-sensing power, selectivity, and the complexity of the cardiomyocyte interface; author revisions and final validation of the energy calculation."
        }
      ]
    },
    {
      "id": "019f95f1-29e5-7ea2-a96e-03b7e9d296cb",
      "entry_schema_version": "2.0.0",
      "slug": "dreamer-4-offline-minecraft-diamonds",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/dreamer-4-offline-minecraft-diamonds/",
      "revision_id": "01a0348c-f887-73b7-9172-f704da383072",
      "revision_number": 3,
      "revision_published_at": "2026-08-24T16:14:09.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and clarified the offline training scope and authored task hierarchy.",
      "title": "Dreamer 4 becomes first reported agent to obtain Minecraft diamonds using only offline training data",
      "summary": "Google DeepMind researchers report that Dreamer 4 obtained diamonds in 0.7% of 1,000 60-minute Minecraft evaluations after learning from a fixed 2,541-hour contractor gameplay dataset and improving its policy through reinforcement learning inside a learned world model, without collecting new Minecraft experience during training.",
      "reality_check": "Offline describes the learning process, not the final evaluation. Dreamer 4’s world model, initial policy, and reward model first learned from recorded video, actions, and event annotations; reinforcement learning inside the world model was the final policy-improvement stage. The evaluation also followed a hand-specified sequence of 20 subtasks, so the result does not establish unrestricted task decomposition or reliable Minecraft mastery.",
      "context": "Dreamer 4 was trained in three broad stages: world-model pretraining on recorded video and actions, task-conditioned policy and reward-model training from the dataset, and reinforcement learning on imagined trajectories generated by the learned world model. The final policy was then evaluated in the actual Minecraft engine without using that evaluation experience for further training.\n\nFor the Offline Diamond Challenge, the researchers used approximately 2,541 hours of contractor gameplay containing 360p video, mouse and keyboard actions, and event annotations. Episodes started in randomly generated worlds with empty inventories, lasted up to 60 minutes, and used raw pixels with Minecraft’s native low-level control interface. Dreamer 4 obtained diamonds in 0.7% of 1,000 episodes, equivalent to seven successful episodes.\n\nThe agent was guided through a linear sequence of annotated subtasks covering resource gathering, tool crafting, mining iron, furnace use, and diamond mining. This still required long-horizon control under procedural variation, but it limits claims about open-ended planning. A 0.7% success rate establishes a non-zero reported capability, not dependable task mastery.",
      "takeaway": "Dreamer 4 provides strong initial evidence that a learned world model can become useful enough for imagination-based reinforcement learning to improve a policy on a difficult, long-horizon Minecraft task and transfer that improvement to the real evaluation environment. The frontier change is offline policy training for the complete diamond progression, not Minecraft play itself.\n\nThe result remains narrow: it depends on a large action-labelled dataset, substantial training, an authored task hierarchy, and a low final success rate. It does not establish a reliable general-purpose agent or a physical robot trained entirely through imagination.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because the primary preprint directly reports the evaluation design, 0.7% diamond result, offline training procedure, and comparison baselines, while public project materials expose uncut evaluation artifacts. The claim is deliberately limited to the reported Minecraft experiment.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong because the preprint provides detailed benchmark methodology and the authors released public evaluation artifacts. It is not Very Strong because Dreamer 4 remains a preprint and no independent reproduction of the Minecraft diamond result was located.",
      "evidence_strength_score": 3,
      "domains": [
        "ai_capabilities",
        "ai_evaluation"
      ],
      "primary_topic_trail": {
        "id": "019f95f1-29e6-73e2-8d15-188f7e0593bf",
        "name": "World Models for Agent Training",
        "slug": "world-models-for-agent-training",
        "description": "Tracks claims about using learned world models to train or improve agents through simulated or imagined experience.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/world-models-for-agent-training/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-07-24",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "peer_reviewed_paper",
        "developer_vendor_claim",
        "benchmark_result",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "dreamer-4-paper",
          "title": "Training Agents Inside of Scalable World Models",
          "publisher_or_domain": "arXiv / Google DeepMind researchers",
          "url": "https://arxiv.org/abs/2509.24527",
          "evidence_types": [
            "preprint",
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Main Dreamer 4 claim, training procedure, dataset and evaluation setup, milestone success rates, behavioral-cloning ablations, world-model comparison, task prompt sequence, compute details, and disclosed limitations. Because the authors are reporting a result about their own system, this is a Developer / Vendor Claim rather than an Official Claim."
        },
        {
          "citation_id": "dreamer-4-project-page",
          "title": "Dreamer 4 project page and uncut evaluations",
          "publisher_or_domain": "Dreamer 4 research team",
          "url": "https://danijar.com/project/dreamer4/",
          "evidence_types": [
            "developer_vendor_claim",
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Public project description, uncut successful evaluation videos, decoded imagination-training sequences, interactive world-model comparisons, and robotics-world-model demonstrations."
        },
        {
          "citation_id": "dreamer-3-nature-paper",
          "title": "Mastering diverse control tasks through world models",
          "publisher_or_domain": "Nature",
          "url": "https://www.nature.com/articles/s41586-025-08744-2",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous frontier established by Dreamer 3, including online reinforcement learning from sparse rewards, Minecraft input and action accommodations, and diamond acquisition after environment interaction."
        },
        {
          "citation_id": "unofficial-dreamer-4-pytorch",
          "title": "Unofficial Dreamer 4 PyTorch implementation",
          "publisher_or_domain": "Nicklas Hansen",
          "url": "https://github.com/nicklashansen/dreamer4",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Reproducibility context. The repository explicitly describes itself as an unofficial and incomplete implementation applied to continuous-control tasks rather than a reproduction of the original Minecraft result. It must not be labelled Independent Replication."
        },
        {
          "citation_id": "openai-vpt-paper",
          "title": "Video PreTraining: Learning to Act by Watching Unlabeled Online Videos",
          "publisher_or_domain": "NeurIPS 2022",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/9c7008aff45b5d8f0973b23e1a22ada0-Abstract-Conference.html",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Earlier native-interface Minecraft frontier, the estimated 24,000-action task horizon, large-scale video pretraining, and diamond-tool performance after online reinforcement-learning fine-tuning."
        }
      ]
    },
    {
      "id": "019fc73f-49c8-736d-9522-4826f88a1134",
      "entry_schema_version": "2.0.0",
      "slug": "epoch-frontier-ai-benchmark-progress-acceleration-2024",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/epoch-frontier-ai-benchmark-progress-acceleration-2024/",
      "revision_id": "01a0348d-0057-79f9-ae94-074d40f6725c",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:11.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and clarified the fitted ECI acceleration and uncertainty.",
      "title": "Epoch estimates frontier AI benchmark progress nearly doubled in pace around April 2024",
      "summary": "Epoch AI’s December 2025 retrospective estimated that the frontier of its composite Epoch Capabilities Index improved 1.85 times faster after a best-fit breakpoint on 8 April 2024: 8.337 ECI points per year before the breakpoint and 15.459 afterward.",
      "reality_check": "These are fitted trends on an arbitrary latent scale, not a direct measurement of a universal AI growth rate. The breakpoint is inferred rather than observed, with a 90% bootstrap interval from 17 March 2023 to 19 August 2024. Epoch’s later study found acceleration in some metrics but not WeirdML V2 and did not identify one precise universal growth law.",
      "context": "The Epoch Capabilities Index combines results from many benchmarks using a logistic latent-variable model that estimates model capability, benchmark difficulty, and benchmark discrimination. Epoch states that the resulting scale is arbitrary and currently anchored to Claude 3.5 Sonnet at 130 and GPT-5 at 150.\n\nFor the December 2025 retrospective, Epoch refit the index using 149 models released from December 2021 through December 2025, selected 17 models that established new running maxima, and tested 5,000 candidate breakpoint dates. The minimum-residual continuous two-segment fit placed the breakpoint on 8 April 2024, with slopes of 8.337 and 15.459 ECI points per year. The ratio was 1.85×. Across 2,000 resampled frontier datasets, the two-segment model beat a single-line model on AIC in 90% of runs and on BIC in approximately 80%. The 90% intervals were 1.3–3.1× for the ratio and 17 March 2023–19 August 2024 for the breakpoint.\n\nEpoch’s April 2026 follow-up tested four metrics, eight trend families, and four data-preparation modes. ECI, log METR time horizon, and the mathematics index showed acceleration relative to a global linear trend, while WeirdML V2 did not. Reasoning and non-reasoning models were best represented by separate trends in the positive metrics, but several superlinear forms also fit well.\n\nThe result is therefore sensitive to benchmark composition, model inclusion, elicitation and scoring choices, refitting, and the selected trend family. No independent reproduction of Epoch’s exact public ECI series, April 2024 breakpoint, and 1.85× slope ratio was located.",
      "takeaway": "The evidence supports a faster frontier trend in selected, benchmark-heavy capabilities after approximately 2024 than one constant linear rate across the full 2021–2025 period. It does not show that every AI capability accelerated by 85%, that progress changed suddenly on precisely 8 April 2024, or that reasoning models or reinforcement learning caused the change.\n\nThe strongest supported conclusion is narrower: several readily verifiable mathematics, programming, and related capability measures show evidence of accelerated measured progress, while other measures do not yet show the same pattern.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because Epoch’s primary analysis reports the fitted slopes and uncertainty procedure, its public implementation exposes the ECI method, and the later multi-metric study provides compatible but not identical evidence. The claim remains limited to measured frontier trends and does not assert universal acceleration or causation.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong because the result has detailed benchmark methodology, public fitting code, released data paths, and a later multi-metric follow-up. It is not Very Strong because both major analyses come from Epoch and no independent exact reproduction of the historical breakpoint result was located.",
      "evidence_strength_score": 3,
      "domains": [
        "ai_capabilities",
        "ai_evaluation"
      ],
      "primary_topic_trail": {
        "id": "019fc73f-49c1-739c-ad74-1cce0679efe7",
        "name": "Frontier AI Capability Progress Over Time",
        "slug": "frontier-ai-capability-progress-over-time",
        "description": "Tracks evidence about how quickly and consistently frontier AI capabilities are changing over time, including acceleration, slowdowns, plateaus, and breaks in long-run trends.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/frontier-ai-capability-progress-over-time/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-08-03",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "peer_reviewed_paper",
        "developer_vendor_claim",
        "benchmark_result",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "epoch-eci-acceleration-analysis",
          "title": "AI capabilities progress has sped up",
          "publisher_or_domain": "Epoch AI",
          "url": "https://epoch.ai/data-insights/ai-capabilities-progress-has-sped-up",
          "evidence_types": [
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Disclosure date, retrospective dataset scope, frontier-point selection, breakpoint-search procedure, corrected 8 April 2024 estimate, pre- and post-breakpoint slopes, 1.85× ratio, model comparisons, resampling results, bootstrap intervals, assumptions, limitations, and erratum."
        },
        {
          "citation_id": "epoch-capabilities-acceleration-follow-up",
          "title": "Have AI Capabilities Accelerated?",
          "publisher_or_domain": "Epoch AI",
          "url": "https://epoch.ai/publications/have-ai-capabilities-accelerated",
          "evidence_types": [
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Later testing across ECI, METR time horizon, a combined mathematics index, and WeirdML V2; comparison of eight trend families and four dataset-preparation modes; reasoning-versus-non-reasoning results; domain-generalization limits; and the non-accelerating WeirdML result."
        },
        {
          "citation_id": "epoch-eci-public-code",
          "title": "ECI: Epoch Capabilities Index",
          "publisher_or_domain": "Epoch AI / GitHub",
          "url": "https://github.com/epoch-research/eci-public",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Public implementation of the ECI model, fitting scripts, benchmark-data loading, capability, difficulty and discrimination parameters, bootstrap outputs, confidence intervals, and scale anchoring."
        },
        {
          "citation_id": "rosetta-stone-ai-benchmarks",
          "title": "A Rosetta Stone for AI Benchmarks",
          "publisher_or_domain": "arXiv / Epoch AI and collaborating researchers",
          "url": "https://arxiv.org/abs/2512.00193",
          "evidence_types": [
            "preprint",
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Technical basis for placing heterogeneous benchmarks on one latent capability scale, benchmark-stitching assumptions, simulation-based validation, applications to capability trends, and limitations of representing model capability with one number."
        },
        {
          "citation_id": "epoch-eci-methodology",
          "title": "ECI Documentation – Methodology",
          "publisher_or_domain": "Epoch AI",
          "url": "https://epoch.ai/data/eci-documentation/methodology",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Current logistic model specification, nonlinear least-squares fitting, regularization, arbitrary scale construction, anchor models, and bootstrap confidence intervals."
        },
        {
          "citation_id": "epoch-eci-benchmark-page",
          "title": "Epoch Capabilities Index",
          "publisher_or_domain": "Epoch AI",
          "url": "https://epoch.ai/benchmarks/eci",
          "evidence_types": [
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Purpose of the composite index, current benchmark breadth, interpretation of absolute scores and slopes, ongoing rescaling, domain-specific limitations, and the public implementation link."
        },
        {
          "citation_id": "metr-neurips-time-horizon-paper",
          "title": "Measuring AI Ability to Complete Long Software Tasks",
          "publisher_or_domain": "NeurIPS 2025",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/85069585133c4c168c865e65d72e9775-Abstract-Conference.html",
          "evidence_types": [
            "peer_reviewed_paper",
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Separate evidence that software-agent task horizons followed a rapid historical growth trend and may have accelerated during 2024; used as convergent context based on a different metric, not as a reproduction of ECI."
        },
        {
          "citation_id": "anthropic-mythos-preview-system-card",
          "title": "System Card: Claude Mythos Preview",
          "publisher_or_domain": "Anthropic",
          "url": "https://www-cdn.anthropic.com/08ab9158070959f88f296514c21b7facce6f52bc.pdf",
          "evidence_types": [
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "A separate organization’s internal ECI-style trend analysis, its 1.86× to 4.3× slope-ratio range across breakpoint choices, and sensitivity to benchmark selection; not used as a reproduction of Epoch’s exact result."
        }
      ]
    },
    {
      "id": "019f95f1-29e6-706f-a250-e15e16b91b72",
      "entry_schema_version": "2.0.0",
      "slug": "google-deepmind-gdmi-leading-hurricane-guidance-2025",
      "aliases": [
        "google-deepmind-gdmi-hurricane-forecasting-2025"
      ],
      "canonical_url": "https://vydex.pages.dev/entries/google-deepmind-gdmi-leading-hurricane-guidance-2025/",
      "revision_id": "01a0348c-f49f-7bce-9e67-f20b876d21a9",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:08.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and clarified the basin, lead-time, and availability limits.",
      "title": "NHC verification finds Google DeepMind’s GDMI leading individual hurricane guidance in 2025",
      "summary": "During its first season in the National Hurricane Center’s live forecasting workflow, Google DeepMind’s GDMI ensemble mean slightly outperformed the official Atlantic track forecast at 12–72 hours and beat the official forecast and all tested consensus aids in eastern North Pacific track forecasts at 48–120 hours. Its Atlantic intensity skill was comparable to the official forecast.",
      "reality_check": "The result is narrower than the headline suggests. GDMI was unavailable for the first two Atlantic storms and first five eastern North Pacific storms, and NHC remained stronger in several difficult cases, including Hurricane Melissa’s intensity forecast. The model’s ability to generate scenarios as far as 15 days does not demonstrate verified 15-day hurricane accuracy.",
      "context": "The 2025 season was the first in which NHC incorporated AI-based model output into real-time operations. Its post-season verification compared available guidance under homogeneous samples, so every model was evaluated only on cases where the required guidance was available.\n\nIn the Atlantic, GDMI had lower mean track errors than the official NHC forecast at 12–72 hours, while its intensity skill was described as comparable to the official forecast. In the eastern North Pacific, GDMI performed better than the other individual models and beat both the official forecast and every tested consensus aid from 48–120 hours.\n\nThese results varied by basin, lead time, and metric. NHC’s official forecast remained essential, and the final Melissa verification found that the human intensity forecast outperformed the models at nearly every lead time. The season therefore demonstrates operationally useful AI guidance, not replacement of forecasters or universally superior performance.",
      "takeaway": "GDMI crossed a meaningful operational threshold: an AI model became useful inside a high-stakes real-time hurricane workflow for both track and intensity, and achieved leading individual-model performance under formal NHC verification. The result remains conditional on basin, forecast range, availability, and the limited evidence of one season.",
      "claim_status": "confirmed",
      "claim_status_rationale": "Confirmed because NHC’s durable post-season government verification directly evaluates GDMI under defined homogeneous comparisons, while official operational records document its live use. The conclusion is limited to the stated basins, lead times, metrics, and 2025 sample.",
      "evidence_strength": "very_strong",
      "evidence_strength_rationale": "Very Strong because the claim is supported by a formal government verification report and contemporaneous official forecast records, with additional model documentation and technical context. The evidence does not establish that GDMI will outperform all guidance everywhere or generalize across future seasons.",
      "evidence_strength_score": 4,
      "domains": [
        "ai_capabilities",
        "ai_evaluation"
      ],
      "primary_topic_trail": {
        "id": "019f95f1-29e6-783b-9df9-0bc9b2342563",
        "name": "AI in Operational Weather Forecasting",
        "slug": "ai-in-operational-weather-forecasting",
        "description": "Tracks the use and verified performance of AI systems inside real-world weather-forecasting workflows.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-in-operational-weather-forecasting/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-07-24",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "peer_reviewed_paper",
        "official_claim",
        "developer_vendor_claim",
        "benchmark_result",
        "government_report",
        "media_report"
      ],
      "sources": [
        {
          "citation_id": "nhc-2025-hurricane-season-verification",
          "title": "Forecast Verification Report: 2025 Hurricane Season",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/verification/pdfs/Verification_2025.pdf",
          "evidence_types": [
            "benchmark_result",
            "government_report"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Final post-season verification methodology; Atlantic and eastern North Pacific model comparisons; GDMI’s track and intensity performance; homogeneous sample boundaries; model availability limitations; official 2025 season findings. The report was published on 30 March 2026."
        },
        {
          "citation_id": "nhc-2025-verification-preview",
          "title": "2025 NHC Verification Report Preview",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/pdf/NHC_Verification_Report_2025_Preview.pdf",
          "evidence_types": [
            "benchmark_result",
            "government_report"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "Confirmation that 2025 was NHC’s first season incorporating AI models in real time; preliminary rapid-intensification evaluation; Melissa forecast verification; statement that GDMI was useful but experimental systems were not always available on time."
        },
        {
          "citation_id": "melissa-landfall-update",
          "title": "Category 5 Melissa makes landfall in Jamaica",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/archive/2025/al13/al132025.update.10281700.shtml",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "Official observed landfall outcome: Category 5 intensity, estimated 185 mph sustained winds and 892-millibar central pressure."
        },
        {
          "citation_id": "melissa-discussion-13",
          "title": "Hurricane Melissa Forecast Discussion 13",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/archive/2025/al13/al132025.discus.013.shtml",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "Real-time evidence that every DeepMind ensemble member forecast Category 4 intensity or higher; documentation that NHC blended GDMI into its track and intensity forecasts while Melissa was still weak."
        },
        {
          "citation_id": "melissa-discussion-18",
          "title": "Hurricane Melissa Forecast Discussion 18",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/archive/2025/al13/al132025.discus.018.shtml",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "Real-time evidence that approximately four-fifths of the DeepMind ensemble projected Category 5 intensity and that NHC regarded GDMI as its best-performing intensity guidance to that point in the season."
        },
        {
          "citation_id": "melissa-discussion-22",
          "title": "Hurricane Melissa Forecast Discussion 22",
          "publisher_or_domain": "National Hurricane Center",
          "url": "https://www.nhc.noaa.gov/archive/2025/al13/al132025.discus.022.shtml",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "Confirmation that 48 of 50 DeepMind ensemble members projected Category 5 intensity and that GDMI remained part of NHC’s operational track blend."
        },
        {
          "citation_id": "ai-tropical-cyclone-operations-evaluation",
          "title": "An Operations-Based Evaluation of Tropical Cyclone Track and Intensity Forecasts from Artificial Intelligence Weather Prediction Models",
          "publisher_or_domain": "DeMaria et al. / AMS",
          "url": "https://journals.ametsoc.org/view/journals/aies/4/4/AIES-D-24-0085.1.xml",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous-frontier baseline showing that GraphCast, Pangu-Weather and FourCastNet produced competitive tracks but severe intensity underprediction, substantial low bias and no improvement to the intensity consensus."
        },
        {
          "citation_id": "deepmind-tropical-cyclone-prediction",
          "title": "How we’re supporting better tropical cyclone prediction with AI",
          "publisher_or_domain": "Google DeepMind",
          "url": "https://deepmind.google/blog/how-were-supporting-better-tropical-cyclone-prediction-with-ai/",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Weather Lab launch date; 50-scenario and 15-day output design; training-data description; retrospective 2023–2024 testing; commencement of live NHC evaluation. Its performance claims are treated as developer evidence, not as independent verification."
        },
        {
          "citation_id": "deepmind-hurricane-melissa",
          "title": "How WeatherNext helped NHC better predict Hurricane Melissa",
          "publisher_or_domain": "Google DeepMind",
          "url": "https://deepmind.google/blog/how-weathernext-helped-the-national-hurricane-center-better-predict-hurricane-melissas-historic-landfall-in-jamaica/",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "DeepMind’s merged-basin interpretation of the 2025 results; its account of WeatherNext’s Melissa probabilities; the developer’s description of its continuing NHC collaboration."
        },
        {
          "citation_id": "skillful-joint-probabilistic-weather-forecasting",
          "title": "Skillful joint probabilistic weather forecasting from marginals",
          "publisher_or_domain": "Alet et al. / arXiv",
          "url": "https://arxiv.org/abs/2506.10772",
          "evidence_types": [
            "preprint"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Technical description of the Forecast Generative Network underlying Weather Lab’s probabilistic forecasts. It supports architectural context but is not used to establish the season-level operational result."
        },
        {
          "citation_id": "weather-network-ai-hurricane-model",
          "title": "This AI weather model helped predict 2025’s huge hurricanes",
          "publisher_or_domain": "The Weather Network",
          "url": "https://www.theweathernetwork.com/en/news/weather/severe/this-ai-weather-model-helped-forecasters-predict-2025s-huge-hurricanes",
          "evidence_types": [
            "media_report"
          ],
          "source_role": "media_report",
          "source_role_label": "Media Report",
          "used_for": "Independent journalistic explanation of the completed NHC verification report, including the exact Atlantic and eastern North Pacific comparisons. It is contextual coverage rather than the primary basis for confirmation."
        }
      ]
    },
    {
      "id": "019fcc3d-5f44-706a-8286-8a6bdaa6a9bf",
      "entry_schema_version": "2.0.0",
      "slug": "gpt-5-erdos-literature-search-status-changes",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/gpt-5-erdos-literature-search-status-changes/",
      "revision_id": "01a0348d-043f-7615-a288-688c7f703104",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:12.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and corrected the title and interpretation to literature recovery.",
      "title": "GPT-5 literature searches prompted six Erdős database status changes after finding earlier human results",
      "summary": "Between 12 and 14 October 2025, GPT-5-assisted literature searches helped the Erdős Problems project identify earlier human results for Problems 339, 494, 621, 822, 903, and 1043, after which maintainers changed the six database statuses.",
      "reality_check": "These were literature-recovery events, not six new GPT-5 proofs. Contributors examined the cited papers and decided that existing human results matched the database problems. The public record does not expose every query, rejected candidate, or unsuccessful search, so this is not a controlled benchmark of GPT-5’s literature-search reliability.",
      "context": "The Erdős Problems project maintains more than one thousand problems spanning decades of mathematical literature, different terminology, and many subfields. A problem can remain listed as open even when a relevant human result exists elsewhere in the literature.\n\nResearchers used GPT-5 through ChatGPT, the OpenAI API, and internal tools to search many problems in parallel and filter candidate references. Human contributors then checked whether the papers actually answered the database entries.\n\nA 12 October repository commit changed Problem 339 from open to proved and Problem 1043 from falsifiable to disproved. A later commit changed Problems 494, 822, and 903 from open to solved and Problem 621 from falsifiable to solved. The project’s maintained contribution table continues to classify all six as GPT-5 literature-search recoveries.\n\nThe project warns that its collection is not a controlled benchmark and that success counts are affected by selection and reporting bias. Later GPT-5 work expanded beyond this six-entry milestone and should not be conflated with it.",
      "takeaway": "GPT-5 did not solve six previously unsolved Erdős problems from scratch. It helped researchers find earlier human results that had been missed by a maintained problem database, and those findings survived expert review strongly enough to change six public status records. The milestone is evidence for AI-assisted knowledge maintenance and retrieval, not autonomous theorem proving.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because versioned repository commits record the six status changes, the project’s contribution table attributes them to GPT-5 literature review, and the later technical account describes human verification. The evidence supports search assistance, not independent mathematical proof production.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong because the event is anchored in durable database commits and maintained attribution records, with contemporaneous expert discussion and a later technical description. It remains below Very Strong because complete search logs, rejected candidates, and controlled recall or precision measurements were not publicly released.",
      "evidence_strength_score": 3,
      "domains": [
        "ai_capabilities",
        "mathematics"
      ],
      "primary_topic_trail": {
        "id": "019fcc3d-5f46-75ff-9ae8-fd9a65062332",
        "name": "AI-Assisted Scientific Literature Discovery",
        "slug": "ai-assisted-scientific-literature-discovery",
        "description": "Tracks evidence that AI systems can find, interpret, and connect scientific literature well enough to change expert-maintained knowledge records or research decisions.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-assisted-scientific-literature-discovery/"
      },
      "secondary_topic_trails": [
        {
          "id": "019fcc3d-5f46-75ff-9ae9-02c7d98621f7",
          "name": "AI in Research Mathematics",
          "slug": "ai-in-research-mathematics",
          "description": "Tracks verified changes in how AI systems contribute to research mathematics through literature discovery, proof development, formalization, computation, and expert collaboration.",
          "canonical_url": "https://vydex.pages.dev/topic-trails/ai-in-research-mathematics/"
        }
      ],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-08-04",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "researcher_organization_claim",
        "developer_vendor_claim",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "early-science-acceleration-gpt-5",
          "title": "Early science acceleration experiments with GPT-5",
          "publisher_or_domain": "arXiv / OpenAI researchers and collaborators",
          "url": "https://arxiv.org/abs/2511.16072",
          "evidence_types": [
            "preprint",
            "developer_vendor_claim"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Search methodology; use of ChatGPT, the OpenAI API, and internal tools; human verification; the later ten-problem full-solution list; partial-progress findings; detailed examples; and reproducibility limits."
        },
        {
          "citation_id": "tao-october-ai-literature-review-thread",
          "title": "I am increasingly of the opinion that the most productive near-term adoptions of AI in mathematics…",
          "publisher_or_domain": "Terence Tao / Mathstodon",
          "url": "https://mathstodon.xyz/@tao/115385022005130505",
          "evidence_types": [
            "researcher_organization_claim"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Contemporaneous identification of the exact six-entry cluster, the 16 October disclosure date, the use of systematic AI-assisted literature review, and the requirement for contributor review before findings were added to the project."
        },
        {
          "citation_id": "erdos-ai-contributions-table",
          "title": "AI contributions to Erdős problems",
          "publisher_or_domain": "Erdős Problems project / GitHub",
          "url": "https://github.com/teorth/erdosproblems/wiki/AI-contributions-to-Erd%C5%91s-problems",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Maintained attribution and dates for GPT-5 literature-search contributions, the distinction between literature recovery and primary mathematical contributions, and project warnings about incompleteness, selection bias, and benchmark interpretation."
        },
        {
          "citation_id": "erdos-status-commit-october-14",
          "title": "Mark various problems as solved, add 1081",
          "publisher_or_domain": "Erdős Problems project / GitHub",
          "url": "https://github.com/teorth/erdosproblems/commit/2b5bce4617fa80d369561d37049ba24b07dae0c2",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Direct repository evidence that Problems 494, 621, 822, and 903 changed to solved on 14 October 2025, establishing the event date for the six-entry cluster."
        },
        {
          "citation_id": "erdos-status-commit-339-1043",
          "title": "Update 339 to proved and 1043 to disproved",
          "publisher_or_domain": "Erdős Problems project / GitHub",
          "url": "https://github.com/teorth/erdosproblems/commit/f9502bad032dce9a572f1824923e5d0336a5606b",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Direct repository evidence that Problem 339 changed from open to proved and Problem 1043 changed from falsifiable to disproved on 12 October 2025."
        }
      ]
    },
    {
      "id": "019fd2d8-c7f3-7108-9297-21223bf62764",
      "entry_schema_version": "2.0.0",
      "slug": "kosmos-ai-neuron-clearance-signal",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/kosmos-ai-neuron-clearance-signal/",
      "revision_id": "01a0348d-0827-7926-a064-35181bd6aed0",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:13.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and narrowed the conclusion to a candidate cross-dataset flippase signal.",
      "title": "Kosmos AI surfaces a candidate flippase signal linked to neuron vulnerability in Alzheimer’s disease",
      "summary": "During a human-defined investigation of ageing-related neuron vulnerability, Edison Scientific’s Kosmos surfaced an age-linked transcriptomic association: flippase-family expression fell in vulnerable entorhinal neurons while phagocytosis-related markers rose in microglia. The project reported the same directional pattern in another mouse dataset and a corresponding regional pattern in human Alzheimer’s data.",
      "reality_check": "This is a computational association and candidate mechanism, not evidence that flippase loss causes phosphatidylserine exposure or microglial engulfment. The original Discovery 7 dataset is not publicly available for unaffiliated end-to-end reproduction, and the validation was conducted by the developer-led project and collaborators. The claim that the original dataset’s researchers had not identified the pattern is reported novelty, not independently audited fact.",
      "context": "Kosmos received a human-supplied research objective and curated single-nucleus RNA-sequencing data, then ran parallel data-analysis and literature-search tasks through a structured world model for up to approximately 12 hours. In Discovery 7, it identified age-associated reductions in *Atp10a* and other P4-ATPase flippase genes in vulnerable entorhinal neurons and connected them to increased phagocytosis-related gene expression in microglia.\n\nThe project performed repeated checks in the original mouse data, a separately published mouse ageing dataset, and a human Alzheimer’s dataset. The reported direction was preserved in those checks, although effects were more modest in the additional mouse data and only four of five prompted human analyses supported the corresponding regional pattern.\n\nThe technical report, linked trajectories, and Figure 8 code provide more than a bare product announcement. However, the original Discovery 7 data remain unavailable, the validation was performed within the same developer-led project, and the proposed biological pathway was not tested experimentally. Across representative reports, project evaluators found substantially weaker support for synthesis and interpretation than for direct data-analysis statements.",
      "takeaway": "Kosmos crossed a narrower scientific-discovery threshold: it coordinated a long computational search that surfaced a biologically coherent, cross-dataset candidate association rather than merely summarizing literature. The finding is useful enough to motivate independent replication and wet-lab testing, but it does not establish a causal Alzheimer’s pathway or show that an AI scientist can replace expert design, validation, or judgment.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because the technical report, linked report artifacts, and validation analyses provide more than a vendor announcement and show the stated directional association across the original, additional mouse, and human datasets. The result remains developer-led, partly unpublished, and mechanistically unverified.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong because public report and code artifacts, repeated analyses, and cross-dataset checks support the stated association. It is not Very Strong because there is no unaffiliated end-to-end replication, no causal wet-lab validation, and no complete public record of all runs and rejected candidate findings.",
      "evidence_strength_score": 3,
      "domains": [
        "ai_capabilities",
        "biology"
      ],
      "primary_topic_trail": {
        "id": "019fd2d8-c7ef-72cf-86b1-8f8c171ff7b7",
        "name": "AI Systems in Scientific Discovery",
        "slug": "ai-systems-in-scientific-discovery",
        "description": "Tracks verified changes in how AI systems contribute to novel scientific findings through literature work, computational analysis, experiment design or execution, and expert validation.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-systems-in-scientific-discovery/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-08-05",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "peer_reviewed_paper",
        "developer_vendor_claim",
        "benchmark_result",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "kosmos-technical-report",
          "title": "Kosmos: An AI Scientist for Autonomous Discovery",
          "publisher_or_domain": "arXiv / Edison Scientific researchers and academic collaborators",
          "url": "https://arxiv.org/abs/2511.02824",
          "evidence_types": [
            "preprint",
            "developer_vendor_claim",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Kosmos architecture; run duration and rollout volume; statement-level evaluation; case-study classification; Discovery 7 analyses and validation methods; human-time estimates; data availability; contributor roles; and system limitations."
        },
        {
          "citation_id": "kosmos-figure-eight-artifacts",
          "title": "Kosmos Figure 8 aging neuron vulnerability artifacts",
          "publisher_or_domain": "Edison Scientific / GitHub",
          "url": "https://github.com/EdisonScientific/kosmos-figures/tree/main/Figure_8_aging_neuron_vulnerability",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Public code, notebooks, and plotted data supporting the technical report’s Discovery 7 figure and validation analyses."
        },
        {
          "citation_id": "kosmos-discovery-seven-report",
          "title": "Mechanism of neuron vulnerability in aging",
          "publisher_or_domain": "Edison Scientific Platform",
          "url": "https://platform.edisonscientific.com/kosmos/28c427d2-be31-48b5-b272-28d5a1e3ea5c",
          "evidence_types": [
            "developer_vendor_claim",
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Public Discovery 7 report, linked analysis trajectories, literature provenance, candidate hypotheses, and the path from initial exploration to the reported flippase–phagocytosis signature."
        },
        {
          "citation_id": "google-ai-co-scientist",
          "title": "Accelerating scientific breakthroughs with an AI co-scientist",
          "publisher_or_domain": "Google Research",
          "url": "https://research.google/blog/accelerating-scientific-breakthroughs-with-an-ai-co-scientist/",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous multi-agent hypothesis-generation frontier; laboratory testing of drug-repurposing candidates and liver-fibrosis targets; and evidence against describing Kosmos as the first AI system to generate experimentally tested scientific hypotheses."
        },
        {
          "citation_id": "jin-2025-mouse-brain-aging",
          "title": "Brain-wide cell-type-specific transcriptomic signatures of healthy ageing in mice",
          "publisher_or_domain": "Nature",
          "url": "https://www.nature.com/articles/s41586-024-08350-8",
          "evidence_types": [
            "peer_reviewed_paper"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Provenance, design, and public availability of the separately produced mouse single-cell ageing dataset used for the project’s cross-dataset validation. The paper did not report the Kosmos flippase–phagocytosis finding."
        },
        {
          "citation_id": "kosmos-announcement",
          "title": "Kosmos: An AI Scientist for Autonomous Discovery",
          "publisher_or_domain": "Edison Scientific",
          "url": "https://edisonscientific.com/news/announcing-kosmos",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Public disclosure date; developer interpretation of Discovery 7; the stated wet-lab validation plan; explanation of the six-month estimate; repeated-run practice; pricing and product context; and developer-stated caveats."
        },
        {
          "citation_id": "leng-2021-vulnerable-alzheimers-neurons",
          "title": "Molecular characterization of selectively vulnerable neurons in Alzheimer’s disease",
          "publisher_or_domain": "Nature Neuroscience",
          "url": "https://www.nature.com/articles/s41593-020-00764-7",
          "evidence_types": [
            "peer_reviewed_paper"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Provenance and design of the public human entorhinal-cortex and superior-frontal-gyrus dataset used to assess whether the mouse signature had a corresponding regional pattern in Alzheimer’s disease."
        },
        {
          "citation_id": "robin-ai-scientist-paper",
          "title": "Robin: A multi-agent system for automating scientific discovery",
          "publisher_or_domain": "arXiv / FutureHouse researchers",
          "url": "https://arxiv.org/abs/2505.13400",
          "evidence_types": [
            "preprint",
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous Edison/FutureHouse frontier; integration of literature search, hypothesis generation, experiment planning, and data analysis; the ripasudil candidate and follow-up RNA-sequencing interpretation; and evidence that validated AI-led biological discovery preceded Kosmos."
        },
        {
          "citation_id": "independent-kosmos-radiation-biology-evaluation",
          "title": "When AI Does Science: Evaluating the Autonomous AI Scientist KOSMOS in Radiation Biology",
          "publisher_or_domain": "arXiv",
          "url": "https://arxiv.org/abs/2511.13825",
          "evidence_types": [
            "preprint",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Separate-party evidence that useful Kosmos findings can coexist with uncertain or false hypotheses and that random-gene null comparisons can materially change interpretation. This study did not test Discovery 7."
        }
      ]
    },
    {
      "id": "019f95f1-29e6-7b1a-b120-8c2d9d628ed9",
      "entry_schema_version": "2.0.0",
      "slug": "metr-software-task-horizons-doubling-seven-months",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/metr-software-task-horizons-doubling-seven-months/",
      "revision_id": "01a0348c-f0b7-7740-91b7-1063d205f254",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:07.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2, broadened the title to the full task set, and revised Evidence Strength to Strong.",
      "title": "METR finds frontier AI task-completion horizons doubling about every seven months",
      "summary": "METR’s human-time-calibrated benchmark found that the frontier 50% task-completion horizon for model agents on its software and research tasks roughly doubled every seven months from 2019 to early 2025.",
      "reality_check": "This is a benchmark trend, not a measure of how long an AI can operate unattended. A 50% horizon means the evaluated agent configuration is predicted to complete tasks of that human-difficulty level half the time; it does not mean the system can reliably perform every task of that duration or replace that amount of human work.",
      "context": "METR’s original study combined 170 software and research tasks and modelled agent success against the amount of time an expert human would need to complete each task. The resulting 50% horizon provided a common human-time scale for comparing model-agent systems across generations.\n\nThe peer-reviewed NeurIPS version retained the long-run trend and reported an approximately 110-minute horizon for o3. METR’s TH1.1 update expanded the suite to 228 tasks; its stitched long-run estimate was about 196.5 days per doubling, close to the original 195.8-day estimate. Faster post-2023 and post-2024 slopes were more sensitive to task composition, model selection, and modelling assumptions.\n\n> A time horizon is a human-time difficulty measure, not the amount of wall-clock time an agent can work without supervision.\n\nThe benchmark mostly uses self-contained, automatically gradable software and research tasks. Results depend on the model, scaffold, tools, prompts, and resource limits. Human baselines are also noisy; in TH1.1, only five of the 31 tasks estimated at eight hours or more had measured human baselines. BRIDGE recovered a similar exponential trend with a different statistical model, but its calibration still partly uses METR’s human-time annotations.",
      "takeaway": "The durable result is rapid improvement on METR-like software and research tasks across multiple model generations. It does not establish that a one-hour horizon equals one hour of dependable workplace autonomy, that every one-hour professional task is within reach, or that the faster recent slope will continue indefinitely.",
      "claim_status": "confirmed",
      "claim_status_rationale": "Confirmed for the bounded claim because the peer-reviewed study, public analysis, TH1.1 update, and separate BRIDGE method all support rapid growth in the measured METR task-completion horizon. The confirmation applies to the stated benchmark and historical period, not to general labour automation.",
      "evidence_strength": "strong",
      "evidence_strength_rationale": "Strong evidence comes from peer-reviewed analysis, inspectable code and data, a benchmark expansion that preserves the long-run estimate, and a separate statistical reconstruction. It is not Very Strong because precise rates are sensitive to task composition, human-time baselines, model/scaffold choices, and modelling assumptions.",
      "evidence_strength_score": 3,
      "domains": [
        "ai_capabilities",
        "ai_evaluation"
      ],
      "primary_topic_trail": {
        "id": "019f95f1-29e6-7321-8eae-45113baba7cd",
        "name": "AI Agents in Software Engineering",
        "slug": "ai-agents-in-software-engineering",
        "description": "Tracks the capability of AI agents to complete software-engineering tasks across increasing scope, duration, and autonomy.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-agents-in-software-engineering/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-07-24",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "peer_reviewed_paper",
        "independent_replication",
        "researcher_organization_claim",
        "benchmark_result",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "metr-neurips-time-horizon-paper",
          "title": "Measuring AI Ability to Complete Long Software Tasks",
          "publisher_or_domain": "NeurIPS 2025",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/85069585133c4c168c865e65d72e9775-Abstract-Conference.html",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Peer-reviewed methodology, composition of the original task suite, model results, long-run doubling trend, robustness work, and limitations."
        },
        {
          "citation_id": "metr-time-horizon-1-1",
          "title": "Time Horizon 1.1",
          "publisher_or_domain": "METR",
          "url": "https://metr.org/blog/2026-1-29-time-horizon-1-1/",
          "evidence_types": [
            "researcher_organization_claim",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Expansion from 170 to 228 tasks, revised model estimates, long-run and recent-window doubling rates, task-composition sensitivity, and updated Claude 3.7 Sonnet estimates."
        },
        {
          "citation_id": "bridge-time-horizon-replication",
          "title": "BRIDGE: Predicting Human Task Completion Time From Model Performance",
          "publisher_or_domain": "ICML 2026 / arXiv",
          "url": "https://arxiv.org/abs/2602.07267",
          "evidence_types": [
            "peer_reviewed_paper",
            "independent_replication"
          ],
          "source_role": "independent_replication",
          "source_role_label": "Independent Replication",
          "used_for": "Separate-team reproduction of the exponential task-horizon trend using a different item-response framework and additional benchmark data; also used to assess how independent the replication is from METR’s calibration data."
        },
        {
          "citation_id": "metr-time-horizon-analysis",
          "title": "METR Time Horizon Analysis",
          "publisher_or_domain": "METR / GitHub",
          "url": "https://github.com/METR/eval-analysis-public",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Public analysis code, data pipeline, logistic fitting procedure, bootstrap analysis, and reproducibility of the reported figures and time-horizon estimates."
        },
        {
          "citation_id": "metr-time-horizon-limitations",
          "title": "Clarifying Limitations of Time Horizon",
          "publisher_or_domain": "METR",
          "url": "https://metr.org/notes/2026-01-22-time-horizon-limitations/",
          "evidence_types": [
            "researcher_organization_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Clarification that time horizon is not wall-clock autonomous runtime, domain and task-distribution limits, uncertainty in individual estimates, reliability limits, and warnings against treating the metric as a direct automation rate."
        },
        {
          "citation_id": "metr-modelling-assumptions",
          "title": "Impact of Modelling Assumptions on Time Horizon Results",
          "publisher_or_domain": "METR",
          "url": "https://metr.org/notes/2026-03-20-impact-of-modelling-assumptions-on-time-horizon-results/",
          "evidence_types": [
            "researcher_organization_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "The corrected regularization mistake, benchmark-saturation concerns, sensitivity to curve fitting and task-length noise, and the scale of uncertainty in recent model estimates."
        },
        {
          "citation_id": "metr-original-time-horizon-post",
          "title": "Measuring AI Ability to Complete Long Tasks",
          "publisher_or_domain": "METR",
          "url": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "evidence_types": [
            "researcher_organization_claim",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Original March 19, 2025 disclosure, public explanation of the 50% task-completion time horizon, original Claude 3.7 Sonnet result, and the approximately seven-month trend claim."
        }
      ]
    },
    {
      "id": "019febd4-0920-73f4-b129-8d777415ead2",
      "entry_schema_version": "2.0.0",
      "slug": "single-operator-ai-assisted-mexican-government-intrusions",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/single-operator-ai-assisted-mexican-government-intrusions/",
      "revision_id": "01a0348d-0c0f-7cd8-85cb-6cf6b910de07",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:14.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and attributed the disputed campaign scope directly to Gambit.",
      "title": "Gambit says one operator used Claude Code and GPT-4.1 in a campaign targeting nine Mexican government organizations",
      "summary": "Gambit Security says a single operator used Claude Code and GPT-4.1 throughout a late-December 2025 to mid-February 2026 campaign targeting nine Mexican government organizations. The reported AI-assisted activity is detailed, but the extent of confirmed victim impact remains disputed.",
      "reality_check": "The public record does not independently establish that all nine organizations were breached or that the reported data volumes were exfiltrated. SAT says its log review found no illegitimate access or anomalous behavior, while INE says it found no breach, unauthorized access, or corroborated exfiltration. Gambit’s initial disclosure also described a small group before its later report attributed the campaign to one operator.",
      "context": "Gambit’s later report says recovered forensic materials showed Claude Code and GPT-4.1 used as operational tools from late December 2025 through mid-February 2026. The report describes 1,088 logged prompts, 5,317 AI-executed commands across 34 sessions, more than 400 custom attack scripts, 20 tailored exploits, a 17,550-line Python tool, and 2,597 structured reports across 305 internal servers. Gambit estimates that Claude Code generated and executed about 75% of remote-command activity.\n\nThose figures come from a private security firm’s account of recovered material. The public record does not include a complete, independently audited chain of custody for the underlying servers, logs, scripts, or alleged exfiltrated data. SAT and INE publicly disputed the reported impact on their systems. The separate Anthropic case shows that AI can support substantial cyber operations, but it is not evidence that this Mexican campaign achieved the same level of access or autonomy.\n\n> The strongest public finding concerns how AI was reportedly used; the weakest concerns the total scope of confirmed victim impact.\n\nA high share of AI-generated commands is not the same as a high share of the operator’s decisions, and evidence of attempted intrusion is not automatically evidence of successful breach or exfiltration.",
      "takeaway": "Gambit’s report is meaningful evidence that a security firm identified a detailed AI-assisted intrusion workflow in recovered materials. It is not yet public proof that one person successfully breached all nine named organizations or stole the reported data volumes. The case should be treated as a disputed but important signal about the operational leverage of AI in cyberattacks.",
      "claim_status": "disputed",
      "claim_status_rationale": "Disputed because Gambit reports detailed recovered-material evidence, while SAT and INE deny relevant access or exfiltration and the public record does not independently verify the full victim scope. The initial and later Gambit accounts also differ on whether the operation involved a small group or one operator.",
      "evidence_strength": "moderate",
      "evidence_strength_rationale": "Moderate evidence reflects the technical detail and apparent forensic basis of Gambit’s report, balanced against the absence of publicly verifiable raw artifacts or chain-of-custody documentation, official denials from named organizations, and unresolved differences in the reported operator count and impact.",
      "evidence_strength_score": 2,
      "domains": [
        "ai_capabilities",
        "cybersecurity"
      ],
      "primary_topic_trail": {
        "id": "019febd4-0924-7237-bae4-5129c9bf9691",
        "name": "AI in Cyberattacks",
        "slug": "ai-in-cyberattacks",
        "description": "Tracks evidence that AI systems can help plan, carry out, or scale real-world cyberattacks across increasing technical scope, autonomy, and impact.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-in-cyberattacks/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-08-10",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "official_claim",
        "researcher_organization_claim",
        "developer_vendor_claim",
        "technical_artifact",
        "government_report",
        "media_report"
      ],
      "sources": [
        {
          "citation_id": "gambit-full-technical-report",
          "title": "A Single Operator, Two AI Platforms, Nine Government Agencies: The Full Technical Report",
          "publisher_or_domain": "Gambit Security",
          "url": "https://gambit.security/blog-posts/a-single-operator-two-ai-platforms-nine-government-agencies-the-full-technical-report",
          "evidence_types": [
            "researcher_organization_claim",
            "technical_artifact"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Campaign dates and preparation; three recovered servers; one-person assessment; nine-organization scope; Claude Code and GPT-4.1 roles; prompt, command, session, script, exploit, server, and report counts; 75% remote-command estimate; translated excerpts; published file indicators; disclosed evidence limits; and standard-control context."
        },
        {
          "citation_id": "gambit-initial-disclosure",
          "title": "Prevention has lost its edge. Resilience is the winning play.",
          "publisher_or_domain": "Gambit Security",
          "url": "https://gambit.security/blog-posts/prevention-has-lost-its-edge-resilience-is-the-winning-play",
          "evidence_types": [
            "researcher_organization_claim"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Initial 24 February 2026 public disclosure, the earlier one-month description, the original characterization of a small group rather than one operator, and notice that a later technical report would follow responsible disclosure."
        },
        {
          "citation_id": "ine-breach-denial",
          "title": "Rechaza INE vulneración a sus bases de datos",
          "publisher_or_domain": "Instituto Nacional Electoral",
          "url": "https://centralelectoral.ine.mx/2026/02/26/rechaza-ine-vulneracion-a-sus-bases-de-datos/",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "INE's denial of a security breach, unauthorized access, or exfiltration; its statement that no corroborated incident was found; and its criticism that no publicly verifiable forensic evidence had then been supplied."
        },
        {
          "citation_id": "sat-cyberattack-response",
          "title": "Tarjeta informativa 6. Sobre el supuesto ciberataque a instituciones por medio de inteligencia artificial",
          "publisher_or_domain": "Servicio de Administración Tributaria / Government of Mexico",
          "url": "https://www.gob.mx/sat/prensa/tarjeta-informativa-6-420283?idiom=es",
          "evidence_types": [
            "official_claim"
          ],
          "source_role": "official_record",
          "source_role_label": "Official Record",
          "used_for": "SAT's statement that it reviewed logs for potentially related systems and found no illegitimate access or anomalous operational behavior."
        },
        {
          "citation_id": "anthropic-ai-orchestrated-espionage",
          "title": "Disrupting the first reported AI-orchestrated cyber espionage campaign",
          "publisher_or_domain": "Anthropic",
          "url": "https://www.anthropic.com/news/disrupting-AI-espionage",
          "evidence_types": [
            "developer_vendor_claim"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous frontier: Anthropic's high-confidence Chinese state-sponsored attribution, roughly 30 attempted targets, limited successful intrusions, extensive Claude Code execution, and continuing human decision points."
        },
        {
          "citation_id": "nids-ai-cyberattack-scaling-commentary",
          "title": "NIDS Commentary No. 434: Impact and countermeasures of scaling through AI misuse in advanced cyberattack campaigns",
          "publisher_or_domain": "National Institute for Defense Studies / Japan Ministry of Defense",
          "url": "https://www.nids.mod.go.jp/publication/commentary/commentary434.html",
          "evidence_types": [
            "government_report"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Independent strategic analysis of how AI could reduce human coordination constraints in advanced cyberattack campaigns, limits on current conclusions, and contemporaneous acknowledgment that named Mexican organizations disputed the reported impact."
        },
        {
          "citation_id": "bloomberg-mexican-government-breach",
          "title": "Hacker Used Anthropic's Claude to Steal Mexican Data Trove",
          "publisher_or_domain": "Bloomberg",
          "url": "https://www.bloomberg.com/news/articles/2026-02-25/hacker-used-anthropic-s-claude-to-steal-sensitive-mexican-data",
          "evidence_types": [
            "media_report"
          ],
          "source_role": "media_report",
          "source_role_label": "Media Report",
          "used_for": "Initial reporting; the 150 GB allegation; named organizations; Spanish-language prompts and safeguard behavior; institutional responses; statements from Anthropic and OpenAI about investigations, refusals, disruption, and account bans; and statements attributed to Gambit researchers."
        }
      ]
    },
    {
      "id": "019ff105-27a1-7150-9bc7-da4df518e53e",
      "entry_schema_version": "2.0.0",
      "slug": "sparse-neuron-subsets-factual-hallucination-predictors",
      "aliases": [],
      "canonical_url": "https://vydex.pages.dev/entries/sparse-neuron-subsets-factual-hallucination-predictors/",
      "revision_id": "01a0348d-0ff7-7297-80d5-b6d13b010fcb",
      "revision_number": 2,
      "revision_published_at": "2026-08-24T16:14:15.031Z",
      "latest_update_summary": "Migrated the Entry to schema v2 and narrowed the conclusion to predictive, domain-specific signals.",
      "title": "Researchers find sparse neuron subsets can predict factual hallucination in several open models",
      "summary": "A Tsinghua research team found that very small, model-specific groups of feed-forward neurons predicted factual hallucinations across several tests in six open-weight models. Later studies suggest that the signal is domain-specific and is not, by itself, a reliable control handle.",
      "reality_check": "Prediction is not proof that these neurons cause factual hallucinations. The original interventions changed over-compliance behaviours rather than establishing a reduction in factual hallucination, and later studies found weak factual-rate effects and more distributed signals.",
      "context": "The H-Neurons study used sparse logistic probes over neuron-contribution features to distinguish faithful and hallucinated answers. Across six open-weight models and several factual-question settings, selected subsets usually outperformed matched random subsets. The reported subsets were approximately 0.001–0.035% of feed-forward neurons, depending on the model and test. The same indices retained predictive signal when applied to corresponding base models, which supports persistence of the signal across model stages but does not prove a causal origin.\n\nThe study’s activation-scaling experiments changed behaviours such as accepting false premises, following misleading context, sycophancy, and harmful compliance. They did not establish that scaling those neurons changes factual hallucination rates. An independent cross-domain study found mean AUROC of 0.783 within a trained domain but 0.563 after transfer to another domain, indicating that detectors require domain-specific calibration. A separate medical study found that hallucination signals can be distributed and redundant, with detectability not reliably translating into neuron-level control.\n\n> A detector can read a signal without controlling the behaviour that produced it.\n\nEarlier work on internal truthfulness, entity awareness, and shared uncertainty circuits is consistent with the broader idea that models encode information about factual reliability internally, but it does not establish a universal hallucination circuit.",
      "takeaway": "The evidence supports a useful, model-specific measurement result: sparse internal features can help predict factual hallucination in tested settings. It does not establish a universal circuit, show that fewer than 0.1% of neurons are solely responsible for hallucinations, or provide a reliable correction mechanism. Detection should be evaluated by domain, and factual reliability still requires external verification and task-specific testing.",
      "claim_status": "supported",
      "claim_status_rationale": "Supported because the original preprint, inspectable implementation, and later independent studies support the bounded claim that sparse neuron subsets can predict hallucination-related outcomes in several open models. The support does not extend to a universal mechanism or reliable causal control.",
      "evidence_strength": "moderate",
      "evidence_strength_rationale": "Moderate evidence reflects a detailed primary preprint and reproducible implementation, strengthened by independent follow-up work, but limited by the preprints’ review status, domain-transfer failures, distributed medical signals, and the distinction between predicting hallucination and controlling it.",
      "evidence_strength_score": 2,
      "domains": [
        "ai_capabilities",
        "ai_evaluation"
      ],
      "primary_topic_trail": {
        "id": "019ff105-27a3-707f-a3ec-6444bf066784",
        "name": "AI Model Hallucinations",
        "slug": "ai-model-hallucinations",
        "description": "Tracks evidence about why AI models produce factually incorrect or unsupported outputs and how reliably those failures can be detected, predicted, reduced, or prevented.",
        "canonical_url": "https://vydex.pages.dev/topic-trails/ai-model-hallucinations/"
      },
      "secondary_topic_trails": [],
      "methodology": {
        "id": "01a03111-1c00-7000-8691-64d211a4cf43",
        "version": "2.0.0",
        "canonical_url": "https://vydex.pages.dev/methodology/2.0.0/"
      },
      "dates": {
        "date_added": "2026-08-11",
        "date_updated": "2026-08-24",
        "date_last_checked": "2026-08-24"
      },
      "evidence_types": [
        "preprint",
        "peer_reviewed_paper",
        "independent_replication",
        "benchmark_result",
        "technical_artifact"
      ],
      "sources": [
        {
          "citation_id": "h-neurons-paper",
          "title": "H-Neurons: On the Existence, Impact, and Origin of Hallucination-Associated Neurons in LLMs",
          "publisher_or_domain": "arXiv / Tsinghua University researchers",
          "url": "https://arxiv.org/abs/2512.01797",
          "evidence_types": [
            "preprint",
            "benchmark_result"
          ],
          "source_role": "primary_evidence",
          "source_role_label": "Primary Evidence",
          "used_for": "Model set; neuron-selection method; training-data construction; selected-neuron ratios; detection results; intervention benchmarks; base-model transfer; parameter-drift analysis; and acknowledged mitigation trade-offs."
        },
        {
          "citation_id": "cross-domain-h-neuron-study",
          "title": "Do Hallucination Neurons Generalize? Evidence from Cross-Domain Transfer in LLMs",
          "publisher_or_domain": "arXiv / independent researchers",
          "url": "https://arxiv.org/abs/2604.19765",
          "evidence_types": [
            "preprint",
            "independent_replication",
            "benchmark_result"
          ],
          "source_role": "independent_replication",
          "source_role_label": "Independent Replication",
          "used_for": "Independent reconstruction of H-Neuron detection; within-domain and cross-domain AUROC; model and domain coverage; and the negative factual-hallucination activation-scaling result."
        },
        {
          "citation_id": "h-neurons-code",
          "title": "H-Neurons official implementation",
          "publisher_or_domain": "THUNLP / GitHub",
          "url": "https://github.com/thunlp/H-Neurons",
          "evidence_types": [
            "technical_artifact"
          ],
          "source_role": "strong_artifact",
          "source_role_label": "Strong Artifact",
          "used_for": "Inspectable data collection, answer-token extraction, CETT activation extraction, sparse-classifier and intervention code, example data, and reproducibility boundaries."
        },
        {
          "citation_id": "ferrando-entity-awareness",
          "title": "Do I Know This Entity? Knowledge Awareness and Hallucinations in Language Models",
          "publisher_or_domain": "ICLR 2025",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/c1c44e46358e0fb94dc94ec495a7fb1a-Abstract-Conference.html",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous sparse-autoencoder frontier for entity recognition, causal steering of refusal and hallucination, and reuse of base-model features after chat fine-tuning."
        },
        {
          "citation_id": "orgad-intrinsic-hallucination-representation",
          "title": "LLMs Know More Than They Show: On the Intrinsic Representation of LLM Hallucinations",
          "publisher_or_domain": "ICLR 2025",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/a712d461e57201efe35d429a6f1731c1-Abstract-Conference.html",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Previous hidden-state detection frontier, concentration of truthfulness information at particular tokens, and earlier evidence that detectors fail to generalize universally across datasets."
        },
        {
          "citation_id": "medical-neuron-hallucination-study",
          "title": "Readable but Not Controllable: Neuron-Level Evidence for Medical LLM Hallucination",
          "publisher_or_domain": "arXiv / independent researchers",
          "url": "https://arxiv.org/abs/2607.00158",
          "evidence_types": [
            "preprint",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Separate medical-domain evidence for strong neuron-level detectability, distributed and redundant predictive signals, and the gap between decoding and reliable intervention."
        },
        {
          "citation_id": "sciurus-shared-circuits",
          "title": "SCIURus: Shared Circuits for Interpretable Uncertainty Representations in Language Models",
          "publisher_or_domain": "NAACL 2025 / ACL Anthology",
          "url": "https://aclanthology.org/2025.naacl-long.618/",
          "evidence_types": [
            "peer_reviewed_paper",
            "benchmark_result"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Evidence that factual recall and expressed uncertainty rely on overlapping network regions across eight models and five datasets, without establishing an impossibility result for self-verification."
        },
        {
          "citation_id": "openai-hallucination-theory",
          "title": "Why Language Models Hallucinate",
          "publisher_or_domain": "arXiv / OpenAI researchers and collaborators",
          "url": "https://arxiv.org/abs/2509.04664",
          "evidence_types": [
            "preprint"
          ],
          "source_role": "context_source",
          "source_role_label": "Context Source",
          "used_for": "Learning-theoretic account of factual errors during next-token pre-training and the separate role of post-training evaluation incentives that reward guessing."
        }
      ]
    }
  ]
}
