{
  "schema": "fak-frontier-infrastructure-index/1",
  "updated_at": "2026-08-27",
  "issue": 9387,
  "root_issue": 9269,
  "latest_slice_issue": 9387,
  "status": "initial_spine_incomplete",
  "scope_statement": "Dated, source-linked evidence about frontier labs, hyperscalers, AI clouds, datacenters, accelerators, serving systems, workload distributions, economics, releases, news, startups, and rumors that can change fak architecture or benchmark assumptions.",
  "methodology": {
    "source_priority": [
      "primary official sources",
      "peer-reviewed or original research",
      "credible reporting",
      "analyst datasets",
      "rumor only with explicit provenance"
    ],
    "rule": "A plan is not delivered capacity; peak hardware is not goodput; users are not requests; synthetic distributions are not production measurements; copied rumors are not corroboration.",
    "exhaustiveness": "Exhaustive is operationalized as an explicit taxonomy, dated source ledger, and visible coverage gaps. The open web is not finite and this spine remains incomplete until each declared slice is researched and refreshed."
  },
  "evidence_classes": [
    "production_measurement",
    "production_observation",
    "benchmark_measurement",
    "synthetic_experiment",
    "official_statement",
    "vendor_claim",
    "analyst_estimate",
    "reported_estimate",
    "reported_observation",
    "inference",
    "rumor",
    "vendor_specification"
  ],
  "confidence_levels": [
    "high",
    "medium_high",
    "medium",
    "low"
  ],
  "required_entry_fields": [
    "id",
    "category",
    "entity",
    "topic",
    "published_at",
    "event_at",
    "source_title",
    "source_url",
    "source_kind",
    "evidence_class",
    "confidence",
    "claim",
    "quantified",
    "assumptions",
    "contradictions_or_limits",
    "fak_implications",
    "rumor"
  ],
  "coverage": {
    "entry_count": 272,
    "categories": [
      "accelerator_platform",
      "ai_cloud",
      "datacenter_physical",
      "frontier_lab",
      "hardware_supply",
      "hyperscaler",
      "market_signal",
      "policy_regulation",
      "serving_system",
      "standard",
      "supply_chain",
      "workload_model",
      "workload_trace"
    ],
    "entities": [
      "01.AI",
      "AEP Ohio data-center tariff",
      "AI Singapore SEA-LION",
      "AI datacenter power architecture research",
      "AI optical interconnect sector",
      "AI21 Labs",
      "AMD / Cerebras",
      "AWS",
      "AWS / Anthropic",
      "AWS / Cerebras",
      "AWS / NVIDIA",
      "AWS Bedrock",
      "Agent workflow scheduling study",
      "AgentSysBench",
      "Agentic OS research study",
      "Ai2",
      "Alibaba",
      "Alibaba Cloud / Aliyun",
      "Alibaba Cloud Tair",
      "Alibaba Qwen",
      "Alphabet",
      "Amazon",
      "Ampere Computing / SoftBank",
      "Anchor Browser",
      "Anthropic",
      "Anthropic / AWS",
      "Anthropic / Decart",
      "Anthropic / Fluidstack",
      "Anthropic / Google / Broadcom",
      "Anthropic / Google Cloud",
      "Anthropic / Microsoft / NVIDIA",
      "Anthropic / Mistral / Microsoft",
      "Anthropic / Nscale",
      "Apple",
      "Applied Digital",
      "Arista Networks",
      "Arista XPO MSA",
      "Azure OpenAI / BurstGPT",
      "Baichuan 2",
      "Baidu",
      "Broadcom Tomahawk 6",
      "Browserbase",
      "Builder.ai",
      "BurstGPT v1.1",
      "ByteDance Seed",
      "CacheRoute",
      "Chatbot Arena",
      "Chutes production trace study",
      "Cloudflare / Replicate",
      "Cohere",
      "Continuum multi-turn agent study",
      "CoreWeave",
      "CoreWeave / Nebius / Cerebras",
      "Data Center Watch",
      "DeepSeek",
      "DistServe",
      "Dominion Energy Virginia GS-5",
      "EAGLE",
      "ERCOT Batch Zero large-load process",
      "ERCOT regional transmission planning",
      "Eaton",
      "Equinix Metal",
      "European Commission / EuroHPC",
      "European Union / AI Office",
      "FERC / Susquehanna-Amazon co-location",
      "FineServe",
      "Fireworks AI",
      "GAIA",
      "GE Vernova / Crusoe",
      "GE Vernova T&D India",
      "GitHub Copilot coding agent",
      "Google",
      "Google / utility partners",
      "Google Cloud",
      "Google DeepMind",
      "Google Franklin Township data-center proposal",
      "Google internal developer tools",
      "Graphcore / SoftBank",
      "Groq",
      "HUMAIN / SDAIA",
      "HeteroScale production study",
      "Hitachi Energy",
      "Huawei Atlas 900 A3 SuperPoD",
      "Huawei Cloud Pangu",
      "Hyperbrowser",
      "IBM Cloud / Together AI",
      "IETF Internet-Draft authors",
      "IndiaAI Mission",
      "Intel manufacturing expansion",
      "International Energy Agency",
      "Johnson Controls",
      "Kakao",
      "Kernel",
      "LG AI Research",
      "LMSYS-Chat-1M",
      "Lambda",
      "Large-scale training reliability study",
      "Leyline",
      "Lightmatter",
      "LiquidStack / Trane Technologies",
      "MBZUAI / G42 Institute of Foundation Models",
      "MBZUAI / Inception / Cerebras / Petuum",
      "MLCommons / AMD",
      "MLCommons / NVIDIA",
      "MLCommons / Red Hat",
      "MagicDec",
      "Marvell / Celestial AI",
      "Medusa",
      "Meituan DORA",
      "Meituan LongCat-Flash",
      "Meituan MTGenRec",
      "Meituan MTServe",
      "Meta",
      "Meta Laidley / Entergy Louisiana",
      "Micron HBM3E",
      "Micron HBM4",
      "Microsoft",
      "Microsoft Azure / Splitwise",
      "Microsoft Research Mnemosyne",
      "Microsoft Research Phi",
      "MiniMax",
      "Mistral AI",
      "Modine / Airedale",
      "Moonshot AI",
      "Multi-tenant prefix-cache admission study",
      "NAVER Cloud",
      "NTT tsuzumi",
      "NVIDIA",
      "NVIDIA / Coherent",
      "NVIDIA / Groq",
      "NVIDIA / HBM suppliers",
      "NVIDIA / Hugging Face",
      "NVIDIA / Lumentum",
      "NVIDIA / SchedMD",
      "NVIDIA DSX",
      "NVIDIA Dynamo",
      "NVIDIA Dynamo Planner",
      "NVIDIA Nemotron",
      "NVIDIA TensorRT-LLM",
      "Nebius Group",
      "New York Executive Order 62",
      "OSWorld",
      "OctoAI / NVIDIA",
      "OpenAI",
      "OpenAI / G42 / Oracle / NVIDIA / Cisco / SoftBank",
      "OpenAI / Georgia Power",
      "OpenAI / NVIDIA",
      "OpenAI / Oracle",
      "OpenAI / Oracle / Related Digital / DTE",
      "OpenAI / Oracle / SoftBank",
      "OpenAI / Stargate",
      "OpenAssistant Conversations",
      "OpenRouter",
      "Operator-level LLM autoscaling study",
      "Oracle",
      "Orca",
      "PJM Interconnection",
      "Parrot serving study",
      "Robust KV cache management study",
      "SDAIA / IBM",
      "SGLang",
      "SK Telecom",
      "SK hynix",
      "SMetric agent scheduling study",
      "SageServe production-trace study",
      "Sakana AI",
      "Samsung Electronics",
      "Sarathi-Serve",
      "Sarvam AI",
      "Schneider Electric",
      "Schneider Electric / NVIDIA",
      "Sea AI Lab",
      "SemiAnalysis",
      "SenseTime",
      "ServeGen production study",
      "Shanghai AI Laboratory / InternLM",
      "Siemens Energy",
      "Siemens Energy / Start Campus",
      "Singapore IMDA",
      "SkyWalker cross-region study",
      "SpecInfer",
      "Speculative decoding",
      "Steel",
      "Steel BrowserBench / AnchorBrowser",
      "Steel BrowserBench / Browserbase",
      "Steel BrowserBench / Hyperbrowser",
      "Steel BrowserBench / Kernel",
      "Steel BrowserBench / Steel",
      "StepFun",
      "TIE scheduling study",
      "TSMC",
      "TSMC Arizona Fab 21 N4",
      "TSMC CoWoS-L",
      "TaiChi serving study",
      "Technology Innovation Institute",
      "Tencent",
      "TokenScale autoscaling study",
      "Tool-result caching research",
      "TraceLab coding-agent corpus",
      "Trane Technologies",
      "U.S. AI datacenter pipeline",
      "U.S. Bureau of Industry and Security",
      "U.S. Bureau of Industry and Security / UAE",
      "U.S. datacenter sector",
      "Ultra Ethernet Consortium",
      "United States data-center cancellation cohort",
      "Untether AI",
      "Untether AI / AMD",
      "VAST Data",
      "VAST Data / NVIDIA / CoreWeave",
      "Vertiv",
      "WEKA",
      "WebArena",
      "WildChat",
      "WorkArena / BrowserGym",
      "Xiaomi MiMo",
      "Z.ai / Zhipu AI",
      "iFLYTEK Spark / Open Platform",
      "llm-d / Google",
      "tau-bench",
      "vLLM",
      "vLLM / PagedAttention",
      "xAI",
      "xAI Grok 3"
    ],
    "explicit_gaps": [
      "Entity-complete frontier-lab coverage, especially China, France, Canada, the Middle East, and private labs beyond OpenAI and Anthropic.",
      "Direct production distributions for prompt/prefix popularity, tenant concentration, geography, seasonality, retries, sessions, tool calls, and cache-hit opportunity.",
      "A parameter table extracting exact fitted distributions, quantiles, coefficients of variation, time windows, and sample populations from every workload paper.",
      "Complete hyperscaler filings normalized for AI versus non-AI capex, leases, purchase obligations, depreciation, backlog, and customer concentration.",
      "Colocation/operator, utility/grid queue, cooling/water, construction, HBM, advanced packaging, optics, storage, and electrical-equipment company ledgers.",
      "Full startup census across inference engines, neoclouds, model routers, KV-cache systems, power, cooling, networking, and modular datacenters, including failures and acquisitions.",
      "China and export-control coverage, sovereign AI programs, standards, regulatory changes, and regional supply chains.",
      "A systematic rumor chronology with original source, independent corroboration graph, later confirmation/refutation, and expiry.",
      "Automated link checking, schema validation as a committed Go leaf, and scheduled refresh; the manual contradiction matrix exists but project-level reconciliation remains incomplete.",
      "Representative production retry, server/client fallback, and client-side speculative-decoding outcome distributions remain unavailable in the reviewed primary sources.",
      "Production browser and desktop-agent trajectories remain missing: action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, and retry/failure/tail distributions.",
      "Benchmark submissions still leave many runtime topology, replica-count, achieved-active-batch, numeric SLO-threshold, and exact sequence-envelope fields unstated; benchmark evidence is not production prevalence.",
      "Google AI Hypercomputer control-plane evidence still lacks a current GKE maximum slices-per-JobSet value, an exhaustive region-and-release-stage matrix, production queue-wait and admission distributions, utilization, healthy/schedulable/active scale, failure/retry rates, power, total workload cost, and installed-to-useful-goodput conversion."
    ]
  },
  "entries": [
    {
      "id": "openai-stargate-launch-2025",
      "category": "frontier_lab",
      "entity": "OpenAI / Oracle / SoftBank",
      "topic": [
        "capex",
        "capacity",
        "cluster_scale"
      ],
      "published_at": "2025-01-21",
      "event_at": "2025-01-21",
      "source_title": "Announcing The Stargate Project",
      "source_url": "https://openai.com/index/announcing-the-stargate-project/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Stargate said it intended to invest $500B over four years in U.S. AI infrastructure, beginning with $100B immediately.",
      "quantified": {
        "planned_capex_usd": 500000000000,
        "initial_capex_usd": 100000000000,
        "period_years": 4
      },
      "assumptions": [
        "Frontier model supply and product demand justify sovereign-scale, multi-year infrastructure commitments."
      ],
      "contradictions_or_limits": [
        "An announced intention is not completed spend, installed hardware, or serving capacity."
      ],
      "fak_implications": [
        "Track announced, contracted, built, accepted, and serving capacity separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-stargate-oracle-2025",
      "category": "frontier_lab",
      "entity": "OpenAI / Oracle",
      "topic": [
        "power",
        "chips",
        "capacity"
      ],
      "published_at": "2025-07-22",
      "event_at": "2025-07-22",
      "source_title": "Stargate advances with 4.5 GW partnership with Oracle",
      "source_url": "https://openai.com/index/stargate-advances-with-partnership-with-oracle/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "OpenAI said 4.5 GW of additional Oracle capacity would bring U.S. Stargate capacity under development above 5 GW and run more than two million chips.",
      "quantified": {
        "additional_power_gw": 4.5,
        "capacity_under_development_gw_gt": 5,
        "chip_count_gt": 2000000
      },
      "assumptions": [
        "The relevant operating unit is a multi-gigawatt fleet rather than one conventional cluster."
      ],
      "contradictions_or_limits": [
        "Under development is not online; chip count does not reveal accelerator mix, yield, utilization, or goodput."
      ],
      "fak_implications": [
        "Receipts should distinguish nameplate chips, healthy schedulable chips, and delivered token goodput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-stargate-sites-2025",
      "category": "frontier_lab",
      "entity": "OpenAI / Oracle / SoftBank",
      "topic": [
        "sites",
        "geography",
        "capex"
      ],
      "published_at": "2025-09-23",
      "event_at": "2025-09-23",
      "source_title": "OpenAI, Oracle, and SoftBank expand Stargate with five new AI data center sites",
      "source_url": "https://openai.com/index/five-new-stargate-sites/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Five new U.S. sites brought stated planned capacity near 7 GW and investment above $400B over three years, against a 10 GW/$500B commitment.",
      "quantified": {
        "new_sites": 5,
        "planned_power_gw_approx": 7,
        "investment_usd_gt": 400000000000
      },
      "assumptions": [
        "Frontier capacity is geographically distributed and secured before it is operational."
      ],
      "contradictions_or_limits": [
        "Planned capacity, secured capacity, and online capacity are different states."
      ],
      "fak_implications": [
        "Use lifecycle and site-level ledgers instead of one fleet total."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-stargate-uae-2025",
      "category": "frontier_lab",
      "entity": "OpenAI / G42 / Oracle / NVIDIA / Cisco / SoftBank",
      "topic": [
        "sovereign_cloud",
        "geography",
        "power"
      ],
      "published_at": "2025-05-22",
      "event_at": "2025-05-22",
      "source_title": "Introducing Stargate UAE",
      "source_url": "https://openai.com/index/introducing-stargate-uae/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Stargate UAE announced a 1 GW Abu Dhabi cluster with 200 MW expected online in 2026.",
      "quantified": {
        "cluster_power_gw": 1,
        "initial_online_mw": 200,
        "initial_online_year": 2026
      },
      "assumptions": [
        "Jurisdictional and sovereign capacity will coexist with centralized global fleets."
      ],
      "contradictions_or_limits": [
        "The cluster is prospective and lacks public utilization or workload-mix evidence."
      ],
      "fak_implications": [
        "Region, jurisdiction, residency, and availability date must be first-class routing constraints."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-nvidia-10gw-2025",
      "category": "frontier_lab",
      "entity": "OpenAI / NVIDIA",
      "topic": [
        "roadmap",
        "power",
        "co_design"
      ],
      "published_at": "2025-09-22",
      "event_at": "2025-09-22",
      "source_title": "OpenAI and NVIDIA announce strategic partnership to deploy 10 gigawatts of NVIDIA systems",
      "source_url": "https://openai.com/index/openai-nvidia-systems-partnership/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "A letter of intent targeted at least 10 GW of NVIDIA systems and millions of GPUs, with the first gigawatt on Vera Rubin in the second half of 2026.",
      "quantified": {
        "target_power_gw": 10,
        "nvidia_investment_usd_up_to": 100000000000,
        "first_phase_gw": 1
      },
      "assumptions": [
        "Lab and hardware roadmaps will be co-optimized across several platform generations."
      ],
      "contradictions_or_limits": [
        "A letter of intent is weaker than deployed capacity; the source gives no net-goodput envelope."
      ],
      "fak_implications": [
        "Separate contracted future envelopes from present capability and record silicon generation."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-chatgpt-weekly-people-2026",
      "category": "frontier_lab",
      "entity": "OpenAI",
      "topic": [
        "userbase",
        "demand"
      ],
      "published_at": "2026-08-06",
      "event_at": "2026-08-06",
      "source_title": "Improving GPT-5.6 Sol in ChatGPT—and expanding access to GPT-5.6 Luna for free users",
      "source_url": "https://openai.com/index/improving-gpt-5-6-sol-in-chatgpt/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "OpenAI reported that 1 billion people turn to ChatGPT every week.",
      "quantified": {
        "chatgpt_weekly_people": 1000000000
      },
      "assumptions": [
        "The weekly population establishes ChatGPT product reach, not provider-wide model or API traffic."
      ],
      "contradictions_or_limits": [
        "The wording identifies people using ChatGPT each week; it is not a registered-user, monthly-active-user, daily-active-user, subscriber, paid-seat, organization, developer, request, session, message, token, or concurrency count.",
        "The disclosure does not provide request rate, interarrival law, tenant concentration, geography, model routing, or per-person usage intensity."
      ],
      "fak_implications": [
        "Keep the ChatGPT weekly-people denominator separate from subscribers, business customers, API developers, Codex users, and API traffic.",
        "Treat population scale as a reach constraint only; do not derive traffic shape or capacity from it."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-google-tpu-2025",
      "category": "frontier_lab",
      "entity": "Anthropic / Google Cloud",
      "topic": [
        "tpu",
        "power",
        "capacity"
      ],
      "published_at": "2025-10-23",
      "event_at": "2025-10-23",
      "source_title": "Expanding our use of Google Cloud TPUs and Services",
      "source_url": "https://www.anthropic.com/news/expanding-our-use-of-google-cloud-tpus-and-services",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic planned access to up to one million TPUs, worth tens of billions of dollars, with well over 1 GW expected online in 2026.",
      "quantified": {
        "tpu_count_up_to": 1000000,
        "capacity_gw_gt": 1,
        "online_year": 2026
      },
      "assumptions": [
        "Demand growth warrants million-accelerator commitments and hardware diversity."
      ],
      "contradictions_or_limits": [
        "Access, installed capacity, and sustained utilization are not equivalent."
      ],
      "fak_implications": [
        "Preserve accelerator backend identity in receipts and comparisons."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-aws-5gw-2026",
      "category": "frontier_lab",
      "entity": "Anthropic / AWS",
      "topic": [
        "trainium",
        "capacity",
        "custom_silicon"
      ],
      "published_at": "2026-04-20",
      "event_at": "2026-04-20",
      "source_title": "Anthropic and Amazon expand collaboration for up to 5 GW of new capacity",
      "source_url": "https://www.anthropic.com/news/anthropic-amazon-compute",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic said it was using over one million Trainium2 chips and committed more than $100B over ten years for up to 5 GW spanning Trainium2 through Trainium4 and Graviton.",
      "quantified": {
        "trainium2_in_use_gt": 1000000,
        "commitment_usd_gt": 100000000000,
        "capacity_gw_up_to": 5,
        "period_years": 10
      },
      "assumptions": [
        "Long-term frontier procurement spans multiple silicon generations and host CPUs."
      ],
      "contradictions_or_limits": [
        "Up-to capacity and decade commitments do not equal current serving capacity."
      ],
      "fak_implications": [
        "Model platform transitions and mixed generations explicitly."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-google-broadcom-2026",
      "category": "frontier_lab",
      "entity": "Anthropic / Google / Broadcom",
      "topic": [
        "tpu",
        "roadmap",
        "capacity"
      ],
      "published_at": "2026-04-06",
      "event_at": "2026-04-06",
      "source_title": "Anthropic expands partnership with Google and Broadcom for multiple gigawatts of next-generation compute",
      "source_url": "https://www.anthropic.com/news/google-broadcom-partnership-compute",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic signed for multiple gigawatts of next-generation TPU capacity expected online starting in 2027.",
      "quantified": {
        "capacity": "multiple_gigawatts",
        "start_year": 2027
      },
      "assumptions": [
        "Capacity planning horizons exceed one year and depend on co-designed future silicon."
      ],
      "contradictions_or_limits": [
        "No precise chip count, completion state, or utilization is disclosed."
      ],
      "fak_implications": [
        "Keep 2027 capacity out of present benchmark claims."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-multicloud-2025",
      "category": "frontier_lab",
      "entity": "Anthropic",
      "topic": [
        "heterogeneity",
        "reliability",
        "geography"
      ],
      "published_at": "2025-09-17",
      "event_at": "2025-09-17",
      "source_title": "A postmortem of three recent issues",
      "source_url": "https://www.anthropic.com/engineering/a-postmortem-of-three-recent-issues",
      "source_kind": "engineering_postmortem",
      "evidence_class": "production_observation",
      "confidence": "high",
      "claim": "Anthropic described serving Claude across AWS Trainium, NVIDIA GPUs, and Google TPUs to obtain capacity and geographic distribution for millions of users and multiple cloud channels.",
      "quantified": {
        "user_scale": "millions"
      },
      "assumptions": [
        "Frontier production serving is heterogeneous across hardware, geography, and distribution channel."
      ],
      "contradictions_or_limits": [
        "Traffic split, per-platform SLO, batching, and utilization are undisclosed."
      ],
      "fak_implications": [
        "Incident and execution receipts must retain platform, region, and hardware identity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-azure-2025",
      "category": "frontier_lab",
      "entity": "Anthropic / Microsoft / NVIDIA",
      "topic": [
        "cloud_diversity",
        "capacity",
        "co_design"
      ],
      "published_at": "2025-11-18",
      "event_at": "2025-11-18",
      "source_title": "Microsoft, NVIDIA and Anthropic announce strategic partnerships",
      "source_url": "https://www.anthropic.com/news/microsoft-nvidia-anthropic-announce-strategic-partnerships",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic committed to purchase $30B of Azure capacity and contract up to 1 GW using Grace Blackwell and Vera Rubin, while co-optimizing models and hardware with NVIDIA.",
      "quantified": {
        "azure_commitment_usd": 30000000000,
        "capacity_gw_up_to": 1
      },
      "assumptions": [
        "Frontier providers will keep several clouds and architectures in parallel."
      ],
      "contradictions_or_limits": [
        "Contracted capacity is prospective and lacks measured workload allocation."
      ],
      "fak_implications": [
        "Cross-cloud routing should price migration, cache locality, and platform-specific quality/performance."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-ironwood-2025",
      "category": "hyperscaler",
      "entity": "Google Cloud",
      "topic": [
        "inference",
        "pod_scale",
        "cooling",
        "networking"
      ],
      "published_at": "2025-04-09",
      "event_at": "2025-04-09",
      "source_title": "Ironwood: The first Google TPU for the age of inference",
      "source_url": "https://blog.google/innovation-and-ai/infrastructure-and-cloud/google-cloud/ironwood-tpu-age-of-inference/",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "Ironwood was presented as an inference-focused TPU scaling to 9,216 liquid-cooled chips over nearly 10 MW.",
      "quantified": {
        "chips_per_pod": 9216,
        "pod_power_mw_approx": 10
      },
      "assumptions": [
        "Reasoning and MoE inference require pod-scale, high-bandwidth, liquid-cooled systems."
      ],
      "contradictions_or_limits": [
        "Peak specifications are not application goodput, cost per accepted token, or SLO attainment."
      ],
      "fak_implications": [
        "Measure sustained end-to-end goodput and quality rather than peak FLOPS."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-tpu8-specialization-2026",
      "category": "hyperscaler",
      "entity": "Google Cloud",
      "topic": [
        "training",
        "inference",
        "specialization"
      ],
      "published_at": "2026-04-01",
      "event_at": "2026-04-01",
      "source_title": "Our eighth generation TPUs: two chips for the agentic era",
      "source_url": "https://blog.google/innovation-and-ai/infrastructure-and-cloud/google-cloud/eighth-generation-tpu-agentic-era/",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "Google split TPU 8 into training-oriented 8t and latency-oriented 8i, arguing that small inefficiencies compound across agent interactions.",
      "quantified": {},
      "assumptions": [
        "Agentic demand creates enough divergent workload shape to justify specialized inference silicon."
      ],
      "contradictions_or_limits": [
        "Vendor claims do not disclose the production traffic share matching each chip."
      ],
      "fak_implications": [
        "Route using observed workload shape; do not optimize all requests for one fashionable profile."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-rubin-nvl72-2026",
      "category": "accelerator_platform",
      "entity": "NVIDIA",
      "topic": [
        "rack_scale",
        "moe",
        "networking",
        "ras"
      ],
      "published_at": "2026-01-05",
      "event_at": "2026-01-05",
      "source_title": "NVIDIA Kicks Off the Next Generation of AI With Rubin",
      "source_url": "https://nvidianews.nvidia.com/news/rubin-platform-ai-supercomputer",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "Rubin NVL72 makes a 72-GPU rack the integration unit, claiming 3.6 TB/s NVLink bandwidth per GPU and 260 TB/s per rack with added serviceability and resiliency.",
      "quantified": {
        "gpus_per_rack": 72,
        "nvlink_tbps_per_gpu": 3.6,
        "rack_bandwidth_tbps": 260
      },
      "assumptions": [
        "Large MoE and reasoning workloads need tightly coupled rack-scale fabrics and system-level RAS."
      ],
      "contradictions_or_limits": [
        "Peak bandwidth and vendor comparisons do not prove sustained application goodput."
      ],
      "fak_implications": [
        "Matched envelopes must record rack topology, failures, and sustained collective bandwidth."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-rubin-cpx-2025",
      "category": "accelerator_platform",
      "entity": "NVIDIA",
      "topic": [
        "long_context",
        "prefill",
        "specialization"
      ],
      "published_at": "2025-09-09",
      "event_at": "2025-09-09",
      "source_title": "NVIDIA Unveils Rubin CPX for Massive-Context Inference",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-unveils-rubin-cpx-a-new-class-of-gpu-designed-for-massive-context-inference",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "NVIDIA announced specialized massive-context inference silicon expected at the end of 2026.",
      "quantified": {
        "expected_availability": "2026-12-31"
      },
      "assumptions": [
        "Massive-context prefill will be common enough to justify specialized hardware."
      ],
      "contradictions_or_limits": [
        "Availability and prevalence are prospective; no production context-length share is provided."
      ],
      "fak_implications": [
        "Measure context-length prevalence before generalizing massive-context optimizations."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-dynamo-2025",
      "category": "serving_system",
      "entity": "NVIDIA Dynamo",
      "topic": [
        "batching",
        "disaggregation",
        "kv_cache",
        "routing",
        "autoscaling"
      ],
      "published_at": "2025-03-18",
      "event_at": "2025-03-18",
      "source_title": "Introducing NVIDIA Dynamo",
      "source_url": "https://developer.nvidia.com/blog/introducing-nvidia-dynamo-a-low-latency-distributed-inference-framework-for-scaling-reasoning-ai-models/",
      "source_kind": "official_engineering_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "Dynamo treats prefill/decode disaggregation, dynamic GPU scheduling, KV-aware routing, asynchronous transfer, and multi-tier cache offload as datacenter-scale inference primitives.",
      "quantified": {
        "throughput_gain_up_to": 30
      },
      "assumptions": [
        "Reasoning traffic creates differing prefill/decode bottlenecks and reusable KV state."
      ],
      "contradictions_or_limits": [
        "Benefits depend on topology, workload mix, cache locality, and SLO; up-to 30x is a vendor envelope."
      ],
      "fak_implications": [
        "Ablate reuse-aware routing and disaggregation net of transfer, coordination, and failure cost."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-dynamo-topology-2026",
      "category": "serving_system",
      "entity": "NVIDIA Dynamo",
      "topic": [
        "topology",
        "disaggregation",
        "latency"
      ],
      "published_at": "2026-03-16",
      "event_at": "2026-03-16",
      "source_title": "How NVIDIA Dynamo 1.0 Powers Multi-Node Inference at Scale",
      "source_url": "https://developer.nvidia.com/blog/nvidia-dynamo-1-production-ready/",
      "source_kind": "official_engineering_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "Dynamo 1.0 added topology APIs to colocate prefill and decode within NVL72 racks, confine stacks to a datacenter, and place CPU frontends nearby.",
      "quantified": {},
      "assumptions": [
        "Physical topology materially changes KV-transfer latency and service placement."
      ],
      "contradictions_or_limits": [
        "Reference placement guidance is not proof that disaggregation wins every workload."
      ],
      "fak_implications": [
        "Topology must be explicit in scheduling receipts and comparative tests."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "servegen-2026",
      "category": "workload_trace",
      "entity": "ServeGen production study",
      "topic": [
        "arrival_distribution",
        "token_distribution",
        "tenant_decomposition",
        "multimodal",
        "reasoning"
      ],
      "published_at": "2026-05-11",
      "event_at": "2026-05-11",
      "source_title": "ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production",
      "source_url": "https://arxiv.org/html/2505.09999v3",
      "source_kind": "peer_reviewed_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "ServeGen characterizes one week of Alibaba Cloud Model Studio production traffic and reports source-specific distribution fits rather than one universal workload family: one-minute interarrival-time samples are tested against Exponential, Gamma, and Weibull families; category input lengths are fitted with a Pareto/lognormal mixture and category output lengths with an Exponential distribution. It separately reconstructs aggregate five-minute request-rate series with a linear trend and 24-hour and 12-hour Fourier terms.",
      "quantified": "The trace covers 300+ APIs from 2024-11-25 through 2024-12-01. Arrival fitting uses one-minute IAT samples; Figure 2 reports KS statistics and p-values for Exponential, Gamma, and Weibull candidates and names the best family by workload tier (Gamma for M-large, Weibull for M-mid, Exponential for M-small). Figure 10 reports the input-length Pareto/lognormal-mixture fit and output-length Exponential fit qualitatively; the paper does not publish the mixture weights, Pareto/lognormal parameters, or an input/output goodness-of-fit statistic. The reconstruction model uses aggregate five-minute rate buckets with linear trend plus 24h and 12h Fourier periods.",
      "assumptions": "Treat each reported family as bounded to its stated random variable, workload tier/category, one-week production population, and paper version v3. Do not transfer the arrival-family fits to per-client traffic or the token-length fits across categories. Use the official generator only with source-derived or explicitly supplied parameters; its Weibull option is not by itself production evidence.",
      "contradictions_or_limits": "The paper explicitly rejects neither every alternative family nor a universal family: it tests only Exponential, Gamma, and Weibull for one-minute IAT samples, while its token-length fits use different families. KS p-values vary across minutes and may be insignificant; no stationarity proof, per-client fit, mixture parameter table, or token-length goodness-of-fit statistic is reported. The 24h/12h components model aggregate seasonality, not interarrival or token-length distributions.",
      "fak_implications": "Preserve the source-specific fitted families and KS evidence as separate benchmark cases, with workload tier/category and variable labels. Replay the observed week for fidelity; when generating traffic, label fitted-IAT, fitted-token-length, and trend/seasonality components separately and do not collapse them into a universal Poisson, Pareto, lognormal, or Weibull law.",
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "fineserve-2026",
      "category": "workload_trace",
      "entity": "FineServe",
      "topic": [
        "multi_model",
        "task_mix",
        "arrival_distribution",
        "token_geometry"
      ],
      "published_at": "2026-07-25",
      "event_at": "2026-07-25",
      "source_title": "FineServe: A Fine-Grained Dataset and Characterization of Global LLM Serving Workloads",
      "source_url": "https://arxiv.org/html/2607.19349v1",
      "source_kind": "preprint",
      "evidence_class": "production_measurement",
      "confidence": "medium_high",
      "claim": "FineServe analyzed a 23-day production trace of 29 open-source models and 9 task intents from a global marketplace, finding architecture/scale-dependent arrival burstiness and token geometry rather than one universal request process.",
      "quantified": {
        "trace_days": 23,
        "models": 29,
        "task_intents": 9,
        "arrival_window_seconds": 300,
        "short_horizon_signal_metrics": [
          "coefficient_of_variation",
          "mean_squared_successive_difference"
        ],
        "arrival_model_selection": "Poisson-family, negative-binomial-family, and self-exciting candidates compared per model/scale; no universal winner published",
        "dense_output_peak_tokens": 620,
        "dense_output_peak_input_tokens": 1500,
        "dense_output_stable_tokens": 165,
        "dense_output_stable_input_tokens_min": 5000,
        "moe_output_start_tokens": 100,
        "moe_output_start_input_tokens": 500,
        "moe_output_mid_tokens": 290,
        "moe_output_mid_input_tokens": 5000,
        "moe_output_max_tokens": 400,
        "moe_output_max_input_tokens_min": 8000,
        "task_output_peak_parameters": {
          "programme": [
            1100,
            950,
            520
          ],
          "science": [
            1200,
            860,
            540
          ],
          "law": [
            750,
            620,
            280
          ]
        },
        "task_flat_output_tokens": {
          "social": 140,
          "writing": 155
        }
      },
      "assumptions": [
        "Request-arrival and token-length structure varies by architecture, parameter scale, and task intent.",
        "Five-minute CV and MSSD expose different aspects of short-window burstiness; a single average request rate is insufficient.",
        "Architecture-level input lengths can be modeled with compact parametric marginals while task inputs require mixtures, point masses, or bounded plateaus.",
        "Piecewise token-geometry values are approximate visual extractions from the paper figures; the 23-day trace does not establish population confidence intervals."
      ],
      "contradictions_or_limits": [
        "The trace is one 23-day commercial-marketplace window and is not a census of all providers, users, or geographies.",
        "The paper compares several arrival-process families per workload but does not publish one universal fitted family or parameter vector for all models.",
        "Task and architecture parameters describe this observed population; they are not production SLO, cache-hit, retry, tenant-concentration, or service-health measurements.",
        "Maximum context and model scale are model attributes, not evidence of typical request lengths.",
        "Arrival-family selection varies by workload, and neither universal fitted parameters nor population confidence intervals are reported."
      ],
      "fak_implications": [
        "Benchmark arrival processes and token geometry per model/task class rather than applying one Zipf, Poisson, or lognormal default globally.",
        "Use the published piecewise output envelopes to stress dynamic memory allocation, while preserving residual uncertainty and source population.",
        "Record fit candidates, selection criterion, window, and omitted confidence intervals explicitly."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "year-in-serving-2026",
      "category": "workload_trace",
      "entity": "Chutes production trace study",
      "topic": [
        "heavy_tail",
        "model_popularity",
        "user_model_structure",
        "longitudinal"
      ],
      "published_at": "2026-07-03",
      "event_at": "2026-07-03",
      "source_title": "A Year in LLM Serving: Workload Evolution, Caching and Load-Balancing",
      "source_url": "https://arxiv.org/abs/2608.13573",
      "source_kind": "preprint",
      "evidence_class": "production_measurement",
      "confidence": "medium_high",
      "claim": "Chutes.ai reports one year of production serving telemetry spanning 2025-01-01 through 2025-12-31: 212B requests and 46T tokens across 60K models. Distribution evidence is descriptive: request traffic is highly concentrated among the leading models, hot-model rankings change materially over time, and input/output token lengths and interarrival times are shown as empirical distributions for selected models and use cases. No Zipf/Pareto/lognormal/Poisson/Hawkes/MMPP fit or goodness-of-fit test is reported.",
      "quantified": "212B requests; 46T tokens; 60K models; calendar-year 2025. Paper figures use product-wide or selected-model empirical distributions at the figure-specific granularity; model popularity is request-count rank/share, while token-length plots count tokens per request. No family parameter is reported.",
      "assumptions": "The population is requests served by Chutes.ai during calendar 2025. Model-rank and selected-model distributions use different denominators. The year contains workload growth and model turnover, so pooled distributions are not stationary estimates.",
      "contradictions_or_limits": "Heavy concentration is not a power-law or Zipf proof. Request popularity is not token, spend, prefix, or user share. Selected-model interarrival and length plots cannot be generalized to every model or client, and the year-long aggregate does not imply stationarity.",
      "fak_implications": "Replay both concentration and churn: model request-share rank, selected-model interarrival, and per-request input/output lengths should be separate axes. Do not collapse them into one Zipf exponent or one stationary arrival process.",
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "burstgpt-2026",
      "category": "workload_trace",
      "entity": "BurstGPT v1.1",
      "topic": [
        "burstiness",
        "capacity_planning",
        "token_volume"
      ],
      "published_at": "2026-08-03",
      "event_at": "2026-08-03",
      "source_title": "Token-Burst-Aware Capacity Planning for LLM Inference Services",
      "source_url": "https://jtie.stekom.ac.id/index.php/jtie/article/view/565",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "medium",
      "claim": "The study used 5.29M requests over 121 days and argues capacity planning must use coupled input/output token bursts, queueing, service mix, and reliability rather than request counts alone.",
      "quantified": {
        "raw_requests": 5288173,
        "trace_days": 121,
        "completed_requests": 5188507
      },
      "assumptions": [
        "Token volume and bursts are more predictive of compute load than raw request rate."
      ],
      "contradictions_or_limits": [
        "The paper's planning method and venue need independent replication; trace population limits apply."
      ],
      "fak_implications": [
        "Capacity ledgers should operate in token-phase work and SLOs, not requests per second alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mnemosyne-2024",
      "category": "serving_system",
      "entity": "Microsoft Research Mnemosyne",
      "topic": [
        "long_context",
        "batching",
        "parallelism",
        "slo"
      ],
      "published_at": "2024-09-25",
      "event_at": "2024-09-25",
      "source_title": "Mnemosyne: Parallelization Strategies for Multi-Million Context LLM Inference",
      "source_url": "https://arxiv.org/abs/2409.17264",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "Mnemosyne combines adaptive chunking, sequence pipeline parallelism, and KV-cache parallelism for interactive requests up to 10M-token contexts while targeting 30 ms between tokens.",
      "quantified": {
        "context_tokens_up_to": 10000000,
        "target_time_between_tokens_ms": 30
      },
      "assumptions": [
        "Extreme long context requires different prefill and decode parallelism."
      ],
      "contradictions_or_limits": [
        "Benchmark capability does not establish the prevalence of 10M-token production requests."
      ],
      "fak_implications": [
        "Gate specialized long-context paths on measured distribution and net SLO benefit."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "iea-energy-ai-2025",
      "category": "datacenter_physical",
      "entity": "International Energy Agency",
      "topic": [
        "power",
        "cooling",
        "capacity_planning"
      ],
      "published_at": "2025-04-10",
      "event_at": "2025-04-10",
      "source_title": "Energy and AI",
      "source_url": "https://www.iea.org/reports/energy-and-ai/energy-demand-from-ai",
      "source_kind": "intergovernmental_report",
      "evidence_class": "analyst_estimate",
      "confidence": "high",
      "claim": "The IEA identifies electricity as a determinant of AI scale and estimates cooling at about 7% of efficient hyperscale consumption but over 30% for less-efficient enterprise facilities.",
      "quantified": {
        "cooling_share_pct_hyperscale_approx": 7,
        "cooling_share_pct_enterprise_gt": 30
      },
      "assumptions": [
        "Facility overhead varies materially with datacenter design and climate."
      ],
      "contradictions_or_limits": [
        "Forecasts depend on uncertain model efficiency, utilization, and buildout."
      ],
      "fak_implications": [
        "Report IT and facility power separately and carry cooling/PUE uncertainty."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-ai-datacenter-2023",
      "category": "hyperscaler",
      "entity": "Meta",
      "topic": [
        "datacenter_design",
        "liquid_cooling",
        "networking"
      ],
      "published_at": "2023-05-18",
      "event_at": "2023-05-18",
      "source_title": "Reimagining Our Infrastructure for the AI Age",
      "source_url": "https://about.fb.com/news/2023/05/metas-infrastructure-for-ai/",
      "source_kind": "official_engineering_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta described a next-generation datacenter design supporting liquid-cooled AI hardware and high-performance networks connecting thousands of chips.",
      "quantified": {
        "cluster_scale": "thousands_of_chips"
      },
      "assumptions": [
        "The datacenter rather than the server is the design boundary for large training and inference systems."
      ],
      "contradictions_or_limits": [
        "No delivered utilization or failure-domain distribution is disclosed."
      ],
      "fak_implications": [
        "Topology, cooling, and failure domains belong in cluster operating envelopes."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-nuclear-2026",
      "category": "datacenter_physical",
      "entity": "Meta",
      "topic": [
        "power",
        "nuclear",
        "capacity"
      ],
      "published_at": "2026-01-09",
      "event_at": "2026-01-09",
      "source_title": "Meta Announces Nuclear Energy Projects, Unlocking Up to 6.6 GW",
      "source_url": "https://about.fb.com/news/2026/01/meta-nuclear-energy-projects-power-american-ai-leadership/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta announced agreements associated with up to 6.6 GW of nuclear energy for AI and datacenter growth.",
      "quantified": {
        "power_gw_up_to": 6.6
      },
      "assumptions": [
        "Firm long-duration power procurement is strategic infrastructure rather than a passive utility input."
      ],
      "contradictions_or_limits": [
        "Project timelines, delivered output, and attribution to sites vary."
      ],
      "fak_implications": [
        "Capacity forecasts need power-delivery dates and firmness."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-stargate-michigan-2025",
      "category": "datacenter_physical",
      "entity": "OpenAI / Oracle / Related Digital / DTE",
      "topic": [
        "water",
        "grid",
        "community",
        "power"
      ],
      "published_at": "2025-10-30",
      "event_at": "2025-10-30",
      "source_title": "Expanding Stargate to Michigan",
      "source_url": "https://openai.com/index/expanding-stargate-to-michigan/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "A greater-than-1-GW Michigan campus proposed closed-loop cooling, use of excess transmission capacity, and project-funded grid upgrades.",
      "quantified": {
        "campus_power_gw_gt": 1
      },
      "assumptions": [
        "Water, grid impact, and ratepayer allocation are now project design requirements."
      ],
      "contradictions_or_limits": [
        "Promised mitigations need operational verification after commissioning."
      ],
      "fak_implications": [
        "Record cooling design, grid-upgrade payer, and local operating constraints."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "coreweave-platform-2026",
      "category": "ai_cloud",
      "entity": "CoreWeave",
      "topic": [
        "goodput",
        "reliability",
        "utilization"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "CoreWeave AI cloud platform",
      "source_url": "https://www.coreweave.com/",
      "source_kind": "vendor_site",
      "evidence_class": "vendor_claim",
      "confidence": "medium",
      "claim": "CoreWeave markets up to 96% cluster goodput and up to 50% fewer daily interruptions from AI-native health, lifecycle, and observability systems.",
      "quantified": {
        "cluster_goodput_pct_up_to": 96,
        "interruptions_reduction_pct_up_to": 50
      },
      "assumptions": [
        "Useful output depends on operational recovery and health, not allocated GPU-hours."
      ],
      "contradictions_or_limits": [
        "The landing page does not disclose workload, baseline, customer distribution, or methodology."
      ],
      "fak_implications": [
        "Prefer interruption-adjusted goodput over nominal allocation."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "groq-funding-scale-2026",
      "category": "ai_cloud",
      "entity": "Groq",
      "topic": [
        "inference_cloud",
        "geography",
        "demand",
        "funding"
      ],
      "published_at": "2026-08-17",
      "event_at": "2026-08-17",
      "source_title": "Groq Closes $350 million Series A",
      "source_url": "https://groq.com/newsroom/groq-closes-usd350-million-series-a-building-the-world-s-leading-ai-inference-cloud",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Groq reported a $350M round after $650M in June 2026 to scale an inference cloud, with planned NVIDIA participation.",
      "quantified": {
        "funding_usd": 350000000,
        "recent_funding_total_usd": 1000000000,
        "valuation_usd": 3500000000
      },
      "assumptions": [
        "Specialized inference clouds require substantial capital and incumbent ecosystem links."
      ],
      "contradictions_or_limits": [
        "Financing does not prove workload economics; planned participation is not completed."
      ],
      "fak_implications": [
        "Track capital, deployed MW, tokens, active tenants, and concentration separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "neocloud-q2-2026",
      "category": "market_signal",
      "entity": "CoreWeave / Nebius / Cerebras",
      "topic": [
        "economics",
        "revenue",
        "losses",
        "capacity"
      ],
      "published_at": "2026-08-14",
      "event_at": "2026-08-14",
      "source_title": "Neocloud results Q2 2026: CoreWeave, Nebius, Cerebras",
      "source_url": "https://www.datacenterdynamics.com/en/news/neocloud-results-q2-2026-coreweave-nebius-cerebras/",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Q2 reporting showed accelerating neocloud revenue alongside widening losses as operators expanded capacity.",
      "quantified": {},
      "assumptions": [
        "Demand growth can coexist with poor near-term capital efficiency and financing risk."
      ],
      "contradictions_or_limits": [
        "Cross-company accounting and backlog definitions differ."
      ],
      "fak_implications": [
        "Compare net economics including financing, idle capacity, and customer concentration."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "datacenter-opposition-2026",
      "category": "datacenter_physical",
      "entity": "U.S. datacenter sector",
      "topic": [
        "permitting",
        "community",
        "water",
        "power"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Data center controversies transform the midterm elections",
      "source_url": "https://apnews.com/article/e5a350af1bc12e6a4f470d3b68be51d7",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "high",
      "claim": "AP reported intensifying bipartisan opposition and tighter state rules involving datacenter water, power, land, pollution, and incentives.",
      "quantified": {},
      "assumptions": [
        "Social license and permitting can delay capacity independently of capital and chip supply."
      ],
      "contradictions_or_limits": [
        "Outcomes vary by jurisdiction and project."
      ],
      "fak_implications": [
        "Add permitting, community, water, and local-cost-allocation states to capacity forecasts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "datacenter-delays-2026",
      "category": "market_signal",
      "entity": "U.S. AI datacenter pipeline",
      "topic": [
        "delays",
        "power",
        "electrical_supply_chain"
      ],
      "published_at": "2026-04-07",
      "event_at": "2026-04-07",
      "source_title": "US AI expansion hit by power shortages, half planned data centers delayed or canceled",
      "source_url": "https://www.thestandard.com.hk/innovation/article/328687/US-AI-expansion-hit-by-power-shortages-half-planned-data-centers-delayed-or-canceled",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_estimate",
      "confidence": "medium",
      "claim": "Reporting citing Bloomberg said 30-50% of planned 2026 U.S. AI datacenters could be delayed or canceled because of power and electrical-component shortages.",
      "quantified": {
        "delay_or_cancel_pct_low": 30,
        "delay_or_cancel_pct_high": 50
      },
      "assumptions": [
        "Announced project pipelines substantially overstate delivered near-term capacity."
      ],
      "contradictions_or_limits": [
        "The estimate is secondary reporting and depends on how projects and delay are defined."
      ],
      "fak_implications": [
        "Apply delivery probabilities and component/power gates to announced capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "datacenterwatch-2026",
      "category": "datacenter_physical",
      "entity": "Data Center Watch",
      "topic": [
        "community",
        "delays",
        "capex_at_risk"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "$64 billion of data center projects have been blocked or delayed amid local opposition",
      "source_url": "https://www.datacenterwatch.org/report",
      "source_kind": "advocacy_dataset",
      "evidence_class": "analyst_estimate",
      "confidence": "medium",
      "claim": "Data Center Watch estimates $64B of U.S. projects blocked or delayed by local opposition.",
      "quantified": {
        "projects_blocked_or_delayed_usd": 64000000000
      },
      "assumptions": [
        "Community opposition is a quantifiable pipeline risk."
      ],
      "contradictions_or_limits": [
        "Advocacy methodology and project attribution require independent checking."
      ],
      "fak_implications": [
        "Carry source bias and confidence when pricing nontechnical project risk."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-memory-price-rumor-2026",
      "category": "supply_chain",
      "entity": "NVIDIA / HBM suppliers",
      "topic": [
        "hbm",
        "pricing",
        "supply_constraint"
      ],
      "published_at": "2026-08-23",
      "event_at": "2026-08-23",
      "source_title": "Nvidia reportedly warns biggest customers of 15% price hikes on AI servers",
      "source_url": "https://www.tomshardware.com/pc-components/dram/nvidia-reportedly-warns-biggest-customers-of-15-percent-price-hikes-on-ai-servers",
      "source_kind": "credible_reporting",
      "evidence_class": "rumor",
      "confidence": "low",
      "claim": "Bloomberg-origin reporting said some NVIDIA AI-server configurations would rise by more than 15% because of memory costs. NVIDIA’s August 26 earnings call corroborated severe memory-cost pressure and planned product price increases, but not the exact 15% figure, customer notice, or configuration list.",
      "quantified": {
        "reported_price_increase_pct_gt": 15
      },
      "assumptions": [
        "HBM availability and price can become a system-level accelerator bottleneck."
      ],
      "contradictions_or_limits": [
        "NVIDIA corroborated memory pressure and planned price increases, but not the reported >15% magnitude or exact affected systems/customers.",
        "Reported server-system prices may include OEM, memory configuration, and delivery-window effects rather than one NVIDIA chip list price."
      ],
      "fak_implications": [
        "Treat accelerator price and delivery as uncertain, memory-sensitive variables; do not present this report as confirmed."
      ],
      "rumor": {
        "is_rumor": true,
        "origin": "industry reporting relayed by Tom's Hardware",
        "corroboration": "NVIDIA fiscal-Q2-2027 earnings materials and the August 26 call corroborate extreme memory-cost pressure and planned product price increases; they do not confirm the reported >15% magnitude, customer notices, named configurations, or shipment scope.",
        "status": "partially_corroborated_open",
        "last_checked_at": "2026-08-27",
        "expires_at": "2026-11-30",
        "resolution": "Direction corroborated by NVIDIA primary materials on August 26, 2026; magnitude and configuration/customer scope remain unverified."
      }
    },
    {
      "id": "optics-squeeze-2026",
      "category": "supply_chain",
      "entity": "AI optical interconnect sector",
      "topic": [
        "optics",
        "cpo",
        "supply_constraint"
      ],
      "published_at": "2026-05-26",
      "event_at": "2026-05-26",
      "source_title": "AI boom squeezes optical tech and Huawei makes a chip comeback",
      "source_url": "https://www.ft.com/content/0d9fe3e2-904f-494c-b10d-6292aebc334c",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Financial Times reporting described shortages and price pressure in optical fiber, lasers, and indium-phosphide substrates as AI networks scale toward co-packaged optics.",
      "quantified": {},
      "assumptions": [
        "Network optics and packaging can gate cluster delivery even when accelerators are available."
      ],
      "contradictions_or_limits": [
        "Company-specific shortages and timelines can change rapidly."
      ],
      "fak_implications": [
        "Capacity BOMs should include optics, switches, packaging, and serviceability, not GPUs alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-dsx-2026",
      "category": "datacenter_physical",
      "entity": "NVIDIA DSX",
      "topic": [
        "digital_twin",
        "power",
        "cooling",
        "operations"
      ],
      "published_at": "2026-03-16",
      "event_at": "2026-03-16",
      "source_title": "NVIDIA Releases Vera Rubin DSX AI Factory Reference Design",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-releases-vera-rubin-dsx-ai-factory-reference-design-and-omniverse-dsx-digital-twin-blueprint-with-broad-industry-support",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "DSX proposes simulating power topology, thermal behavior, network layout, and operational policy before physical deployment.",
      "quantified": {},
      "assumptions": [
        "Facility and workload decisions are sufficiently coupled to warrant system digital twins."
      ],
      "contradictions_or_limits": [
        "Simulation fidelity and business benefit require site-specific validation."
      ],
      "fak_implications": [
        "Preserve facility/workload coupling in scenario models and label simulated evidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "amd-cerebras-2026",
      "category": "market_signal",
      "entity": "AMD / Cerebras",
      "topic": [
        "partnership",
        "specialized_inference",
        "heterogeneous_systems"
      ],
      "published_at": "2026-07-23",
      "event_at": "2026-07-23",
      "source_title": "AMD inks deal with AI chip startup Cerebras",
      "source_url": "https://www.axios.com/2026/07/23/amd-cerebras-ai-chips",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "AMD and Cerebras announced an integration intended to split inference work across their systems, with Cerebras adding AMD Helios systems to its cloud.",
      "quantified": {},
      "assumptions": [
        "Different inference phases or workload classes may be assigned to distinct accelerator architectures."
      ],
      "contradictions_or_limits": [
        "The offering was not yet generally available and no matched production economics were reported."
      ],
      "fak_implications": [
        "Evaluate heterogeneous phase placement using end-to-end transfer and scheduling costs."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-stargate-over-10gw-2026",
      "category": "frontier_lab",
      "entity": "OpenAI / Stargate",
      "topic": [
        "capacity",
        "delivery",
        "ecosystem"
      ],
      "published_at": "2026-04-29",
      "event_at": "2026-04-29",
      "source_title": "Building the compute infrastructure for the Intelligence Age",
      "source_url": "https://openai.com/index/building-the-compute-infrastructure-for-the-intelligence-age/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "OpenAI said it had surpassed its 10 GW U.S. infrastructure goal, adding more than 3 GW in the prior 90 days, and explicitly named power, land, permitting, transmission, workforce, community support, and partner readiness as site requirements.",
      "quantified": {
        "secured_power_gw_gt": 10,
        "power_added_prior_90_days_gw_gt": 3
      },
      "assumptions": [
        "Capacity acquisition is an ecosystem coordination problem spanning technical, financial, labor, government, and community actors."
      ],
      "contradictions_or_limits": [
        "Secured or planned capacity is not necessarily online; the publication does not disclose accepted goodput."
      ],
      "fak_implications": [
        "Represent each capacity project as a gated delivery pipeline with non-compute constraints."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-georgia-3p2gw-2026",
      "category": "datacenter_physical",
      "entity": "OpenAI / Georgia Power",
      "topic": [
        "power",
        "delivery_schedule",
        "community"
      ],
      "published_at": "2026-07-22",
      "event_at": "2026-07-22",
      "source_title": "Building AI infrastructure with the Effingham County community",
      "source_url": "https://openai.com/index/building-ai-infrastructure-with-the-effingham-county-community/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "OpenAI said it was contracting for 3.2 GW in Georgia, delivered in phases from 2028 through 2032.",
      "quantified": {
        "contracted_power_gw": 3.2,
        "delivery_start_year": 2028,
        "delivery_end_year": 2032
      },
      "assumptions": [
        "Power delivery is staged over multi-year horizons rather than instantly available with a datacenter announcement."
      ],
      "contradictions_or_limits": [
        "The source describes contracted future power, not installed IT load or utilization."
      ],
      "fak_implications": [
        "Capacity forecasts need phased dates and should not count future utility delivery as present compute."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-multisite-million-tpu-2026",
      "category": "hyperscaler",
      "entity": "Google",
      "topic": [
        "distributed_training",
        "multi_site",
        "cluster_scale"
      ],
      "published_at": "2026-05-19",
      "event_at": "2026-05-19",
      "source_title": "Google I/O 2026: Sundar Pichai opening keynote",
      "source_url": "https://blog.google/innovation-and-ai/sundar-pichai-io-2026/",
      "source_kind": "official_keynote",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Google said JAX and Pathways can distribute training across multiple sites and more than one million TPUs globally, explicitly moving beyond a single massive datacenter.",
      "quantified": {
        "global_tpu_scale_gt": 1000000,
        "expected_2026_capex_usd_low": 180000000000,
        "expected_2026_capex_usd_high": 190000000000
      },
      "assumptions": [
        "Frontier training can cross site boundaries; wide-area coordination and failure modes are part of the cluster."
      ],
      "contradictions_or_limits": [
        "The statement provides no topology, synchronization efficiency, or measured training goodput."
      ],
      "fak_implications": [
        "Do not assume one low-latency fabric or one site when defining frontier training envelopes."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-agent-tokens-2026",
      "category": "workload_trace",
      "entity": "Google internal developer tools",
      "topic": [
        "agentic",
        "token_volume",
        "growth",
        "enterprise"
      ],
      "published_at": "2026-05-19",
      "event_at": "2026-05-19",
      "source_title": "Google I/O 2026: Sundar Pichai opening keynote",
      "source_url": "https://blog.google/innovation-and-ai/sundar-pichai-io-2026/",
      "source_kind": "official_keynote",
      "evidence_class": "production_observation",
      "confidence": "high",
      "claim": "Google reported internal AI developer-tool traffic growing from about 0.5 trillion tokens per day in March 2026 to more than 3 trillion per day by May, while saying top companies process about 1 trillion tokens daily.",
      "quantified": {
        "internal_tokens_per_day_march_2026": 500000000000,
        "internal_tokens_per_day_may_2026_gt": 3000000000000,
        "top_company_tokens_per_day_approx": 1000000000000
      },
      "assumptions": [
        "Agentic coding workloads can grow by multiples over weeks and can dominate enterprise token budgets."
      ],
      "contradictions_or_limits": [
        "Aggregate token volume omits active users, request count, cache reuse, modalities, and input/output split."
      ],
      "fak_implications": [
        "Capacity planning should include rapid growth regimes and token-phase decomposition, not static annual averages."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aliyun-kvcache-wild-2025",
      "category": "workload_trace",
      "entity": "Alibaba Cloud / Aliyun",
      "topic": [
        "prefix_reuse",
        "cache",
        "consumer_vs_api",
        "reuse_skew"
      ],
      "published_at": "2025-06-03",
      "event_at": "2025-06-03",
      "source_title": "KVCache Cache in the Wild: Characterizing and Optimizing KVCache Cache at a Large Cloud Provider",
      "source_url": "https://arxiv.org/html/2506.02634v3",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "A production trace from Alibaba Cloud Model Studio covers 58 LLM serving workloads across 49 days (June 7-July 25, 2024) and reports empirical KV-prefix sharing/reuse behavior. The retained distribution variables are request-level input/output lengths, shared-prefix length and ratio, prefix-tree height/width, session request count, and cacheable-prefix reuse frequency/lifetime. Results are empirical CDFs and workload summaries; the paper reports no Zipf/Pareto/lognormal fit or goodness-of-fit test.",
      "quantified": "58 workloads; 49 days from 2024-06-07 through 2024-07-25. Example reported shares include 17.7% of requests sharing no prefix and 16.2% sharing only a system prompt; more than 60% of workloads have over 20% shareable tokens and more than 1,000 cacheable KV blocks. No rank exponent or family parameter is reported.",
      "assumptions": "Workload and figure denominators vary: requests, workloads, prefixes, sessions, or cache blocks. Prefix frequency is defined by exact/structured prefix reuse in the authors' analysis, not semantic prompt similarity or per-user demand.",
      "contradictions_or_limits": "Prefix reuse frequency is not model popularity, user share, token-spend share, or a Zipf law. Empirical CDFs do not establish fitted families. The 49-day window and strong cross-workload heterogeneity do not establish stationarity or universality.",
      "fak_implications": "Replay cacheability using measured prefix/session variables and per-workload heterogeneity. Keep prefix popularity distinct from model popularity and require explicit trace-derived ranks before testing Zipf-shaped cache scenarios.",
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tracelab-coding-agents-2026",
      "category": "workload_trace",
      "entity": "TraceLab coding-agent corpus",
      "topic": [
        "agents",
        "prefix_cache",
        "tool_calls",
        "heavy_tail",
        "human_gaps"
      ],
      "published_at": "2026-06-29",
      "event_at": "2026-06-29",
      "source_title": "TraceLab: Characterizing Coding Agent Workloads for LLM Serving",
      "source_url": "https://arxiv.org/abs/2606.30560",
      "source_kind": "preprint_and_open_dataset",
      "evidence_class": "production_measurement",
      "confidence": "medium_high",
      "claim": "A corpus of roughly 4,300 coding-agent sessions, 350,000 LLM steps, and 430,000 tool calls found long autonomous loops, long contexts with short outputs, heavily tailed tool calls, and high but imperfect prefix-cache hit rates.",
      "quantified": {
        "sessions_approx": 4300,
        "llm_steps_approx": 350000,
        "tool_calls_approx": 430000,
        "developers": 43,
        "collection_months_approx": 8,
        "llm_calls_per_request_avg": 8.8,
        "tool_calls_per_request_avg": 10.8,
        "completion_minutes_avg": 4.3,
        "completion_minutes_p90_gt": 6.4
      },
      "assumptions": [
        "Coding-agent demand differs from chat: closed loops, appended context, tool latency, and human-paced idle gaps affect cache residency."
      ],
      "contradictions_or_limits": [
        "The trace comes from the authors own use rather than a provider-wide random user sample."
      ],
      "fak_implications": [
        "Use coding-agent-specific workload profiles and include tool turns, idle gaps, append length, and imperfect cache hits."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "llmd-prefix-bimodal-2026",
      "category": "serving_system",
      "entity": "llm-d / Google",
      "topic": [
        "prefix_cache",
        "routing",
        "bimodal_distribution"
      ],
      "published_at": "2026-03-13",
      "event_at": "2026-03-13",
      "source_title": "Predicted-Latency Based Scheduling for LLMs",
      "source_url": "https://llm-d.ai/blog/predicted-latency-based-scheduling-for-llms",
      "source_kind": "engineering_blog",
      "evidence_class": "production_observation",
      "confidence": "medium_high",
      "claim": "llm-d reports an internal production prefix-match distribution that is roughly bimodal: about half of request-pod pairs are above 0.80 and half below, motivating an affinity gate rather than a smooth global score.",
      "quantified": {
        "affinity_threshold": 0.8
      },
      "assumptions": [
        "Conversation state often creates near-binary cache locality at a replica: the history is present or absent."
      ],
      "contradictions_or_limits": [
        "The underlying trace, sample size, and tenant mix are not public."
      ],
      "fak_implications": [
        "Benchmark session-affinity routing against load balancing and record hit-ratio distributions, not just averages."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "output-length-heavy-tail-2026",
      "category": "workload_model",
      "entity": "TIE scheduling study",
      "topic": [
        "output_length",
        "heavy_tail",
        "continuous_batching",
        "tail_latency"
      ],
      "published_at": "2026-05-25",
      "event_at": "2026-05-25",
      "source_title": "Scheduling LLM Inference with Uncertainty-Aware Output Length Predictions",
      "source_url": "https://arxiv.org/html/2604.00499",
      "source_kind": "research_paper",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium_high",
      "claim": "Across 1,000 LMSYS prompts with 100 generations each, output lengths were strongly heavy-tailed: average skewness 3.10, mean coefficient of variation 1.09, and P99/P50 10.77; the top decile contributed 35.7% of generated length.",
      "quantified": {
        "prompts": 1000,
        "samples_per_prompt": 100,
        "average_skewness": 3.1,
        "average_coefficient_variation": 1.09,
        "cv_gt_1_share_pct": 78.6,
        "top_decile_length_share_pct": 35.7,
        "p90_p50": 4.62,
        "p99_p50": 10.77
      },
      "assumptions": [
        "Even identical prompts produce uncertain, heavy-tailed service times; point output-length prediction is structurally weak."
      ],
      "contradictions_or_limits": [
        "Generated responses over public prompts are not a direct production arrival trace; fitted tails depend on model and decoding."
      ],
      "fak_implications": [
        "Schedulers should price output-length uncertainty, continuous batching, preemption, and tail risk rather than use one expected length."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-bedrock-batch-discount-2025",
      "category": "serving_system",
      "entity": "AWS Bedrock",
      "topic": [
        "offline_batch",
        "pricing",
        "latency_tier"
      ],
      "published_at": "2025-09-18",
      "event_at": "2025-09-18",
      "source_title": "Monitor Amazon Bedrock batch inference using Amazon CloudWatch metrics",
      "source_url": "https://aws.amazon.com/blogs/machine-learning/monitor-amazon-bedrock-batch-inference-using-amazon-cloudwatch-metrics/",
      "source_kind": "official_engineering_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AWS distinguishes bulk batch inference from on-demand serving and advertises a 50% lower price for workloads that can trade interactive latency for scheduled throughput.",
      "quantified": {
        "discount_vs_on_demand_pct": 50
      },
      "assumptions": [
        "A meaningful share of inference can be delayed and optimized for cost/throughput instead of TTFT."
      ],
      "contradictions_or_limits": [
        "List-price discount does not prove realized cost once queueing, data movement, retries, and deadlines are included."
      ],
      "fak_implications": [
        "Keep offline batch and online continuous batching as distinct workload classes and compare net completion cost."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ietf-llm-workload-profiles-2026",
      "category": "standard",
      "entity": "IETF Internet-Draft authors",
      "topic": [
        "benchmarking",
        "workload_profiles",
        "concurrency",
        "cache_behavior"
      ],
      "published_at": "2026-04-10",
      "event_at": "2026-04-10",
      "source_title": "Benchmarking Workload Profiles for Large Language Model Inference Serving",
      "source_url": "https://datatracker.ietf.org/doc/draft-mondal-llm-serving-workload-profiles/",
      "source_kind": "internet_draft",
      "evidence_class": "official_statement",
      "confidence": "medium",
      "claim": "An Internet-Draft proposes six benchmark profile groups defined by input/output distributions, output structure, latency sensitivity, concurrency, and caching behavior: non-generative, minimal-output, interactive streaming, prefill-heavy, decode-heavy, and multi-step chained.",
      "quantified": {
        "profile_groups": 6
      },
      "assumptions": [
        "No single benchmark profile represents all production inference; latency and caching behavior are workload-defining dimensions."
      ],
      "contradictions_or_limits": [
        "An Internet-Draft is mutable and not an adopted standard or production measurement."
      ],
      "fak_implications": [
        "Map fak benchmarks to named workload profiles and retain the exact draft/version used."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "xai-series-e-2026",
      "category": "frontier_lab",
      "entity": "xAI",
      "topic": [
        "cluster_scale",
        "userbase",
        "funding"
      ],
      "published_at": "2026-01-06",
      "event_at": "2026-01-06",
      "source_title": "xAI Raises $20B Series E",
      "source_url": "https://x.ai/news/series-e",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "xAI said Colossus I and II ended 2025 with more than one million H100-equivalent GPUs and that X plus Grok reached about 600 million monthly active users.",
      "quantified": {
        "funding_usd": 20000000000,
        "h100_equivalent_gpu_count_gt": 1000000,
        "monthly_active_users_approx": 600000000
      },
      "assumptions": [
        "A single lab can pair million-accelerator infrastructure with a consumer-scale distribution channel."
      ],
      "contradictions_or_limits": [
        "H100-equivalent is not a physical SKU count, and monthly active users do not reveal request or token distributions."
      ],
      "fak_implications": [
        "Keep physical hardware, normalized equivalents, active users, requests, and tokens as separate measures."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "xai-grok4-colossus-2025",
      "category": "frontier_lab",
      "entity": "xAI",
      "topic": [
        "reinforcement_learning",
        "cluster_scale",
        "test_time_compute"
      ],
      "published_at": "2025-07-09",
      "event_at": "2025-07-09",
      "source_title": "Grok 4",
      "source_url": "https://x.ai/news/grok-4",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "xAI said Grok 4 reinforcement learning used a 200,000-GPU Colossus cluster at pretraining scale and that stack changes improved training compute efficiency sixfold.",
      "quantified": {
        "gpu_count": 200000,
        "claimed_training_compute_efficiency_gain": 6
      },
      "assumptions": [
        "Reasoning post-training can consume pretraining-scale compute rather than being a small finishing stage."
      ],
      "contradictions_or_limits": [
        "The efficiency baseline, GPU mix, goodput, and energy are not fully specified."
      ],
      "fak_implications": [
        "Capacity accounting should separate pretraining, post-training/RL, evaluation, and inference demand."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "xai-reasoning-duration-2025",
      "category": "workload_model",
      "entity": "xAI Grok 3",
      "topic": [
        "reasoning",
        "output_length",
        "test_time_compute"
      ],
      "published_at": "2025-02-19",
      "event_at": "2025-02-19",
      "source_title": "Grok 3 Beta — The Age of Reasoning Agents",
      "source_url": "https://x.ai/news/grok-3",
      "source_kind": "official_model_release",
      "evidence_class": "production_observation",
      "confidence": "medium_high",
      "claim": "xAI described reasoning requests as spending from seconds to minutes and reported a highest-compute evaluation using 64 sampled chains.",
      "quantified": {
        "high_compute_sampling": "cons@64"
      },
      "assumptions": [
        "Reasoning-service time and sampled-compute demand can vary by orders of magnitude across requests and settings."
      ],
      "contradictions_or_limits": [
        "Product prose does not provide a production duration distribution or user-selection rate."
      ],
      "fak_implications": [
        "Model reasoning budget as a request-controlled distribution, not a fixed output length."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mistral-compute-2025",
      "category": "frontier_lab",
      "entity": "Mistral AI",
      "topic": [
        "sovereign_cloud",
        "vertical_integration",
        "deployment_choice"
      ],
      "published_at": "2025-06-11",
      "event_at": "2025-06-11",
      "source_title": "Mistral Compute",
      "source_url": "https://mistral.ai/news/mistral-compute/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Mistral announced a vertically integrated European infrastructure offering spanning GPUs, orchestration, APIs, products, bare metal, and managed PaaS, motivated by scarce expensive GPUs and patchy tooling.",
      "quantified": {},
      "assumptions": [
        "Sovereign buyers want infrastructure and control boundaries, not only a serverless model API."
      ],
      "contradictions_or_limits": [
        "The announcement does not disclose fleet size, utilization, or delivered economics."
      ],
      "fak_implications": [
        "Treat deployment/control preference and sovereignty as workload-placement constraints."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mistral-environmental-lca-2025",
      "category": "frontier_lab",
      "entity": "Mistral AI",
      "topic": [
        "energy",
        "water",
        "lifecycle",
        "inference"
      ],
      "published_at": "2025-07-22",
      "event_at": "2025-07-22",
      "source_title": "Our contribution to a global environmental standard for AI",
      "source_url": "https://mistral.ai/news/our-contribution-to-a-global-environmental-standard-for-ai/",
      "source_kind": "official_report",
      "evidence_class": "production_observation",
      "confidence": "medium_high",
      "claim": "Mistral reported lifecycle impacts for training Mistral Large 2 and marginal impacts for a 400-token Le Chat response, including carbon, water, and resource-depletion metrics.",
      "quantified": {
        "training_ghg_tco2e": 20.4,
        "training_water_m3": 281000,
        "response_tokens": 400,
        "inference_ghg_gco2e": 1.14,
        "inference_water_ml": 45
      },
      "assumptions": [
        "Environmental accounting can be attached to concrete training and inference units rather than facility totals alone."
      ],
      "contradictions_or_limits": [
        "The scope, geography, amortization, and model-serving stack limit cross-provider comparison."
      ],
      "fak_implications": [
        "Report energy/water boundaries and per-accepted-token impacts with methodology, not marketing aggregates."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mistral3-training-2025",
      "category": "frontier_lab",
      "entity": "Mistral AI",
      "topic": [
        "training_cluster",
        "moe",
        "inference_stack"
      ],
      "published_at": "2025-12-02",
      "event_at": "2025-12-02",
      "source_title": "Introducing Mistral 3",
      "source_url": "https://mistral.ai/news/mistral-3/",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Mistral said its 675B-total/41B-active MoE Large 3 was trained from scratch on 3,000 H200 GPUs and was being optimized for prefill/decode disaggregation and speculative decoding on rack-scale systems.",
      "quantified": {
        "training_h200_gpu_count": 3000,
        "total_parameters": 675000000000,
        "active_parameters": 41000000000
      },
      "assumptions": [
        "Frontier-capable open models may be trained on clusters far smaller than million-accelerator fleet announcements, while serving still benefits from phase-specific optimization."
      ],
      "contradictions_or_limits": [
        "The source does not disclose training duration, utilization, tokens, failures, or full cost."
      ],
      "fak_implications": [
        "Do not infer required cluster size solely from model capability; preserve architecture and training-efficiency context."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "deepseek-v3-cluster-2024",
      "category": "frontier_lab",
      "entity": "DeepSeek",
      "topic": [
        "training_cluster",
        "moe",
        "networking",
        "efficiency"
      ],
      "published_at": "2024-12-27",
      "event_at": "2024-12-27",
      "source_title": "DeepSeek-V3 Technical Report",
      "source_url": "https://arxiv.org/html/2412.19437v1",
      "source_kind": "technical_report",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "DeepSeek-V3 used 2,048 H800 GPUs in 8-GPU NVLink/NVSwitch nodes with InfiniBand across nodes, reporting 2.788M H800 GPU-hours for full training and no irrecoverable loss spikes or rollback.",
      "quantified": {
        "h800_gpu_count": 2048,
        "gpu_hours": 2788000,
        "pretraining_tokens": 14800000000000
      },
      "assumptions": [
        "Model/hardware co-design and overlap of communication can make a modest cluster competitive with much larger public buildouts."
      ],
      "contradictions_or_limits": [
        "Self-reported training cost excludes broader R&D, failed experiments, data, inference, and facility overhead."
      ],
      "fak_implications": [
        "Compare net training runs with boundaries and reliability events, not headline GPU count alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "deepseek-hardware-reflections-2025",
      "category": "frontier_lab",
      "entity": "DeepSeek",
      "topic": [
        "memory",
        "interconnect",
        "low_precision",
        "co_design"
      ],
      "published_at": "2025-05-14",
      "event_at": "2025-05-14",
      "source_title": "Insights into DeepSeek-V3: Scaling Challenges and Reflections on Hardware for AI Architectures",
      "source_url": "https://arxiv.org/abs/2505.09343",
      "source_kind": "technical_report",
      "evidence_class": "production_observation",
      "confidence": "high",
      "claim": "DeepSeek identified memory capacity, compute efficiency, and interconnect bandwidth as binding constraints, and highlighted FP8, MLA, MoE communication overlap, and multi-plane networking as co-design responses.",
      "quantified": {},
      "assumptions": [
        "Frontier efficiency depends on matching model architecture to hardware bottlenecks rather than scaling hardware alone."
      ],
      "contradictions_or_limits": [
        "The analysis is authored by the model team and does not independently compare every alternative."
      ],
      "fak_implications": [
        "Prioritize native kernels and topology-aware communication where the real model exposes those bottlenecks."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "qwen3-thinking-budget-2025",
      "category": "frontier_lab",
      "entity": "Alibaba Qwen",
      "topic": [
        "reasoning",
        "budgeting",
        "model_mix"
      ],
      "published_at": "2025-05-14",
      "event_at": "2025-05-14",
      "source_title": "Qwen3 Technical Report",
      "source_url": "https://arxiv.org/abs/2505.09388",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Qwen3 unified thinking and non-thinking modes and exposed a thinking-budget mechanism to allocate inference compute by task complexity.",
      "quantified": {
        "model_size_range_billion_parameters": [
          0.6,
          235
        ]
      },
      "assumptions": [
        "User- or router-controlled reasoning budgets create multiple service classes within one model family."
      ],
      "contradictions_or_limits": [
        "The report does not establish production selection frequencies or exact service-time distributions."
      ],
      "fak_implications": [
        "Benchmarks should sweep reasoning budgets and routing policies rather than use one fixed mode."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "kimi-k2-cluster-2025",
      "category": "frontier_lab",
      "entity": "Moonshot AI",
      "topic": [
        "training_cluster",
        "roce",
        "fault_tolerance",
        "moe"
      ],
      "published_at": "2025-07-28",
      "event_at": "2025-07-28",
      "source_title": "Kimi K2: Open Agentic Intelligence",
      "source_url": "https://arxiv.org/html/2507.20534v1",
      "source_kind": "technical_report",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "Kimi K2 was trained on H800 nodes with eight GPUs, 2 TB host RAM, and eight 400 Gbps RoCE links across nodes; the report treats fault tolerance and large-scale MoE communication as core infrastructure.",
      "quantified": {
        "gpus_per_node": 8,
        "host_ram_tb": 2,
        "roce_links_per_node": 8,
        "roce_link_gbps": 400
      },
      "assumptions": [
        "Frontier clusters can use Ethernet/RoCE rather than InfiniBand if topology, congestion, and collective behavior are engineered together."
      ],
      "contradictions_or_limits": [
        "The indexed excerpt does not state total GPU count or production serving distribution."
      ],
      "fak_implications": [
        "Keep fabric type, topology, collective efficiency, and failure recovery in training receipts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "minimax-m3-context-2026",
      "category": "frontier_lab",
      "entity": "MiniMax",
      "topic": [
        "long_context",
        "multimodal",
        "agents"
      ],
      "published_at": "2026-06-01",
      "event_at": "2026-06-01",
      "source_title": "MiniMax M3: Frontier Coding, 1M Context, Native Multimodality",
      "source_url": "https://www.minimax.io/blog/minimax-m3",
      "source_kind": "official_model_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "MiniMax M3 advertises a one-million-token context, native image/video input, and computer-use capability for coding and agentic workloads.",
      "quantified": {
        "context_tokens": 1000000
      },
      "assumptions": [
        "Long-context multimodal agents create heterogeneous prefill, media-processing, tool, and decode demand."
      ],
      "contradictions_or_limits": [
        "Maximum supported context does not show the production context-length distribution or SLO at that limit."
      ],
      "fak_implications": [
        "Keep supported maximum distinct from observed percentiles and benchmark multimodal/tool phases separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-richland-5gw-2026",
      "category": "datacenter_physical",
      "entity": "Meta",
      "topic": [
        "power",
        "capex",
        "community",
        "generation"
      ],
      "published_at": "2026-07-13",
      "event_at": "2026-07-13",
      "source_title": "Teachers and Local Businesses Win as Meta Expands Louisiana Data Center",
      "source_url": "https://about.fb.com/news/2026/07/teachers-local-businesses-win-as-meta-expands-louisiana-data-center/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta said the Richland Parish datacenter would expand to 5 GW of compute capacity with more than $50B invested, over $1B in local infrastructure, and new gas generation, batteries, and nuclear uprates.",
      "quantified": {
        "compute_capacity_gw": 5,
        "investment_usd_gt": 50000000000,
        "local_infrastructure_usd_gt": 1000000000
      },
      "assumptions": [
        "A single AI campus can become a regional power, infrastructure, labor, and fiscal program."
      ],
      "contradictions_or_limits": [
        "Compute-capacity wording may not equal delivered IT load; community-benefit claims are company-authored."
      ],
      "fak_implications": [
        "Attach generation mix, transmission, water, workforce, and local-cost allocation to site capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tair-kvcache-hisim-2026",
      "category": "serving_system",
      "entity": "Alibaba Cloud Tair",
      "topic": [
        "kv_cache",
        "simulation",
        "tiering",
        "slo"
      ],
      "published_at": "2026-05-22",
      "event_at": "2026-05-22",
      "source_title": "Alibaba Cloud Tair KVCache Simulation Analysis",
      "source_url": "https://www.alibabacloud.com/blog/alibaba-cloud-tair-kvcache-simulation-analysis-high-precision-computational-and-caching-simulation-design-and-implementation_603164",
      "source_kind": "official_engineering_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Tair describes multi-tier KV cache as system infrastructure for agents and long context, using real-workload replay to co-optimize GPU, storage media, bandwidth, prefetch, eviction, latency, throughput, and cost under SLOs.",
      "quantified": {
        "claimed_prediction_error_pct_lt": 5,
        "claimed_cost_advantage_vs_hardware_trials": 390000
      },
      "assumptions": [
        "KV storage, compute, network, and policy form one high-dimensional operating envelope."
      ],
      "contradictions_or_limits": [
        "Accuracy and cost are vendor claims and need independent trace/hardware validation."
      ],
      "fak_implications": [
        "Use simulation for search, then require real native endpoint witnesses before accepting a policy."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-rainier-operational-2025",
      "category": "hyperscaler",
      "entity": "AWS / Anthropic",
      "topic": [
        "cluster_scale",
        "multi_site",
        "delivery"
      ],
      "published_at": "2025-06-24",
      "event_at": "2025-06-24",
      "source_title": "AWS activates Project Rainier: One of the world’s largest AI compute clusters comes online",
      "source_url": "https://www.aboutamazon.com/news/aws/aws-project-rainier-ai-trainium-chips-compute-cluster",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AWS said Project Rainier became operational with nearly 500,000 Trainium2 chips across multiple U.S. datacenters less than a year after announcement, with Anthropic already running workloads.",
      "quantified": {
        "trainium2_chips_approx": 500000
      },
      "assumptions": [
        "A named cluster can span multiple datacenters and move from announcement to workload use on a sub-year delivery cycle."
      ],
      "contradictions_or_limits": [
        "AWS does not disclose healthy-chip fraction, topology efficiency, failures, utilization, or useful goodput."
      ],
      "fak_implications": [
        "Record cluster geography and delivery state plus healthy/schedulable/goodput conversion."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-ml-demand-response-2025",
      "category": "hyperscaler",
      "entity": "Google",
      "topic": [
        "demand_response",
        "power",
        "batching"
      ],
      "published_at": "2025-08-04",
      "event_at": "2025-08-04",
      "source_title": "How we’re making data centers more flexible to benefit power grids",
      "source_url": "https://blog.google/innovation-and-ai/infrastructure-and-cloud/global-network/how-were-making-data-centers-more-flexible-to-benefit-power-grids/",
      "source_kind": "official_engineering_release",
      "evidence_class": "production_observation",
      "confidence": "high",
      "claim": "Google said it shifted or reduced machine-learning workload power during grid events and signed utility agreements to make datacenter demand response operational.",
      "quantified": {},
      "assumptions": [
        "Some ML work is deferrable or movable enough to act as a grid resource."
      ],
      "contradictions_or_limits": [
        "The source does not disclose which AI workloads moved, performance impact, or capacity fraction."
      ],
      "fak_implications": [
        "Classify work by deadline/mobility and account for grid-driven scheduling as an operating constraint."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "kv-robust-reservation-2026",
      "category": "serving_system",
      "entity": "Robust KV cache management study",
      "topic": [
        "output_uncertainty",
        "kv_cache",
        "reservation",
        "heterogeneous_clusters"
      ],
      "published_at": "2026-07-18",
      "event_at": "2026-07-18",
      "source_title": "Robust KV Cache Management for LLM Serving under Output Token Length Uncertainty",
      "source_url": "https://arxiv.org/abs/2607.16892",
      "source_kind": "preprint",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium_high",
      "claim": "The study jointly optimizes parallelism, per-class KV reservation, heterogeneous-group routing, and prefix caching under distribution shift, reporting up to 56% lower cost on trace-driven evaluations.",
      "quantified": {
        "claimed_cost_reduction_pct_up_to": 56
      },
      "assumptions": [
        "Unknown output length creates a quantile-dependent tradeoff between memory waste and preemption/recompute risk."
      ],
      "contradictions_or_limits": [
        "Trace-driven optimization is not direct production deployment evidence; results depend on cost and SLO assumptions."
      ],
      "fak_implications": [
        "Model reservation uncertainty and distribution shift; verify on native production-like traces before adoption."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "cacheroute-2026",
      "category": "serving_system",
      "entity": "CacheRoute",
      "topic": [
        "prefix_affinity",
        "routing",
        "hot_keys",
        "load_balance"
      ],
      "published_at": "2026-08-20",
      "event_at": "2026-08-20",
      "source_title": "CacheRoute: Planned Prefix-Affinity Routing for Large-Scale LLM Serving",
      "source_url": "https://arxiv.org/abs/2608.19677",
      "source_kind": "preprint",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium_high",
      "claim": "CacheRoute plans stable warm destinations for high-rate prefix keys; on a semi-synthetic 60-H100 workload it raised served cache hit rate from 64.1% to 93.2% and throughput 2.3x, but counterexamples erased gains when recoverable KV work was small.",
      "quantified": {
        "h100_gpu_count": 60,
        "baseline_cache_hit_pct": 64.1,
        "planned_cache_hit_pct": 93.2,
        "throughput_gain": 2.3,
        "p99_slo_seconds": 3.5
      },
      "assumptions": [
        "Prefix-key popularity and affinity can be exploited only when expected reuse outweighs load skew."
      ],
      "contradictions_or_limits": [
        "Primary evidence is semi-synthetic and includes negative 32B counterexamples."
      ],
      "fak_implications": [
        "Require shadow replay and a no-gain gate rather than enabling affinity universally."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "multitenant-cache-admission-2026",
      "category": "serving_system",
      "entity": "Multi-tenant prefix-cache admission study",
      "topic": [
        "multi_tenancy",
        "fairness",
        "cache_admission"
      ],
      "published_at": "2026-08-03",
      "event_at": "2026-08-03",
      "source_title": "Preserving Admission Responsibility in Multi-Tenant Large Language Model Inference",
      "source_url": "https://arxiv.org/html/2608.01657v1",
      "source_kind": "preprint",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium",
      "claim": "The paper identifies an admission-responsibility gap: one tenant can materialize KV blocks that evict another tenant’s reusable state, while ordinary request schedulers and eviction policies do not assign responsibility.",
      "quantified": {},
      "assumptions": [
        "Shared prefix cache is persistent cross-tenant state whose admission can impose externalities."
      ],
      "contradictions_or_limits": [
        "The proposal is research evidence, not a disclosed frontier-provider policy."
      ],
      "fak_implications": [
        "Account cache admission and eviction by tenant, cap externalities, and expose fairness/read-back metrics."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ibm-together-b300-2026",
      "category": "ai_cloud",
      "entity": "IBM Cloud / Together AI",
      "topic": [
        "inference_cluster",
        "partnership",
        "enterprise"
      ],
      "published_at": "2026-08-11",
      "event_at": "2026-08-11",
      "source_title": "IBM and Together AI Sign Multi-Year Agreement to Scale Open-Source AI Inference",
      "source_url": "https://newsroom.ibm.com/2026-08-11-IBM-and-Together-AI-Sign-Multi-Year-Agreement-to-Scale-Open-Source-AI-Inference-with-NVIDIA-AI-Infrastructure-on-IBM-Cloud",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "IBM and Together AI announced a multi-year agreement for a large-scale enterprise inference cluster using NVIDIA HGX B300 systems on IBM Cloud.",
      "quantified": {},
      "assumptions": [
        "Open-model inference platforms increasingly pair specialized serving software with incumbent cloud enterprise distribution."
      ],
      "contradictions_or_limits": [
        "Cluster size, customer distribution, SLO, and economics are undisclosed."
      ],
      "fak_implications": [
        "Track go-live state and measured tenant/workload mix rather than partnership announcements alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-nscale-45b-2026",
      "category": "market_signal",
      "entity": "Anthropic / Nscale",
      "topic": [
        "lease",
        "power",
        "natural_gas",
        "ipo"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Anthropic agrees $45bn AI data centre deal with UK start-up Nscale",
      "source_url": "https://www.ft.com/content/0ec76ba3-5f7f-4085-88fb-acf21954bc85",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Financial Times reported a six-year $45B agreement for Anthropic to use 460 MW of Vera Rubin capacity from a planned West Virginia Nscale facility beginning in late 2027.",
      "quantified": {
        "contract_value_usd": 45000000000,
        "term_years": 6,
        "anthropic_capacity_mw": 460,
        "facility_power_gw": 1.35,
        "generation_power_gw": 2
      },
      "assumptions": [
        "Frontier labs are using long-duration leases from startups backed by dedicated generation to diversify supply."
      ],
      "contradictions_or_limits": [
        "The site and processors are prospective, and commercial terms are reported rather than fully public."
      ],
      "fak_implications": [
        "Model credit, construction, generation, and counterparty risk in future-capacity leases."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-decart-rumor-2026",
      "category": "market_signal",
      "entity": "Anthropic / Decart",
      "topic": [
        "acquisition",
        "inference_infrastructure",
        "rumor"
      ],
      "published_at": "2026-08-13",
      "event_at": "2026-08-13",
      "source_title": "Anthropic in Talks for $6 Billion AI Infrastructure Bet",
      "source_url": "https://www.youtube.com/watch?v=XeXMUOigCaQ",
      "source_kind": "credible_reporting",
      "evidence_class": "rumor",
      "confidence": "low",
      "claim": "Bloomberg reported that Anthropic was in talks to acquire Decart for roughly $6B; Reuters later independently reported the acquisition talks, while both companies declined to comment and no final agreement was recorded.",
      "quantified": {
        "reported_deal_value_usd": 6000000000
      },
      "assumptions": [
        "Frontier labs may acquire infrastructure optimization capability rather than only lease capacity."
      ],
      "contradictions_or_limits": [
        "Independent reporting supports that talks occurred, but the $6B term, final scope, closing, and integration remain unconfirmed.",
        "Talks can change or collapse and must not be treated as acquired capacity, staff, IP, or product."
      ],
      "fak_implications": [
        "Watch Decart’s technical seam and any official filing, but do not treat the acquisition as fact."
      ],
      "rumor": {
        "is_rumor": true,
        "origin": "Bloomberg reporting summarized in Bloomberg Technology video",
        "corroboration": "Bloomberg, Reuters, Axios, and Israeli technology reporting describe talks; neither Anthropic nor Decart announced a signed or completed transaction by August 27, 2026.",
        "status": "independently_corroborated_open",
        "last_checked_at": "2026-08-27",
        "expires_at": "2026-11-30",
        "resolution": "Talks remain independently corroborated but open; the roughly $6B term, final scope, signing, closing, and integration remain unconfirmed."
      }
    },
    {
      "id": "openai-datacenter-leadership-rumor-2026",
      "category": "market_signal",
      "entity": "OpenAI",
      "topic": [
        "organization",
        "datacenter_strategy",
        "rumor"
      ],
      "published_at": "2026-08-25",
      "event_at": "2026-08-25",
      "source_title": "OpenAI’s Head of Data Centers Has Left the Company",
      "source_url": "https://www.wsj.com/tech/ai/openais-head-of-data-centers-has-left-company-6d24fd83",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "OpenAI confirmed that data-center head Chris Malone left the company. Reporting also described an infrastructure-team reorganization and renewed full-facility leasing, but causal and strategic interpretations remain reported rather than primary-confirmed.",
      "quantified": {},
      "assumptions": [
        "Organizational execution and build-versus-lease strategy can alter capacity delivery independent of announced capital."
      ],
      "contradictions_or_limits": [
        "The personnel departure is confirmed; claims about why he left and the strategic effect of the reorganization are not established by a primary detailed statement.",
        "A leadership change does not prove cancellation, delay, or change in delivered capacity."
      ],
      "fak_implications": [
        "Treat leadership and strategy reports as delivery-risk signals only until confirmed."
      ],
      "rumor": {
        "is_rumor": false,
        "resolution_status": "partially_confirmed",
        "resolved_at": "2026-08-26",
        "confirmed_fragment": "Chris Malone left OpenAI.",
        "unresolved_fragment": "Causal and strategic interpretations of the departure and team reorganization.",
        "resolution_sources": [
          "OpenAI spokesperson confirmation reported by The Next Web and subsequent outlets",
          "Original Wall Street Journal report"
        ]
      }
    },
    {
      "id": "anthropic-fluidstack-50b-2025",
      "category": "frontier_lab",
      "entity": "Anthropic / Fluidstack",
      "topic": [
        "datacenter_buildout",
        "jobs",
        "delivery"
      ],
      "published_at": "2025-11-12",
      "event_at": "2025-11-12",
      "source_title": "Anthropic invests $50 billion in American AI infrastructure",
      "source_url": "https://www.anthropic.com/news/anthropic-invests-50-billion-in-american-ai-infrastructure",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic announced a $50B U.S. infrastructure program with Fluidstack, with initial sites expected online through 2026.",
      "quantified": {
        "investment_usd": 50000000000,
        "permanent_jobs_approx": 800,
        "construction_jobs_approx": 2400
      },
      "assumptions": [
        "Frontier labs combine hyperscaler commitments with direct or partner-developed sites."
      ],
      "contradictions_or_limits": [
        "Investment headline and expected online dates do not prove installed capacity or useful goodput."
      ],
      "fak_implications": [
        "Track developer, site, delivery, and acceptance states for each capacity channel."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-energy-model-scale-2025",
      "category": "datacenter_physical",
      "entity": "Anthropic",
      "topic": [
        "training_power",
        "sector_forecast",
        "grid"
      ],
      "published_at": "2025-07-21",
      "event_at": "2025-07-21",
      "source_title": "Build AI in America: Anthropic Energy Report",
      "source_url": "https://www.anthropic.com/news/build-ai-in-america",
      "source_kind": "official_report",
      "evidence_class": "analyst_estimate",
      "confidence": "medium_high",
      "claim": "Anthropic projected that developing a single advanced model could need 2 GW in 2027 and 5 GW in 2028, framing power buildout as a model-development constraint.",
      "quantified": {
        "single_model_power_gw_2027": 2,
        "single_model_power_gw_2028": 5
      },
      "assumptions": [
        "Training demand may consume site-scale gigawatts for one frontier program."
      ],
      "contradictions_or_limits": [
        "These are company projections, not observed metered demand, and boundaries may include supporting capacity."
      ],
      "fak_implications": [
        "Keep scenario probabilities and do not promote forecast gigawatts into current workload facts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-sector-50gw-2026",
      "category": "datacenter_physical",
      "entity": "Anthropic",
      "topic": [
        "sector_power",
        "ratepayer",
        "grid"
      ],
      "published_at": "2026-02-11",
      "event_at": "2026-02-11",
      "source_title": "Covering electricity price increases from our data centers",
      "source_url": "https://www.anthropic.com/news/covering-electricity-price-increases",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic said the U.S. AI sector would need at least 50 GW over the next several years and proposed that datacenter developers cover attributable electricity-system costs.",
      "quantified": {
        "us_ai_sector_capacity_gw_at_least": 50
      },
      "assumptions": [
        "Ratepayer allocation and utility tariff design are part of infrastructure feasibility and social license."
      ],
      "contradictions_or_limits": [
        "The sector demand is a company estimate and lacks a fully disclosed forecasting model."
      ],
      "fak_implications": [
        "Record who pays generation, transmission, distribution, and interconnection upgrades."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-govcloud-1p3gw-2025",
      "category": "hyperscaler",
      "entity": "AWS",
      "topic": [
        "government_cloud",
        "sovereignty",
        "capacity"
      ],
      "published_at": "2025-11-24",
      "event_at": "2025-11-24",
      "source_title": "Amazon to invest up to $50 billion to expand AI and supercomputing infrastructure for US government agencies",
      "source_url": "https://www.aboutamazon.com/news/company-news/amazon-ai-investment-us-federal-agencies",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AWS announced up to $50B to add nearly 1.3 GW across Top Secret, Secret, and GovCloud regions, with construction starting in 2026.",
      "quantified": {
        "investment_usd_up_to": 50000000000,
        "capacity_gw_approx": 1.3
      },
      "assumptions": [
        "Classified and sovereign workloads need dedicated regional infrastructure and security boundaries."
      ],
      "contradictions_or_limits": [
        "Construction start is not online capacity; AI and HPC are combined."
      ],
      "fak_implications": [
        "Preserve classification/sovereignty and separate mixed HPC/AI capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-ai-factories-2025",
      "category": "hyperscaler",
      "entity": "AWS",
      "topic": [
        "on_premises",
        "sovereign",
        "integrated_system"
      ],
      "published_at": "2025-12-02",
      "event_at": "2025-12-02",
      "source_title": "New AWS AI Factories transform customers existing infrastructure into AI-optimized environments",
      "source_url": "https://www.aboutamazon.com/news/aws/aws-data-centers-ai-factories",
      "source_kind": "official_product_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AWS AI Factories place AWS-managed NVIDIA/Trainium, networking, storage, and AI services in customer-provided datacenter space, power, and connectivity.",
      "quantified": {},
      "assumptions": [
        "Cloud control planes can extend into customer facilities when sovereignty, latency, or existing power/space dominate placement."
      ],
      "contradictions_or_limits": [
        "The announcement provides no production scale, economics, or workload distribution."
      ],
      "fak_implications": [
        "Model ownership and receipt boundaries separately from physical location."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-cerebras-phase-split-2026",
      "category": "serving_system",
      "entity": "AWS / Cerebras",
      "topic": [
        "prefill_decode",
        "heterogeneous_accelerators",
        "networking"
      ],
      "published_at": "2026-03-13",
      "event_at": "2026-03-13",
      "source_title": "AWS and Cerebras collaboration aims to set a new standard for AI inference speed",
      "source_url": "https://www.aboutamazon.com/news/aws/aws-cerebras-ai-inference",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "AWS announced a Bedrock system pairing Trainium for prefill with Cerebras CS-3 for decode over EFA networking.",
      "quantified": {},
      "assumptions": [
        "Inference phases may be mapped to different accelerator architectures, not only different GPU pools."
      ],
      "contradictions_or_limits": [
        "The solution was announced for future launch and superlative speed claims lacked a public matched production receipt."
      ],
      "fak_implications": [
        "Measure cross-architecture transfer, queueing, quality, availability, and cost before adopting phase specialization."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-domestic-supply-rfp-2026",
      "category": "supply_chain",
      "entity": "OpenAI",
      "topic": [
        "manufacturing",
        "procurement",
        "resilience"
      ],
      "published_at": "2026-01-15",
      "event_at": "2026-01-15",
      "source_title": "Strengthening the US AI supply chain through domestic manufacturing",
      "source_url": "https://openai.com/index/strengthening-the-us-ai-supply-chain/",
      "source_kind": "official_rfp",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "OpenAI launched an RFP for U.S. manufacturing capacity across datacenter inputs, consumer electronics, and robotics to shorten infrastructure timelines and improve resilience.",
      "quantified": {},
      "assumptions": [
        "Frontier buildout depends on a broad manufactured bill of materials and procurement strategy beyond accelerators."
      ],
      "contradictions_or_limits": [
        "An RFP signals a perceived constraint but does not quantify shortages or contracted output."
      ],
      "fak_implications": [
        "Maintain a component-level supply ledger and distinguish discovery from secured supply."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "iea-2026-electricity-update",
      "category": "datacenter_physical",
      "entity": "International Energy Agency",
      "topic": [
        "electricity",
        "forecast",
        "ai_growth"
      ],
      "published_at": "2026-04-16",
      "event_at": "2026-04-16",
      "source_title": "Data centre electricity use surged in 2025, even with tightening bottlenecks",
      "source_url": "https://www.iea.org/news/data-centre-electricity-use-surged-in-2025-even-with-tightening-bottlenecks-driving-a-scramble-for-solutions",
      "source_kind": "intergovernmental_report",
      "evidence_class": "analyst_estimate",
      "confidence": "high",
      "claim": "The IEA said datacenter electricity demand rose 17% in 2025 and projected roughly 485 TWh in 2025 to 950 TWh in 2030, with AI-focused demand growing faster.",
      "quantified": {
        "2025_growth_pct": 17,
        "electricity_twh_2025": 485,
        "electricity_twh_2030_approx": 950
      },
      "assumptions": [
        "Sector demand is growing faster than grid and general electricity demand, increasing site and timing constraints."
      ],
      "contradictions_or_limits": [
        "Forecasts depend on utilization, efficiency, model demand, and project completion."
      ],
      "fak_implications": [
        "Carry forecast ranges and update them rather than freezing one report year."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "iea-grid-delay-2025",
      "category": "datacenter_physical",
      "entity": "International Energy Agency",
      "topic": [
        "grid_queue",
        "delay_probability",
        "capacity_lifecycle"
      ],
      "published_at": "2025-04-10",
      "event_at": "2025-04-10",
      "source_title": "AI and energy security",
      "source_url": "https://www.iea.org/reports/energy-and-ai/ai-and-energy-security",
      "source_kind": "intergovernmental_report",
      "evidence_class": "analyst_estimate",
      "confidence": "high",
      "claim": "A location-specific IEA analysis estimated grid constraints could delay around 20% of global datacenter capacity planned through 2030.",
      "quantified": {
        "global_planned_capacity_delay_pct_approx": 20
      },
      "assumptions": [
        "Grid connection is a probabilistic delivery gate that should discount announced capacity."
      ],
      "contradictions_or_limits": [
        "The estimate is scenario-based and varies by location and policy."
      ],
      "fak_implications": [
        "Attach grid-connection confidence and delay distributions to each planned site."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "semianalysis-half-delayed-rebuttal-2026",
      "category": "market_signal",
      "entity": "SemiAnalysis",
      "topic": [
        "contradiction",
        "delivery_pipeline",
        "methodology"
      ],
      "published_at": "2026-06-18",
      "event_at": "2026-06-18",
      "source_title": "Stop Saying Half of 2026 US Datacenter Capacity Is Delayed",
      "source_url": "https://newsletter.semianalysis.com/p/stop-saying-half-of-2026-us-datacenter",
      "source_kind": "industry_analysis",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "SemiAnalysis challenged the widely repeated claim that half of 2026 U.S. datacenter capacity was delayed or canceled, tracing it to one Bloomberg framing and arguing that boundaries and project states were being conflated.",
      "quantified": {},
      "assumptions": [
        "Headline pipeline failure rates are sensitive to denominator, expected-delivery date, and project-state definitions."
      ],
      "contradictions_or_limits": [
        "The full analysis may rely on proprietary tracking and is itself a secondary interpretation."
      ],
      "fak_implications": [
        "Keep both claim and rebuttal; require a project roster and common lifecycle denominator before using a cancellation percentage."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ai-dc-power-architecture-2026",
      "category": "datacenter_physical",
      "entity": "AI datacenter power architecture research",
      "topic": [
        "power_delivery",
        "rack_voltage",
        "transients"
      ],
      "published_at": "2026-06-23",
      "event_at": "2026-06-23",
      "source_title": "Toward Next-Generation AI Data Centers: Power Delivery Architecture Shifts, Emerging Technologies, and Challenges",
      "source_url": "https://arxiv.org/abs/2606.25095",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "The review argues AI current transients and thermal density expose limits in 48 V rack and conventional AC distribution, motivating higher-voltage DC conversion, facility low-voltage DC, and medium-voltage solid-state transformers.",
      "quantified": {},
      "assumptions": [
        "Rack and facility power architecture must evolve with accelerator density and transient behavior."
      ],
      "contradictions_or_limits": [
        "This is a review and design roadmap, not proof of broad deployment or reliability."
      ],
      "fak_implications": [
        "Include power-delivery topology and transient constraints in next-generation rack assumptions."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "deepseek-v3-economics-2024",
      "category": "frontier_lab",
      "entity": "DeepSeek",
      "topic": [
        "training_economics",
        "gpu_hours",
        "cost_boundary"
      ],
      "published_at": "2024-12-27",
      "event_at": "2024-12-27",
      "source_title": "DeepSeek-V3 Technical Report",
      "source_url": "https://arxiv.org/html/2412.19437v1",
      "source_kind": "technical_report",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "DeepSeek reported 2.788M H800 GPU-hours and a $5.576M rental-price estimate for the official V3 training run.",
      "quantified": {
        "gpu_hours": 2788000,
        "reported_training_cost_usd": 5576000,
        "assumed_gpu_hour_price_usd": 2
      },
      "assumptions": [
        "Algorithm/hardware co-design can reduce the compute cost of one successful run."
      ],
      "contradictions_or_limits": [
        "The headline excludes prior research, ablations, data, salaries, failed runs, infrastructure ownership, and serving."
      ],
      "fak_implications": [
        "Never compare one-run rental estimates with all-in lab or hyperscaler capex."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-amazon-bedrock-customer-scale-2026",
      "category": "frontier_lab",
      "entity": "Anthropic",
      "topic": [
        "userbase",
        "enterprise",
        "demand"
      ],
      "published_at": "2026-04-20",
      "event_at": "2026-04-20",
      "source_title": "Anthropic and Amazon expand collaboration for up to 5 gigawatts of new compute",
      "source_url": "https://www.anthropic.com/news/anthropic-amazon-compute",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Anthropic reported that more than 100,000 customers run Claude on Amazon Bedrock.",
      "quantified": {
        "amazon_bedrock_customers_running_claude_gt": 100000
      },
      "assumptions": [
        "The count establishes a Claude-on-Amazon-Bedrock customer denominator rather than Anthropic-wide users, seats, organizations, or direct API customers."
      ],
      "contradictions_or_limits": [
        "The source does not define whether a customer is an AWS account, organization, contract, or another unit, and one customer may contain many users, applications, workspaces, or API keys.",
        "The count excludes or does not separately identify direct Anthropic API, Claude.ai, Google Vertex AI, or Microsoft Foundry populations and is not requests, sessions, messages, tokens, concurrency, or model-specific traffic."
      ],
      "fak_implications": [
        "Represent the Amazon Bedrock customer population separately from direct Anthropic and other cloud-channel populations.",
        "Do not translate customer count into users, request load, traffic share, or workload shape."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mistral-claude-crosscloud-2026",
      "category": "market_signal",
      "entity": "Anthropic / Mistral / Microsoft",
      "topic": [
        "multi_provider",
        "azure",
        "sovereignty"
      ],
      "published_at": "2026-08-12",
      "event_at": "2026-08-12",
      "source_title": "Anthropic models available through European provider Mistral following Microsoft deal",
      "source_url": "https://www.reuters.com/business/media-telecom/anthropic-models-available-through-european-provider-mistral-following-microsoft-deal-2026-08-12/",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "high",
      "claim": "Reuters reported Claude models becoming available through Mistral on Azure following a Microsoft-Mistral agreement.",
      "quantified": {},
      "assumptions": [
        "Distribution channels can nest providers and clouds, complicating capacity, billing, policy, and provenance."
      ],
      "contradictions_or_limits": [
        "Commercial routing and traffic shares are undisclosed."
      ],
      "fak_implications": [
        "Receipts should retain model provider, reseller/platform, underlying cloud, region, and policy boundary."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "data-center-water-2025",
      "category": "datacenter_physical",
      "entity": "U.S. datacenter sector",
      "topic": [
        "water",
        "reporting",
        "externalities"
      ],
      "published_at": "2025-04-09",
      "event_at": "2025-04-09",
      "source_title": "Why AI data centers are driving up water consumption across the US",
      "source_url": "https://www.cnbc.com/2025/04/09/why-ai-data-centers-are-driving-up-water-consumption-across-the-us.html",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "CNBC reported rising water concern around U.S. AI datacenters and weak disclosure, emphasizing that cooling design, climate, and electricity generation determine direct and indirect water use.",
      "quantified": {},
      "assumptions": [
        "Water impact cannot be inferred from compute or power alone and may be locally binding."
      ],
      "contradictions_or_limits": [
        "Public reporting is incomplete and facility boundaries differ."
      ],
      "fak_implications": [
        "Record direct/indirect water boundary, cooling design, climate, and disclosure confidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "training-bubbles-2025",
      "category": "workload_model",
      "entity": "Large-scale training reliability study",
      "topic": [
        "failures",
        "recovery",
        "scheduling"
      ],
      "published_at": "2025-06-05",
      "event_at": "2025-06-05",
      "source_title": "Training Bubbles: Rescheduling at Scale for Distributed Training",
      "source_url": "https://arxiv.org/abs/2506.05552",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "Training Bubbles studies how to continue large distributed jobs through accelerator failures using redundant or spare capacity rather than full rollback.",
      "quantified": {},
      "assumptions": [
        "At large cluster scale, failures are routine workload events and recovery policy changes useful training goodput."
      ],
      "contradictions_or_limits": [
        "Research evaluation may not match every topology, optimizer, or frontier training stack."
      ],
      "fak_implications": [
        "Count failure detection, lost work, spare capacity, rescheduling, and recovery in training throughput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aws-nvidia-2m-2026",
      "category": "hyperscaler",
      "entity": "AWS / NVIDIA",
      "topic": [
        "capacity",
        "roadmap",
        "heterogeneity",
        "government"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "AWS and NVIDIA to Deliver 2 Million Additional GPUs and Next-Generation Infrastructure for Agentic and Physical AI",
      "source_url": "https://nvidianews.nvidia.com/news/aws-and-nvidia-to-deliver-2-million-additional-gpus-and-next-generation-infrastructure-for-agentic-and-physical-ai",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AWS and NVIDIA announced plans for two million additional Blackwell Ultra, Rubin, and Rubin Ultra GPUs across AWS in 2027-2028, including 100,000 GPUs for secure U.S. government infrastructure and integration of Trainium with NVLink Fusion/NVHBM.",
      "quantified": {
        "additional_gpu_count": 2000000,
        "delivery_start_year": 2027,
        "delivery_end_year": 2028,
        "government_gpu_count": 100000
      },
      "assumptions": [
        "Hyperscaler fleets will mix multiple NVIDIA generations, custom silicon, custom HBM/interconnect, and classified capacity."
      ],
      "contradictions_or_limits": [
        "This is a forward plan; demand, manufacturing, power, and deployment can change delivered counts."
      ],
      "fak_implications": [
        "Track generation, security domain, delivery year, and custom-silicon interoperability separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-groq3-production-2026",
      "category": "accelerator_platform",
      "entity": "NVIDIA / Groq",
      "topic": [
        "agentic_inference",
        "decode",
        "long_context",
        "production"
      ],
      "published_at": "2026-08-24",
      "event_at": "2026-08-24",
      "source_title": "NVIDIA Groq 3 LPX Now in Full Production With World-Class Speed for Agentic AI",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-groq-3-lpx-now-in-full-production-with-world-class-speed-for-agentic-ai",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "high",
      "claim": "NVIDIA said Groq 3 LPX entered full production as a decode/interactivity extension to Vera Rubin, with Nebius first to adopt and Groq planning adoption; a benchmark reported 3,400 output tokens/s for a 31B model at 100k context.",
      "quantified": {
        "benchmark_output_tokens_per_second": 3400,
        "benchmark_context_tokens": 100000,
        "benchmark_model_parameters_billion": 31
      },
      "assumptions": [
        "Agentic demand is being decomposed into context processing and ultra-low-latency generation at the rack-platform level."
      ],
      "contradictions_or_limits": [
        "The record is vendor-selected and model/context specific; availability through clouds and net economics require independent evidence."
      ],
      "fak_implications": [
        "Benchmark phase-specialized hardware on complete agent loops including tool time, quality, transfer, and utilization."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "leyline-agentic-cache-2026",
      "category": "serving_system",
      "entity": "Leyline",
      "topic": [
        "agentic",
        "kv_cache",
        "editing",
        "replay"
      ],
      "published_at": "2026-05-31",
      "event_at": "2026-05-31",
      "source_title": "Leyline: KV Cache Directives for Agentic Inference",
      "source_url": "https://arxiv.org/abs/2606.01065",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "Leyline argues append-only chatbot cache assumptions fail for agents that retry tools, drop stale output, pivot, or splice trajectories; its cache directives preserve valid work across edits.",
      "quantified": {
        "replay_cache_hit_gain_percentage_points": 11.2,
        "latency_reduction_ms_up_to": 241,
        "debug_solve_rate_gain_percentage_points": 14.3
      },
      "assumptions": [
        "Agent trajectories are editable state, not permanently append-only conversations."
      ],
      "contradictions_or_limits": [
        "Open benchmark results do not establish broad production adoption, semantic safety across all edits, or every attention architecture."
      ],
      "fak_implications": [
        "Expose policy-directed cache edits with correctness receipts and compare against re-prefill including quality."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-lumentum-optics-2026",
      "category": "supply_chain",
      "entity": "NVIDIA / Lumentum",
      "topic": [
        "optics",
        "lasers",
        "manufacturing",
        "capacity"
      ],
      "published_at": "2026-03-02",
      "event_at": "2026-03-02",
      "source_title": "NVIDIA Announces Strategic Partnership With Lumentum to Develop State-of-the-Art Optics Technology",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-announces-strategic-partnership-with-lumentum-to-develop-state-of-the-art-optics-technology",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "NVIDIA committed multibillion-dollar purchases and invested $2B in Lumentum to expand U.S. advanced-laser and optics manufacturing capacity for AI datacenters.",
      "quantified": {
        "nvidia_investment_usd": 2000000000
      },
      "assumptions": [
        "Advanced optics supply is strategic enough for accelerator vendors to finance manufacturing directly."
      ],
      "contradictions_or_limits": [
        "Investment and purchase commitments do not disclose delivered component volume or eliminate concentration risk."
      ],
      "fak_implications": [
        "Include lasers/optics capacity rights and manufacturing lead time in cluster delivery ledgers."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-coherent-optics-2026",
      "category": "supply_chain",
      "entity": "NVIDIA / Coherent",
      "topic": [
        "optics",
        "packaging",
        "manufacturing",
        "capacity"
      ],
      "published_at": "2026-03-02",
      "event_at": "2026-03-02",
      "source_title": "NVIDIA and Coherent Announce Strategic Partnership to Develop Optics Technology to Scale Next-Generation Data Center Architecture",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-and-coherent-announce-strategic-partnership-to-develop-optics-technology-to-scale-next-generation-data-center-architecture",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "NVIDIA committed multibillion-dollar purchases and invested $2B in Coherent for advanced laser, optical networking, package integration, R&D, and U.S. manufacturing capacity.",
      "quantified": {
        "nvidia_investment_usd": 2000000000
      },
      "assumptions": [
        "Optical interconnect and package integration are foundational constraints for scaling AI factories."
      ],
      "contradictions_or_limits": [
        "The source is a capacity commitment, not measured delivered network goodput or yield."
      ],
      "fak_implications": [
        "Track optical BOM, packaging integration, yield, and capacity access alongside accelerator orders."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "github-copilot-agentic-production-2026",
      "category": "workload_trace",
      "entity": "GitHub Copilot coding agent",
      "topic": [
        "coding_agents",
        "user_heterogeneity",
        "kv_cache",
        "tool_failures",
        "compaction"
      ],
      "published_at": "2026-07-30",
      "event_at": "2026-06-07",
      "source_title": "Agentic Coding in the Wild: Characterizing GitHub Copilot at Production Scale",
      "source_url": "https://arxiv.org/html/2608.00101v1",
      "source_kind": "production_trace_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "A sampled week of GitHub Copilot coding-agent telemetry shows session-structured cache reuse, alternating tool/model phases, costly compaction and retry loops, and a 50x range in per-turn token use across five user archetypes.",
      "quantified": {
        "users": 3200000,
        "sessions": 13500000,
        "turns": 95100000,
        "llm_calls": 760500000,
        "tool_calls": 774700000,
        "prompt_tokens": 44900000000000,
        "completion_tokens": 39300000000,
        "agent_initiated_llm_calls_pct": 87,
        "cached_tokens_session_avg_pct": 90,
        "cached_tokens_turn_boundary_pct": 55,
        "cached_tokens_after_model_switch_pct": 8,
        "sessions_with_compaction_pct": 7.8,
        "tokens_in_compacted_sessions_pct": 44,
        "prompt_tokens_dropped_by_compaction_pct_gt": 70,
        "turns_with_tool_failure_pct": 9,
        "retry_compute_amplification_up_to": 4,
        "user_archetypes": 5,
        "per_turn_tokens_low": 23000,
        "per_turn_tokens_high": 1100000,
        "container_idle_minutes_avg": 4.1,
        "kv_idle_minutes_avg": 2.9
      },
      "assumptions": [
        "Coding-agent scheduling and caching must operate at session/turn scope and preserve heterogeneous user and tool behavior."
      ],
      "contradictions_or_limits": [
        "The sample covers one product and one week; hidden reasoning tokens were excluded, and models/tools evolve rapidly."
      ],
      "fak_implications": [
        "Replay turns, tools, failures, switches, compaction, and idle gaps; report request-weighted and token-weighted outcomes."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "alphabet-q2-2026-capex",
      "category": "hyperscaler",
      "entity": "Alphabet",
      "topic": [
        "capex",
        "filings",
        "capacity",
        "backlog"
      ],
      "published_at": "2026-07-22",
      "event_at": "2026-07-22",
      "source_title": "Alphabet 2026 Q2 Earnings Call",
      "source_url": "https://abc.xyz/investor/events/event-details/2026/2026-Q2-Earnings-Call-2026-GgTAq7Is0z/default.aspx",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Alphabet reported $44.9B Q2 capex, roughly 60% servers and 40% datacenters/networking within technical infrastructure, raised 2026 guidance to $195-205B, and described third-party capacity as a bridge during internal buildout.",
      "quantified": {
        "q2_capex_usd": 44900000000,
        "technical_servers_share_pct_approx": 60,
        "technical_datacenter_network_share_pct_approx": 40,
        "2026_capex_guidance_low_usd": 195000000000,
        "2026_capex_guidance_high_usd": 205000000000,
        "cloud_backlog_usd": 514000000000
      },
      "assumptions": [
        "AI capacity demand is large enough to use both owned and third-party supply while spending is accelerated."
      ],
      "contradictions_or_limits": [
        "Company capex is not a clean AI-only or useful-capacity measure; backlog includes broad GCP and TPU-system agreements."
      ],
      "fak_implications": [
        "Track ownership, asset mix, backlog duration, and delivered/monetized capacity separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "microsoft-fy26-q4-capex",
      "category": "hyperscaler",
      "entity": "Microsoft",
      "topic": [
        "capex",
        "filings",
        "capacity",
        "backlog"
      ],
      "published_at": "2026-07-29",
      "event_at": "2026-07-29",
      "source_title": "Microsoft Fiscal Year 2026 Fourth Quarter Earnings Conference Call",
      "source_url": "https://www.microsoft.com/en-us/investor/events/fy-2026/earnings-fy-2026-q4",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Microsoft reported $41B quarterly capex, about two-thirds in short-lived CPU/GPU assets, $5.6B in finance leases mainly for large datacenter sites, and $678B commercial RPO while Azure demand exceeded capacity.",
      "quantified": {
        "quarter_capex_usd": 41000000000,
        "short_lived_asset_share_pct_approx": 66.7,
        "finance_leases_usd": 5600000000,
        "cash_ppe_usd": 35800000000,
        "commercial_rpo_usd": 678000000000,
        "rpo_weighted_duration_years": 2.3
      },
      "assumptions": [
        "Incoming supply must be allocated among cloud customers, first-party apps, R&D, and replacement."
      ],
      "contradictions_or_limits": [
        "Capex combines AI/non-AI and finance leases create quarter timing differences."
      ],
      "fak_implications": [
        "Normalize cash, lease, asset-life, allocation, and capacity states before comparing providers."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-q2-2026-capex",
      "category": "hyperscaler",
      "entity": "Meta",
      "topic": [
        "capex",
        "filings",
        "capacity",
        "backlog"
      ],
      "published_at": "2026-07-29",
      "event_at": "2026-07-29",
      "source_title": "Meta Reports Second Quarter 2026 Results",
      "source_url": "https://investor.atmeta.com/investor-news/press-release-details/2026/Meta-Reports-Second-Quarter-2026-Results/default.aspx",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta reported $31.08B Q2 capex including finance-lease principal and narrowed full-year 2026 guidance to $130-145B.",
      "quantified": {
        "q2_capex_including_finance_lease_principal_usd": 31080000000,
        "2026_capex_guidance_low_usd": 130000000000,
        "2026_capex_guidance_high_usd": 145000000000,
        "q2_operating_cash_flow_usd": 31860000000,
        "q2_free_cash_flow_usd": 784000000
      },
      "assumptions": [
        "AI and core-business infrastructure investment can consume nearly all quarterly operating cash generation."
      ],
      "contradictions_or_limits": [
        "Meta capex includes finance-lease principal and mixes AI with core business."
      ],
      "fak_implications": [
        "Retain accounting definition and free-cash-flow boundary in cost comparisons."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "amazon-2026-capex-plan",
      "category": "hyperscaler",
      "entity": "Amazon",
      "topic": [
        "capex",
        "filings",
        "capacity",
        "backlog"
      ],
      "published_at": "2026-04-09",
      "event_at": "2026-04-09",
      "source_title": "CEO Andy Jassy 2025 Letter to Shareholders",
      "source_url": "https://www.aboutamazon.com/news/company-news/amazon-ceo-andy-jassy-2025-letter-to-shareholders",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Amazon said it planned roughly $200B of 2026 capex and that a substantial portion of AWS capex was supported by customer commitments and would monetize in 2027-2028.",
      "quantified": {
        "2026_capex_plan_usd_approx": 200000000000,
        "monetization_start_year": 2027,
        "monetization_end_year": 2028
      },
      "assumptions": [
        "Long-lived customer commitments can underwrite capacity before revenue recognition."
      ],
      "contradictions_or_limits": [
        "The total includes non-AWS investments and predates the reported Q2 increase to about $220B."
      ],
      "fak_implications": [
        "Track commitment coverage, customer concentration, delivery, and recognition separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-data-center-demand-response-2025",
      "category": "datacenter_physical",
      "entity": "Google / utility partners",
      "topic": [
        "demand_response",
        "grid",
        "workload_mobility"
      ],
      "published_at": "2025-08-04",
      "event_at": "2025-08-04",
      "source_title": "How we’re making data centers more flexible to benefit power grids",
      "source_url": "https://blog.google/innovation-and-ai/infrastructure-and-cloud/global-network/how-were-making-data-centers-more-flexible-to-benefit-power-grids/",
      "source_kind": "official_engineering_release",
      "evidence_class": "production_observation",
      "confidence": "high",
      "claim": "Google disclosed shifting or reducing machine-learning load during grid events and new utility agreements for demand response. ",
      "quantified": {},
      "assumptions": [
        "Some machine-learning work can be deferred or moved without violating its deadline."
      ],
      "contradictions_or_limits": [
        "The source does not identify workload classes, fraction shifted, SLO effect, or energy economics."
      ],
      "fak_implications": [
        "Classify workloads by mobility/deadline and include grid events in scheduler scenarios."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "semi-analysis-delay-rebuttal-2026",
      "category": "market_signal",
      "entity": "SemiAnalysis",
      "topic": [
        "datacenter_delays",
        "contradiction",
        "methodology"
      ],
      "published_at": "2026-06-18",
      "event_at": "2026-06-18",
      "source_title": "Stop Saying Half of 2026 US Datacenter Capacity Is Delayed",
      "source_url": "https://newsletter.semianalysis.com/p/stop-saying-half-of-2026-us-datacenter",
      "source_kind": "industry_analysis",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "SemiAnalysis disputed the repeated claim that half of U.S. 2026 datacenter capacity was delayed, arguing that project states and denominators had been conflated.",
      "quantified": {},
      "assumptions": [
        "Delay percentages require a named project roster and consistent capacity lifecycle."
      ],
      "contradictions_or_limits": [
        "The analysis may use proprietary tracking and remains secondary evidence."
      ],
      "fak_implications": [
        "Preserve claim and rebuttal; recompute only from project-level dates and state."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-gemini31-flash-lite-2026",
      "category": "frontier_lab",
      "entity": "Google DeepMind",
      "topic": [
        "model_release",
        "serving_profile",
        "long_context",
        "accelerator",
        "batching"
      ],
      "published_at": "2026-03-03",
      "event_at": "2026-03-03",
      "source_title": "Gemini 3.1 Flash-Lite - Model Card",
      "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
      "source_kind": "official_model_card",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Google describes Gemini 3.1 Flash-Lite as a high-volume, latency-sensitive model with up to a 1M-token input context and 64K-token output, trained on TPU pods with JAX and ML Pathways and distributed through Vertex AI, AI Studio, the Gemini API, the Gemini app, and Search AI Overviews.",
      "quantified": {
        "input_context_tokens": 1000000,
        "max_output_tokens": 64000,
        "reported_output_tokens_per_second": 363,
        "input_price_usd_per_million_tokens": 0.25,
        "output_price_usd_per_million_tokens": 1.5
      },
      "assumptions": [
        "Google operates differentiated frontier and throughput-oriented model tiers on a vertically integrated TPU/JAX serving stack.",
        "High-volume product channels create demand for latency and cost optimization distinct from maximum-capability tiers."
      ],
      "contradictions_or_limits": [
        "The model card does not disclose TPU generation, pod count, batch-size distribution, utilization, traffic share, or production concurrency.",
        "Published speed and price values are vendor-reported and workload-specific; a 1M-token maximum does not establish production prevalence."
      ],
      "fak_implications": [
        "Benchmark fast reasoning tiers separately from frontier tiers and include selectable reasoning/output budgets.",
        "Keep TPU product claims separate from fak-native accelerator receipts and measure matched quality-constrained goodput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-llama4-serving-envelope-2025",
      "category": "frontier_lab",
      "entity": "Meta",
      "topic": [
        "model_release",
        "mixture_of_experts",
        "serving_profile",
        "quantization",
        "long_context"
      ],
      "published_at": "2025-04-05",
      "event_at": "2025-04-05",
      "source_title": "The Llama 4 herd: The beginning of a new era of natively multimodal AI innovation",
      "source_url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta released Llama 4 Scout and Maverick as natively multimodal MoE models with 17B active parameters; Scout has 16 experts and was described as fitting on one H100 with Int4 quantization, while Maverick has 128 experts and fits on one H100 host.",
      "quantified": {
        "active_parameters": 17000000000,
        "scout_experts": 16,
        "maverick_experts": 128,
        "scout_serving_h100_gpus_int4": 1
      },
      "assumptions": [
        "MoE active-parameter count, expert routing, quantization, and host boundary materially change serving memory and throughput.",
        "Open-weight models create a self-hosted serving envelope distinct from API-only frontier systems."
      ],
      "contradictions_or_limits": [
        "Fit is not throughput, latency, quality, concurrency, or cost evidence.",
        "The release does not provide production batch distributions, routing skew, expert-load imbalance, or fleet share."
      ],
      "fak_implications": [
        "Record total and active parameters, expert count, quantization, and physical host boundary in model receipts.",
        "Test expert routing and batching under realistic skew rather than treating active parameters as dense-model equivalence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "amazon-nova2-reasoning-context-2025",
      "category": "frontier_lab",
      "entity": "Amazon",
      "topic": [
        "model_release",
        "reasoning_mode",
        "long_context",
        "multimodal",
        "serving_profile"
      ],
      "published_at": "2025-12-02",
      "event_at": "2025-12-02",
      "source_title": "Amazon Nova 2: Multimodal reasoning and generation models",
      "source_url": "https://www.amazon.science/publications/amazon-nova-2-multimodal-reasoning-and-generation-models",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Amazon Nova 2 comprises Lite, Pro, Omni, and Sonic variants; Lite and Pro expose configurable extended-thinking controls, and the family supports multimodal workloads and contexts up to one million tokens.",
      "quantified": {
        "model_variants": 4,
        "max_context_tokens": 1000000
      },
      "assumptions": [
        "Reasoning-mode selection creates a workload mixture whose output length and latency cannot be inferred from the prompt alone.",
        "A model family can route speech, unified multimodal, fast, and higher-capability work to distinct serving paths."
      ],
      "contradictions_or_limits": [
        "The report does not publish production selection shares for each model or thinking level.",
        "Maximum context and model-family breadth do not identify memory occupancy, batchability, or regional traffic."
      ],
      "fak_implications": [
        "Represent reasoning control and modality as scheduler inputs, not static model metadata.",
        "Benchmark context and output tails by selected mode and count dynamic-thinking overhead end to end."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "apple-foundation-models-2025",
      "category": "frontier_lab",
      "entity": "Apple",
      "topic": [
        "model_release",
        "on_device",
        "private_cloud",
        "kv_cache",
        "quantization",
        "mixture_of_experts"
      ],
      "published_at": "2025-07-17",
      "event_at": "2025-06-09",
      "source_title": "Apple Intelligence Foundation Language Models Tech Report 2025",
      "source_url": "https://machinelearning.apple.com/research/apple-foundation-models-tech-report-2025",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Apple describes a roughly 3B-parameter on-device model using KV-cache sharing and 2-bit quantization-aware training, plus a server model using Parallel-Track MoE, track parallelism, sparse computation, and interleaved global-local attention on Private Cloud Compute.",
      "quantified": {
        "on_device_parameters_approx": 3000000000,
        "quantization_bits": 2
      },
      "assumptions": [
        "Frontier product workloads can be split between device-resident and private-cloud execution based on capability, privacy, and resource envelope.",
        "KV-cache sharing and aggressive quantization are product-level design choices for constrained devices, not only datacenter optimizations."
      ],
      "contradictions_or_limits": [
        "Apple does not disclose server parameter count, fleet size, request split, batch policy, or production latency distribution.",
        "Benchmark comparisons in the report are vendor evaluations and do not establish neutral goodput."
      ],
      "fak_implications": [
        "Model routing should represent device, private-cloud, and general-cloud trust/resource boundaries explicitly.",
        "Include KV sharing and low-bit memory accounting in constrained-device experiments without projecting them onto server workloads."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "cohere-command-a-2025",
      "category": "frontier_lab",
      "entity": "Cohere",
      "topic": [
        "model_release",
        "enterprise",
        "long_context",
        "serving_efficiency"
      ],
      "published_at": "2025-03-13",
      "event_at": "2025-03-13",
      "source_title": "Introducing Command A: Max performance, minimal compute",
      "source_url": "https://cohere.com/blog/command-a",
      "source_kind": "official_model_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Cohere positioned Command A as an enterprise agentic model with a 256K-token context and emphasized deployment efficiency relative to competing frontier APIs.",
      "quantified": {
        "context_tokens": 256000
      },
      "assumptions": [
        "Enterprise traffic may emphasize retrieval, tool use, multilingual work, and long documents rather than consumer chat alone.",
        "Private or sovereign deployment constraints can make hardware footprint and serving efficiency first-order selection criteria."
      ],
      "contradictions_or_limits": [
        "Comparative performance and efficiency are vendor claims and require matched independent reproduction.",
        "The release does not disclose production request, tenant, batching, or length distributions."
      ],
      "fak_implications": [
        "Keep sovereign/private deployment as an explicit workload and trust boundary.",
        "Test long-document agentic workloads with retrieval and tool calls instead of using context capacity alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ai2-olmo2-32b-cluster-2025",
      "category": "frontier_lab",
      "entity": "Ai2",
      "topic": [
        "model_release",
        "training_cluster",
        "training_efficiency",
        "open_model"
      ],
      "published_at": "2025-03-13",
      "event_at": "2025-03-13",
      "source_title": "OLMo 2 32B: First fully open model to outperform GPT 3.5 and GPT 4o mini",
      "source_url": "https://allenai.org/blog/olmo2-32b",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Ai2 reports that OLMo 2 32B trained up to 6T tokens on Augusta, a 160-node Google Cloud AI Hypercomputer with eight H100 GPUs per node and GPUDirect-TCPXO, sustaining over 1,800 tokens/s/GPU at about 38% MFU.",
      "quantified": {
        "parameters": 32000000000,
        "training_tokens": 6000000000000,
        "cluster_nodes": 160,
        "gpus_per_node": 8,
        "physical_h100_gpus": 1280,
        "tokens_per_second_per_gpu": 1800,
        "model_flops_utilization": 0.38
      },
      "assumptions": [
        "A 1,280-H100 cluster with disclosed network and utilization is a reproducible mid-scale frontier-training reference point.",
        "Open data, code, checkpoints, and training stages permit stronger accounting than model-only releases."
      ],
      "contradictions_or_limits": [
        "Reported throughput and MFU are training metrics, not inference goodput.",
        "The post does not establish total downtime, net wall-clock cost, regional power, or serving demand."
      ],
      "fak_implications": [
        "Use OLMo as a reference envelope for cluster receipts that include node count, topology, tokens/s/GPU, MFU, and training tokens.",
        "Do not transfer training utilization directly into serving-capacity assumptions."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "bytedance-seed15-vl-2025",
      "category": "frontier_lab",
      "entity": "ByteDance Seed",
      "topic": [
        "model_release",
        "multimodal",
        "mixture_of_experts",
        "agentic"
      ],
      "published_at": "2025-05-13",
      "event_at": "2025-05-13",
      "source_title": "Seed1.5-VL Technical Report",
      "source_url": "https://seed.bytedance.com/en/public_papers/seed1-5-vl-technical-report",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "ByteDance Seed reports Seed1.5-VL with a 532M-parameter vision encoder and a MoE language model using 20B active parameters, targeting multimodal reasoning and agent-centric GUI/game tasks and exposed through Volcano Engine.",
      "quantified": {
        "vision_encoder_parameters": 532000000,
        "active_language_parameters": 20000000000,
        "public_benchmarks_reported": 60,
        "public_benchmarks_claimed_sota": 38
      },
      "assumptions": [
        "Large consumer platforms can combine multimodal perception, reasoning, and GUI action in a single serving workload.",
        "MoE and separate vision encoders create modality-dependent compute and memory paths."
      ],
      "contradictions_or_limits": [
        "The report does not disclose total parameters, physical training cluster, production request mix, or Volcano Engine traffic.",
        "Benchmark leadership is author-reported."
      ],
      "fak_implications": [
        "Represent modality and agent action as workload dimensions and account for separate encoder/backbone costs.",
        "Avoid inferring ByteDance production distributions from benchmark coverage."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "baidu-ernie5-infrastructure-2026",
      "category": "frontier_lab",
      "entity": "Baidu",
      "topic": [
        "model_release",
        "training_infrastructure",
        "mixture_of_experts",
        "multimodal",
        "parallelism"
      ],
      "published_at": "2026-02-04",
      "event_at": "2026-02-04",
      "source_title": "ERNIE 5.0 Technical Report",
      "source_url": "https://arxiv.org/abs/2602.04705",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Baidu describes ERNIE 5.0 training on PaddlePaddle with hybrid parallelism and fine-grained memory control for an ultra-sparse multimodal MoE, FP8 activation storage, adaptive activation offload, and modality tokenizers physically separated from the backbone onto different GPU nodes.",
      "quantified": {
        "precision_bits_for_activation_storage": 8
      },
      "assumptions": [
        "Ultra-sparse MoE training is limited by inter-node communication and memory pressure, not only nominal FLOPS.",
        "Multimodal tokenizers and the backbone can warrant separate placement and parallelization strategies."
      ],
      "contradictions_or_limits": [
        "The report does not disclose cluster size, GPU SKU/count, net training duration, or production serving distribution.",
        "Training topology does not establish inference topology or goodput."
      ],
      "fak_implications": [
        "Model component placement and communication should be first-class in cluster models.",
        "Treat FP8 storage and offload as envelope-specific choices with transfer overhead counted."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tencent-hunyuan-large-2024",
      "category": "frontier_lab",
      "entity": "Tencent",
      "topic": [
        "model_release",
        "mixture_of_experts",
        "long_context",
        "kv_cache",
        "routing"
      ],
      "published_at": "2024-11-04",
      "event_at": "2024-11-04",
      "source_title": "Hunyuan-Large: An Open-Source MoE Model with 52 Billion Activated Parameters by Tencent",
      "source_url": "https://arxiv.org/abs/2411.02265",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Tencent reports Hunyuan-Large as a 389B-total, 52B-active MoE with up to 256K context, mixed expert routing, KV-cache compression, and expert-specific learning rates.",
      "quantified": {
        "total_parameters": 389000000000,
        "active_parameters": 52000000000,
        "context_tokens": 256000
      },
      "assumptions": [
        "MoE routing and KV compression are coupled model-system choices for long-context serving.",
        "Expert popularity and routing balance may create topology and batching pressure not visible in active-parameter counts."
      ],
      "contradictions_or_limits": [
        "The report does not provide production expert-load distributions or deployed fleet share.",
        "Author benchmarks do not establish neutral service economics."
      ],
      "fak_implications": [
        "Capture expert-routing skew and KV compression in serving receipts.",
        "Do not equate 52B active parameters with a dense 52B memory or communication envelope."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "zai-glm45-2025",
      "category": "frontier_lab",
      "entity": "Z.ai / Zhipu AI",
      "topic": [
        "model_release",
        "mixture_of_experts",
        "reasoning_mode",
        "agentic",
        "training_tokens"
      ],
      "published_at": "2025-07-28",
      "event_at": "2025-07-28",
      "source_title": "GLM-4.5: Reasoning, Coding, and Agentic Abilities",
      "source_url": "https://z.ai/blog/glm-4.5",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Z.ai released GLM-4.5 as a 355B-total, 32B-active MoE trained on 23T tokens with hybrid thinking and direct-response modes for agentic, reasoning, and coding workloads.",
      "quantified": {
        "total_parameters": 355000000000,
        "active_parameters": 32000000000,
        "training_tokens": 23000000000000
      },
      "assumptions": [
        "Hybrid reasoning/direct modes produce distinct output-length and latency distributions within one model endpoint.",
        "Agentic training increases tool-oriented and multi-turn workloads."
      ],
      "contradictions_or_limits": [
        "The release does not disclose cluster hardware, production mode-selection rates, or tenant/request distributions.",
        "Benchmark results are author-reported."
      ],
      "fak_implications": [
        "Log requested and realized reasoning modes and stratify goodput by mode.",
        "Treat tool-call expansion and long-horizon sessions as a separate replay population."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tii-falcon-h1-2025",
      "category": "frontier_lab",
      "entity": "Technology Innovation Institute",
      "topic": [
        "model_release",
        "sovereign_ai",
        "hybrid_architecture",
        "long_context",
        "deployment_efficiency"
      ],
      "published_at": "2025-05-21",
      "event_at": "2025-05-21",
      "source_title": "Middle East’s Leading AI Powerhouse TII Launches Two New AI Models: Falcon Arabic and Falcon-H1",
      "source_url": "https://www.tii.ae/news/middle-easts-leading-ai-powerhouse-tii-launches-two-new-ai-models-falcon-arabic-first-arabic",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Abu Dhabi’s TII launched Falcon-H1 and Falcon Arabic, positioning the hybrid model family for efficient deployment from constrained devices through enterprise infrastructure and emphasizing native Arabic data rather than translated corpora.",
      "quantified": {
        "falcon_arabic_base_parameters": 7000000000
      },
      "assumptions": [
        "Sovereign model programs optimize for regional language/data control and deployment flexibility, not only global benchmark rank.",
        "Hybrid attention/state-space architectures target a different long-context memory and throughput envelope than pure Transformers."
      ],
      "contradictions_or_limits": [
        "Performance comparisons in the launch are vendor claims.",
        "The source does not disclose deployed request volume, compute allocation, or training cluster size."
      ],
      "fak_implications": [
        "Include sovereign-data locality and hybrid-state memory behavior in architecture comparisons.",
        "Separate portability claims from measured throughput on each hardware tier."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sarvam-30b-105b-2026",
      "category": "frontier_lab",
      "entity": "Sarvam AI",
      "topic": [
        "model_release",
        "sovereign_ai",
        "training_tokens",
        "agentic",
        "tool_use"
      ],
      "published_at": "2026-03-06",
      "event_at": "2026-03-06",
      "source_title": "Open-Sourcing Sarvam 30B and 105B",
      "source_url": "https://www.sarvam.ai/blogs/sarvam-30b-105b",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "India’s Sarvam open-sourced 30B and 105B MoE models trained entirely in India on IndiaAI Mission compute and on 16T and 12T tokens respectively; the company says both were already in production for conversational and complex agentic products.",
      "quantified": {
        "small_model_parameters": 30000000000,
        "small_model_training_tokens": 16000000000000,
        "large_model_parameters": 105000000000,
        "large_model_training_tokens": 12000000000000,
        "experts_per_model": 128
      },
      "assumptions": [
        "Sovereign-language programs can prioritize multilingual and population-scale workloads while exposing open deployment options.",
        "Tool use and reasoning create serving demand beyond single-turn generation."
      ],
      "contradictions_or_limits": [
        "The release does not disclose physical cluster size, IndiaAI compute allocation, production request volume, or deployment goodput.",
        "Benchmark and efficiency comparisons are vendor-reported; “in production” does not quantify traffic or health."
      ],
      "fak_implications": [
        "Add Indic multilingual and tool-using workloads without inferring population-scale request volume from policy language.",
        "Keep training-token totals separate from serving capacity and adoption."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "naver-hyperclovax-think-2025",
      "category": "frontier_lab",
      "entity": "NAVER Cloud",
      "topic": [
        "model_release",
        "regional_language",
        "reasoning_mode",
        "agentic",
        "open_model"
      ],
      "published_at": "2025-07-21",
      "event_at": "2025-07-21",
      "source_title": "HyperCLOVA X THINK: From seeds to forest",
      "source_url": "https://clova.ai/en/tech-blog/6203-2",
      "source_kind": "official_model_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "NAVER Cloud presented HyperCLOVA X THINK as a Korean-focused reasoning model trained on a 6T-token Korean/English corpus, with a 128K context window, length-control training, RLVR/RLHF stages, and multimodal and agentic/tool-oriented targets.",
      "quantified": {
        "training_tokens": 6000000000000,
        "context_tokens": 128000,
        "reported_korean_stem_accuracy": 0.464
      },
      "assumptions": [
        "Regional-language frontier programs may have different tokenization, data, evaluation, and latency tradeoffs from English-heavy global models.",
        "Reasoning and tool-use demand should be sampled by language and locale."
      ],
      "contradictions_or_limits": [
        "Comparative accuracy and training-efficiency claims are vendor-reported and do not provide serving cost, cluster size, or production demand.",
        "A regional benchmark score and training-token total do not identify national adoption or workload distribution."
      ],
      "fak_implications": [
        "Include Korean-language tokenization and agentic workloads in multilingual evaluations.",
        "Do not treat English-token counts as comparable cost units across languages without measuring tokenizer expansion."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sea-ai-lab-sailor2-2024",
      "category": "frontier_lab",
      "entity": "Sea AI Lab",
      "topic": [
        "model_release",
        "regional_language",
        "production_demand",
        "speculative_decoding",
        "training_tokens"
      ],
      "published_at": "2024-12-03",
      "event_at": "2024-12-03",
      "source_title": "Sailor2: Sailing in South-East Asia with Inclusive Multilingual LLMs",
      "source_url": "https://sail.sea.com/blog/articles/55",
      "source_kind": "official_research",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Sea AI Lab’s Sailor2 release says its production demand favored 8B and 20B models, with a 1B model for specialized uses including speculative decoding; the family continued pretraining on about 500B tokens across 15 languages.",
      "quantified": {
        "model_sizes_parameters": [
          1000000000,
          8000000000,
          20000000000
        ],
        "continued_pretraining_tokens_approx": 500000000000,
        "supported_languages": 15
      },
      "assumptions": [
        "Regional production users may prefer mid-sized deployment-efficient models over the largest available model.",
        "A small draft model can be a first-class speculative-decoding component rather than a standalone quality tier."
      ],
      "contradictions_or_limits": [
        "The source states strong production demand qualitatively but does not publish request counts, tenant shares, acceptance rates, or hardware.",
        "The 500B-token figure is approximate and benchmark comparisons are author-reported."
      ],
      "fak_implications": [
        "Add Southeast Asian language mixes and 1B/8B/20B deployment tiers to replay design.",
        "Measure speculative acceptance and end-to-end goodput instead of assuming the draft model always helps."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ai21-jamba15-2024",
      "category": "frontier_lab",
      "entity": "AI21 Labs",
      "topic": [
        "model_release",
        "hybrid_architecture",
        "mixture_of_experts",
        "long_context",
        "quantization"
      ],
      "published_at": "2024-08-22",
      "event_at": "2024-08-22",
      "source_title": "The Jamba 1.5 Open Model Family: The Most Powerful and Efficient Long Context Models",
      "source_url": "https://www.ai21.com/blog/announcing-jamba-model-family/",
      "source_kind": "official_model_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Israel-based AI21 released Jamba 1.5 Mini and Large using a hybrid SSM-Transformer MoE architecture with a 256K effective context; it reported that Mini could handle up to 140K tokens on one GPU and that ExpertsInt8 let Large fit its full 256K context on one eight-GPU node.",
      "quantified": {
        "effective_context_tokens": 256000,
        "mini_single_gpu_context_tokens": 140000,
        "large_full_context_gpu_count": 8,
        "long_context_speedup_claim": 2.5
      },
      "assumptions": [
        "Hybrid state-space/attention architectures target long-context memory and throughput limits differently from pure Transformers.",
        "Quantization, context length, and node boundary jointly determine deployability."
      ],
      "contradictions_or_limits": [
        "Speed and quality comparisons are vendor claims; the cited benchmark used batch size 1 and fixed 512-token outputs on A100 hardware.",
        "Fit at one context length does not establish multi-tenant throughput, cache behavior, or production traffic."
      ],
      "fak_implications": [
        "Benchmark hybrid-state models with explicit state/KV memory, context, batch, and output length.",
        "Keep single-request fit claims separate from SLO-constrained service goodput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sakana-series-b-efficient-ai-2025",
      "category": "frontier_lab",
      "entity": "Sakana AI",
      "topic": [
        "regional_lab",
        "efficient_models",
        "edge_deployment",
        "automated_research",
        "enterprise_deployment"
      ],
      "published_at": "2025-11-10",
      "event_at": "2025-11-10",
      "source_title": "Announcing Our Series B",
      "source_url": "https://sakana.ai/series-b/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Tokyo-based Sakana AI said its strategy emphasizes combining existing models, automated AI science, energy-efficient language models that run on edge devices, and enterprise deployments in Japan rather than training another monolithic foundation model from scratch.",
      "quantified": {},
      "assumptions": [
        "A regional frontier lab may advance capability through model merging, search, agents, and edge efficiency rather than only larger pretraining clusters.",
        "Enterprise deployment evidence can exist without public model-traffic or cluster disclosures."
      ],
      "contradictions_or_limits": [
        "The financing announcement provides no model size, cluster size, production traffic, energy measurement, or independent deployment outcomes.",
        "“Energy efficient” and enterprise growth are official claims, not neutral measurements."
      ],
      "fak_implications": [
        "Track composition/search/automated-research labs separately from scale-first pretraining labs.",
        "Require measured device and end-to-end energy envelopes before borrowing efficiency claims."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "coreweave-q1-2026-financials",
      "category": "ai_cloud",
      "entity": "CoreWeave",
      "topic": [
        "revenue",
        "backlog",
        "capacity",
        "debt",
        "customer_contracts"
      ],
      "published_at": "2026-05-07",
      "event_at": "2026-03-31",
      "source_title": "CoreWeave Reports Strong First Quarter 2026 Results",
      "source_url": "https://investors.coreweave.com/news/news-details/2026/CoreWeave-Reports-Strong-First-Quarter-2026-Results/",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "CoreWeave reported Q1 2026 revenue of $2.078B, $99.4B of revenue backlog, more than 1 GW of active power, and more than 3.5 GW of contracted power; its backlog definition includes RPO plus estimated future revenue under committed contracts subject to delivery and service availability.",
      "quantified": {
        "quarter_revenue_usd": 2078000000,
        "revenue_backlog_usd": 99400000000,
        "active_power_gw_gt": 1,
        "contracted_power_gw_gt": 3.5,
        "new_meta_commitment_usd": 21000000000,
        "nonrecourse_delayed_draw_term_loan_usd": 8500000000
      },
      "assumptions": [
        "AI-cloud economics depend on financing and delivering capacity before backlog becomes revenue.",
        "Active power, contracted power, and backlog are separate leading indicators with different delivery risk."
      ],
      "contradictions_or_limits": [
        "Revenue backlog is broader than GAAP RPO and is subject to delivery and availability requirements.",
        "Revenue, backlog, active MW, installed accelerators, and useful goodput are different denominators."
      ],
      "fak_implications": [
        "Do not translate contract value directly into active serving capacity.",
        "Track financing, active power, delivery conditions, revenue recognition, and goodput separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "oracle-fy2026-rpo-ai-financing",
      "category": "hyperscaler",
      "entity": "Oracle",
      "topic": [
        "rpo",
        "ai_contracts",
        "customer_prepayment",
        "customer_supplied_hardware",
        "capex",
        "financing"
      ],
      "published_at": "2026-06-10",
      "event_at": "2026-05-31",
      "source_title": "Oracle Announces Record Q4 and FY 2026 Results Driven by Cloud Infrastructure & Cloud Applications",
      "source_url": "https://investor.oracle.com/investor-news/news-details/2026/Oracle-Announces-Record-Q4-and-FY-2026-Results-Driven-by-Cloud-Infrastructure--Cloud-Applications/default.aspx",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Oracle ended FY2026 with $638B of RPO and said large AI contracts included $75B of prepaid or customer-supplied hardware; FY2026 free cash flow was negative $23.7B while it invested in cloud infrastructure.",
      "quantified": {
        "rpo_usd": 638000000000,
        "rpo_yoy_growth_pct": 363,
        "prepaid_or_customer_supplied_hardware_usd": 75000000000,
        "fy2026_cloud_infrastructure_revenue_usd": 18100000000,
        "fy2026_free_cash_flow_usd": -23700000000,
        "fy2026_debt_financing_usd": 43000000000,
        "fy2026_equity_financing_usd": 5000000000
      },
      "assumptions": [
        "Customer prepayments and customer-supplied GPUs can shift capital requirements without removing delivery obligations."
      ],
      "contradictions_or_limits": [
        "RPO includes future contractual revenue, not installed or accepted capacity.",
        "Customer-supplied hardware complicates comparisons of Oracle capex with capacity controlled or served."
      ],
      "fak_implications": [
        "Model customer-owned hardware and prepayments as separate accounting/ownership states.",
        "Do not divide RPO by GPU price to infer fleet size."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "alibaba-q2-2026-ai-capex",
      "category": "hyperscaler",
      "entity": "Alibaba",
      "topic": [
        "capex",
        "ai_cloud",
        "revenue",
        "cash"
      ],
      "published_at": "2026-08-20",
      "event_at": "2026-06-30",
      "source_title": "Alibaba’s Full-Stack AI Accelerates Monetization with 22-Quarter-High Cloud Growth",
      "source_url": "https://www.alibabagroup.com/en-US/document-2027233133950140416",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Alibaba said it invested nearly $10B of capex in the quarter, up 75% year over year, and reported $1.8B of AI-related product revenue while retaining $69.9B in cash and liquid investments.",
      "quantified": {
        "quarter_capex_usd_approx": 10000000000,
        "capex_yoy_growth_pct": 75,
        "quarter_ai_related_product_revenue_usd": 1800000000,
        "cash_and_liquid_investments_usd": 69900000000
      },
      "assumptions": [
        "Alibaba is funding AI across multiple stack layers and business segments rather than a pure AI-cloud boundary."
      ],
      "contradictions_or_limits": [
        "The capex figure is described as AI-related investment but is not split by accelerators, datacenters, network, or other assets.",
        "AI-related product revenue and capex are not directly comparable capacity measures."
      ],
      "fak_implications": [
        "Keep currency, segment, quarter, and asset split explicit for China-cloud comparisons.",
        "Do not infer serving hardware from aggregate capex."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nebius-microsoft-contract-financing-2025",
      "category": "ai_cloud",
      "entity": "Nebius Group",
      "topic": [
        "customer_contract",
        "capex",
        "secured_debt",
        "capacity",
        "customer_concentration"
      ],
      "published_at": "2025-09-08",
      "event_at": "2025-09-08",
      "source_title": "Nebius announces multi-billion dollar agreement with Microsoft for AI infrastructure",
      "source_url": "https://nebius.com/newsroom/nebius-announces-multi-billion-dollar-agreement-with-microsoft-for-ai-infrastructure",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Nebius announced a multi-year dedicated-capacity contract with Microsoft at its Vineland, New Jersey datacenter and expected to finance associated capex through contract cash flow and debt secured against the contract.",
      "quantified": {},
      "assumptions": [
        "A high-credit counterparty contract can improve financing terms for purpose-built capacity."
      ],
      "contradictions_or_limits": [
        "The release does not state contract value, MW, accelerator count, debt amount, utilization, or revenue-recognition schedule.",
        "A signed contract is not delivered, accepted, or diversified revenue."
      ],
      "fak_implications": [
        "Track contract-backed financing and customer concentration alongside capacity lifecycle.",
        "Keep dedicated single-customer capacity distinct from fungible cloud supply."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "baidu-q2-2026-ai-cloud-revenue",
      "category": "hyperscaler",
      "entity": "Baidu",
      "topic": [
        "ai_cloud",
        "gpu_cloud",
        "revenue",
        "segment_boundary"
      ],
      "published_at": "2026-08-18",
      "event_at": "2026-06-30",
      "source_title": "Baidu Announces Second Quarter 2026 Results",
      "source_url": "https://ir.baidu.com/node/14731/pdf",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Baidu reported Q2 2026 AI Cloud Infra revenue of RMB7.3B, up 50% year over year, and GPU Cloud revenue growth of 283%; it notes the AI-powered business figures come from unaudited internal management records.",
      "quantified": {
        "quarter_ai_cloud_infra_revenue_rmb": 7300000000,
        "ai_cloud_infra_yoy_growth_pct": 50,
        "gpu_cloud_revenue_yoy_growth_pct": 283,
        "core_ai_powered_business_revenue_rmb": 12500000000
      },
      "assumptions": [
        "GPU-cloud revenue growth is demand evidence, but not a physical capacity or utilization disclosure."
      ],
      "contradictions_or_limits": [
        "The AI-powered business data is unaudited management information and GPU Cloud revenue is not reported as an absolute amount.",
        "Revenue does not disclose accelerator mix, capex, customer concentration, or useful goodput."
      ],
      "fak_implications": [
        "Use China-cloud revenue as a market signal only; retain audit and segment-boundary labels.",
        "Do not convert percentage growth into installed capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "vertiv-q4-2025-backlog-2026",
      "category": "supply_chain",
      "entity": "Vertiv",
      "topic": [
        "cooling",
        "power_equipment",
        "backlog",
        "orders",
        "delivery_risk"
      ],
      "published_at": "2026-02-11",
      "event_at": "2025-12-31",
      "source_title": "Vertiv Reports Strong Fourth Quarter with Organic Orders Growth of 252%",
      "source_url": "https://investors.vertiv.com/news/news-details/2026/Vertiv-Reports-Strong-Fourth-Quarter-with-Organic-Orders-Growth-of-252-and-Diluted-EPS-Growth-of-200-Adjusted-Diluted-EPS-37/default.aspx",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Vertiv reported Q4 2025 organic orders up about 252%, a roughly 2.9x book-to-bill ratio, and $15.0B backlog, with hyperscale and colocation datacenters leading demand across power and thermal infrastructure.",
      "quantified": {
        "quarter_organic_orders_yoy_pct_approx": 252,
        "book_to_bill_approx": 2.9,
        "backlog_usd": 15000000000,
        "backlog_yoy_growth_pct": 109
      },
      "assumptions": [
        "Power and cooling equipment delivery can lag compute procurement and gate datacenter commissioning."
      ],
      "contradictions_or_limits": [
        "Backlog is contracted demand, not shipped, installed, commissioned, or healthy equipment.",
        "The disclosure does not split backlog into cooling, UPS, switchgear, regions, projects, or delivery dates."
      ],
      "fak_implications": [
        "Track critical-infrastructure backlog and delivery state separately from accelerator availability.",
        "Do not convert supplier backlog into online MW without project-level shipment and acceptance evidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "skhynix-hbm4-mass-shipments-2026",
      "category": "supply_chain",
      "entity": "SK hynix",
      "topic": [
        "hbm",
        "mass_production",
        "shipments",
        "customer_qualification",
        "memory_supply"
      ],
      "published_at": "2026-07-29",
      "event_at": "2026-06-30",
      "source_title": "SK hynix Announces 2Q26 Financial Results",
      "source_url": "https://news.skhynix.com/en/q2-2026-business-results/",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "SK hynix said HBM4 mass shipments began in Q2 2026 while it expanded long-term customer contracts and production capacity.",
      "quantified": {},
      "assumptions": [
        "HBM lifecycle must distinguish samples, qualification, mass-production readiness, mass shipments, accelerator integration, and accepted systems."
      ],
      "contradictions_or_limits": [
        "Mass shipments do not reveal unit volume, yield, customer allocation, accelerator install rate, or usable fleet capacity.",
        "Supplier statements are not a neutral industry-wide HBM supply census."
      ],
      "fak_implications": [
        "Record HBM generation and shipment lifecycle on accelerator receipts.",
        "Do not infer installed accelerator capacity from memory shipment status alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tsmc-cowos-roadmap-2026",
      "category": "supply_chain",
      "entity": "TSMC",
      "topic": [
        "advanced_packaging",
        "cowos",
        "hbm",
        "roadmap",
        "production_lifecycle"
      ],
      "published_at": "2026-04-23",
      "event_at": "2026-04-23",
      "source_title": "TSMC Debuts A13 Technology at 2026 North America Technology Symposium",
      "source_url": "https://pr.tsmc.com/english/news/3302",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "TSMC said it was producing 5.5-reticle-size CoWoS and planned a 14-reticle version integrating about 10 large compute dies and 20 HBM stacks for production in 2028, followed by beyond-14-reticle expansion in 2029.",
      "quantified": {
        "current_cowos_reticle_size": 5.5,
        "planned_2028_reticle_size": 14,
        "planned_compute_dies_2028_approx": 10,
        "planned_hbm_stacks_2028_approx": 20,
        "planned_production_year": 2028
      },
      "assumptions": [
        "Advanced packaging size and HBM integration are accelerator-platform constraints with multi-year production roadmaps."
      ],
      "contradictions_or_limits": [
        "A roadmap is not delivered capacity, yield, customer allocation, or package shipment volume.",
        "Reticle size does not identify wafer-equivalent throughput or accepted accelerator systems."
      ],
      "fak_implications": [
        "Track package generation, qualification, and volume-production date separately from GPU announcements.",
        "Avoid forecasting fleet capacity from package geometry without throughput and yield."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "gevernova-crusoe-29-turbines-2025",
      "category": "datacenter_physical",
      "entity": "GE Vernova / Crusoe",
      "topic": [
        "onsite_generation",
        "gas_turbines",
        "power",
        "order",
        "emissions"
      ],
      "published_at": "2025-07-22",
      "event_at": "2025-06-30",
      "source_title": "GE Vernova and Crusoe announce major 29-unit aeroderivative gas turbine deal to deliver power to AI data centers",
      "source_url": "https://www.gevernova.com/news/press-releases/ge-vernova-crusoe-announce-major-29-unit-aeroderivative-gas-turbine-deliver-ai-data-centers",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "GE Vernova and Crusoe announced orders totaling 29 LM2500XPRESS turbine packages expected to provide nearly 1 GW of electricity for AI datacenters; 19 units were booked in June 2025 after 10 ordered in December 2024.",
      "quantified": {
        "turbine_packages": 29,
        "expected_generation_gw_approx": 1,
        "units_booked_june_2025": 19,
        "units_ordered_december_2024": 10
      },
      "assumptions": [
        "Onsite generation can bypass grid timing but adds fuel, emissions, permitting, equipment delivery, and operations constraints."
      ],
      "contradictions_or_limits": [
        "An equipment order and nameplate generation are not commissioned continuous IT load.",
        "The source does not state site, delivery dates, capacity factor, PUE, gas interconnection, water use, or accepted compute capacity."
      ],
      "fak_implications": [
        "Maintain ordered, delivered, commissioned, available, and IT-load-equivalent power states.",
        "Count fuel, emissions control, redundancy, PUE, and outages in capacity receipts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "siemens-start-campus-sines-lifecycle-2026",
      "category": "datacenter_physical",
      "entity": "Siemens Energy / Start Campus",
      "topic": [
        "grid_connection",
        "transformers",
        "switchgear",
        "seawater_cooling",
        "phased_delivery",
        "site_lifecycle"
      ],
      "published_at": "2026-06-01",
      "event_at": "2026-06-01",
      "source_title": "Start Campus, Portugal: Renewable energy for hyperscale data center",
      "source_url": "https://www.siemens-energy.com/global/en/home/references/start-campus-sines-datacenter.html",
      "source_kind": "official_case_study",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Siemens Energy describes a phased Sines, Portugal campus delivery: Phase 1 supplied Blue gas-insulated switchgear, transformers, and electrical infrastructure for early operations; future phases add substations, capacity, and seawater cooling.",
      "quantified": {},
      "assumptions": [
        "Named equipment and phased site records are stronger lifecycle evidence than campus target capacity alone.",
        "Cooling and grid infrastructure can enter service in different phases."
      ],
      "contradictions_or_limits": [
        "The case study does not provide shipment/commissioning dates, accepted MW, PUE, water flow, equipment lead times, or independent operating evidence.",
        "Future phases remain plans."
      ],
      "fak_implications": [
        "Track each site phase and equipment system through delivery and acceptance.",
        "Do not treat planned seawater cooling or substations as current operating capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "skhynix-hbm4e-samples-2026",
      "category": "supply_chain",
      "entity": "SK hynix",
      "topic": [
        "hbm4e",
        "samples",
        "qualification",
        "bandwidth",
        "power_efficiency"
      ],
      "published_at": "2026-06-18",
      "event_at": "2026-06-18",
      "source_title": "SK hynix Ships Samples of 12-Layer Next-Gen HBM4E",
      "source_url": "https://news.skhynix.com/en/sk-hynix-ships-samples-of-12-layer-next-gen-hbm4e-2/",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "SK hynix shipped 12-layer HBM4E samples to major customers, reporting up to 16 Gbps per pin, more than 20% better power efficiency, and 17% lower heat resistance, with future mass production still pending.",
      "quantified": {
        "layers": 12,
        "max_gbps_per_pin": 16,
        "power_efficiency_improvement_pct_gt": 20,
        "heat_resistance_reduction_pct": 17
      },
      "assumptions": [
        "Sample shipment is an early qualification state, not supply available for production fleets."
      ],
      "contradictions_or_limits": [
        "Performance is vendor-reported and customer qualification outcomes, yield, volume, allocation, and mass-production timing are undisclosed."
      ],
      "fak_implications": [
        "Represent sample, qualification, and mass-production states explicitly.",
        "Do not benchmark future HBM performance as installed hardware."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "replicate-cloudflare-acquisition-2025",
      "category": "market_signal",
      "entity": "Cloudflare / Replicate",
      "topic": [
        "acquisition",
        "inference_platform",
        "model_hosting",
        "edge_inference"
      ],
      "published_at": "2025-11-17",
      "event_at": "2025-11-17",
      "source_title": "Replicate is joining Cloudflare",
      "source_url": "https://replicate.com/blog/replicate-cloudflare",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Replicate announced it was joining Cloudflare while remaining a distinct brand and keeping its API and existing models running, with planned integration into Cloudflare’s developer platform and network.",
      "quantified": {},
      "assumptions": [
        "Model-hosting and inference abstractions may consolidate into global developer/network platforms."
      ],
      "contradictions_or_limits": [
        "The announcement does not disclose transaction value, profitability, traffic, GPU fleet, customer concentration, or integration milestones.",
        "Product-continuity promises are plans until later service evidence confirms them."
      ],
      "fak_implications": [
        "Track acquisition outcome separately from announced API continuity and later integration.",
        "Expect routing/hosting layers to bundle network, state, storage, and inference."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-schedmd-acquisition-2025",
      "category": "market_signal",
      "entity": "NVIDIA / SchedMD",
      "topic": [
        "acquisition",
        "scheduler",
        "slurm",
        "open_source",
        "cluster_management"
      ],
      "published_at": "2025-12-15",
      "event_at": "2025-12-15",
      "source_title": "NVIDIA Acquires Open-Source Workload Management Provider SchedMD",
      "source_url": "https://blogs.nvidia.com/blog/nvidia-acquires-schedmd/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "NVIDIA acquired SchedMD, the developer of Slurm, and said Slurm would continue as open-source, vendor-neutral software for HPC and AI workload management.",
      "quantified": {},
      "assumptions": [
        "Cluster scheduling and policy are strategic infrastructure control points, not commodity glue."
      ],
      "contradictions_or_limits": [
        "The transaction value and future governance/control mechanisms are undisclosed.",
        "A vendor-neutral commitment is an official promise whose implementation must be observed over time."
      ],
      "fak_implications": [
        "Track scheduler governance, interoperability, and release behavior after acquisition.",
        "Preserve hardware-neutral scheduling interfaces in fak-native paths."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-groq-license-2025",
      "category": "market_signal",
      "entity": "NVIDIA / Groq",
      "topic": [
        "licensing",
        "talent_transfer",
        "inference_accelerator",
        "startup_lifecycle"
      ],
      "published_at": "2026-02-25",
      "event_at": "2025-12-31",
      "source_title": "NVIDIA Announces Financial Results for Fourth Quarter and Fiscal 2026",
      "source_url": "https://nvidianews.nvidia.com/news/nvidia-announces-financial-results-for-fourth-quarter-and-fiscal-2026",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "NVIDIA disclosed a non-exclusive licensing agreement with Groq to accelerate AI inference at global scale; public reporting separately described key Groq engineering talent joining NVIDIA while Groq remained operational.",
      "quantified": {},
      "assumptions": [
        "Infrastructure consolidation can occur through IP licensing and talent transfer without a full-company acquisition."
      ],
      "contradictions_or_limits": [
        "The official disclosure does not state price, licensed IP scope, staff count, product roadmap, or effect on Groq Cloud.",
        "Reported talent-transfer details are not independently quantified in the official filing."
      ],
      "fak_implications": [
        "Represent licensing, hiring, acquisition, and continued independent operation as separate lifecycle events.",
        "Recheck product support and roadmap rather than labeling the company simply acquired."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "untether-ai-shutdown-amd-2025",
      "category": "market_signal",
      "entity": "Untether AI / AMD",
      "topic": [
        "shutdown",
        "acquihire",
        "product_discontinuation",
        "accelerator_startup"
      ],
      "published_at": "2025-06-05",
      "event_at": "2025-06-05",
      "source_title": "Untether AI Team Acquired by AMD, Product Support Discontinued",
      "source_url": "https://www.hpcwire.com/off-the-wire/untether-ai-team-acquired-by-amd-product-support-discontinued/",
      "source_kind": "industry_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Untether AI shut down, its engineering team joined AMD, and support for speedAI products and the imAIgine SDK ended.",
      "quantified": {},
      "assumptions": [
        "Large funding and working silicon do not remove commercialization, ecosystem, and capital-continuity risk for accelerator startups."
      ],
      "contradictions_or_limits": [
        "The report does not establish the full asset/IP disposition or transaction terms.",
        "An engineering-team transfer is not a product acquisition or continuity guarantee."
      ],
      "fak_implications": [
        "Include product-support termination and migration cost in accelerator-platform risk.",
        "Track team, IP, product, support, and legal-entity outcomes separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "untether-ai-bankruptcy-2025",
      "category": "market_signal",
      "entity": "Untether AI",
      "topic": [
        "bankruptcy",
        "failure",
        "liabilities",
        "funding_risk",
        "accelerator_startup"
      ],
      "published_at": "2025-10-27",
      "event_at": "2025-10-14",
      "source_title": "Untether AI files for bankruptcy following AMD acquihire",
      "source_url": "https://betakit.com/untether-ai-files-for-bankruptcy-following-amd-acquihire/",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Bankruptcy filings reported after Untether’s shutdown showed nearly $25M in assets and just over $128M in unsecured liabilities, with lack of working capital or funding cited as the cause.",
      "quantified": {
        "assets_usd_approx": 25000000,
        "unsecured_liabilities_usd_gt": 128000000,
        "deficiency_usd": 103600000
      },
      "assumptions": [
        "Capital intensity and funding continuity can terminate an accelerator product despite prior fundraising and engineering progress."
      ],
      "contradictions_or_limits": [
        "Figures are from reported bankruptcy documents rather than an issuer earnings statement.",
        "The filing does not quantify lifetime product revenue, deployed chips, customer migration cost, or AMD transaction consideration."
      ],
      "fak_implications": [
        "Track runway, liabilities, support continuity, and customer exit plans in alternative-hardware adoption.",
        "Funding raised is not evidence of durable operations."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "bis-ai-diffusion-rescission-2025",
      "category": "policy_regulation",
      "entity": "U.S. Bureau of Industry and Security",
      "topic": [
        "export_controls",
        "ai_chips",
        "china",
        "rule_lifecycle",
        "supply_chain"
      ],
      "published_at": "2025-05-13",
      "event_at": "2025-05-13",
      "source_title": "Department of Commerce Announces Rescission of Biden-Era Artificial Intelligence Diffusion Rule, Strengthens Chip-Related Export Controls",
      "source_url": "https://www.bis.gov/press-release/department-commerce-announces-rescission-biden-era-artificial-intelligence-diffusion-rule-strengthens",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "BIS announced it would rescind the January 15, 2025 AI Diffusion Rule before its May 15 compliance date, instructed enforcement officials not to enforce it, and issued separate guidance on PRC advanced-computing ICs, Chinese-model training/inference, and diversion risk.",
      "quantified": {
        "original_rule_issued_at": "2025-01-15",
        "planned_compliance_at": "2025-05-15"
      },
      "assumptions": [
        "Export-control state can change between publication and compliance, so architecture policy must bind to effective and enforcement dates, not headlines."
      ],
      "contradictions_or_limits": [
        "The announcement said a formal rescission and replacement rule would follow; it is not the full replacement-control text.",
        "Guidance and enforcement posture are not identical to statutory or regulatory classification for a specific transaction."
      ],
      "fak_implications": [
        "Version regional hardware eligibility by jurisdiction, effective date, end user, end use, and enforcement state.",
        "Never encode a withdrawn proposal as an active capacity constraint."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "eu-ai-gigafactories-call-2026",
      "category": "policy_regulation",
      "entity": "European Commission / EuroHPC",
      "topic": [
        "sovereign_compute",
        "ai_gigafactories",
        "public_funding",
        "private_investment",
        "procurement"
      ],
      "published_at": "2026-07-30",
      "event_at": "2026-07-30",
      "source_title": "EU launches AI Gigafactories call to boost Europe’s computing capacity and unlock more than €30 billion in investment",
      "source_url": "https://digital-strategy.ec.europa.eu/en/news/eu-launches-ai-gigafactories-call-boost-europes-computing-capacity-and-unlock-more-eu30-billion",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "The EU launched a tender call for up to seven AI Gigafactories, supported by up to €10B in EU/national funding and intended to unlock at least €20B in private investment for training, inference, and fine-tuning infrastructure.",
      "quantified": {
        "gigafactories_up_to": 7,
        "public_funding_eur_up_to": 10000000000,
        "private_investment_eur_expected_at_least": 20000000000,
        "existing_ai_factories": 19
      },
      "assumptions": [
        "Sovereign compute programs combine procurement, public funding, private capital, access policy, and regional regulatory constraints."
      ],
      "contradictions_or_limits": [
        "A call for tenders and expected investment are not selected sites, financed projects, installed accelerators, or useful capacity.",
        "Processor mix, MW, construction schedule, access allocation, and delivered goodput remain unspecified."
      ],
      "fak_implications": [
        "Track tender, award, financing, construction, acceptance, access, and workload states separately.",
        "Treat sovereign access and EU data/safety obligations as scheduler and deployment constraints only when applicable."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "indiaai-18000-gpu-capacity-2025",
      "category": "policy_regulation",
      "entity": "IndiaAI Mission",
      "topic": [
        "sovereign_compute",
        "gpu_marketplace",
        "public_private_partnership",
        "subsidized_access",
        "allocation"
      ],
      "published_at": "2025-01-31",
      "event_at": "2025-01-31",
      "source_title": "Union Minister announces the availability of 18,000+ affordable AI compute units",
      "source_url": "https://indiaai.gov.in/article/union-minister-of-electronics-it-railways-and-i-b-announces-the-availability-of-18-000-affordable-ai-compute-units",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "IndiaAI announced more than 18,000 GPU compute units through empaneled cloud providers for startups, researchers, students, government, and other approved users, alongside indigenous-model and safety-institute initiatives.",
      "quantified": {
        "gpu_compute_units_gt": 18000,
        "mission_pillars": 7
      },
      "assumptions": [
        "Sovereign compute can be delivered through a subsidized multi-provider marketplace rather than one government-owned cluster."
      ],
      "contradictions_or_limits": [
        "Empaneled/available units are not necessarily simultaneously installed, allocated, healthy, or used.",
        "The headline total mixes GPU types/providers and does not provide time-varying supply, utilization, or useful-goodput conversion."
      ],
      "fak_implications": [
        "Record provider, GPU SKU, allocation, subsidy, project window, and start-by date per sovereign allocation.",
        "Do not treat a heterogeneous marketplace unit count as one matched cluster."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "singapore-green-dc-roadmap-2024",
      "category": "policy_regulation",
      "entity": "Singapore IMDA",
      "topic": [
        "datacenter_regulation",
        "power_capacity",
        "green_energy",
        "energy_efficiency",
        "regional_supply"
      ],
      "published_at": "2024-05-30",
      "event_at": "2024-05-30",
      "source_title": "Singapore announces Green Data Centre Roadmap for sustainable growth",
      "source_url": "https://www.imda.gov.sg/resources/press-releases-factsheets-and-speeches/press-releases/2024/sg-announces-green-data-centre-roadmap",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Singapore’s Green Data Centre Roadmap targeted at least 300 MW of additional near-term capacity, with potentially another 200 MW or more through green-energy deployments, tied to energy-efficiency and low-carbon-energy pathways.",
      "quantified": {
        "additional_capacity_mw_at_least": 300,
        "potential_green_energy_capacity_mw_at_least": 200
      },
      "assumptions": [
        "Datacenter capacity in constrained regions can be allocated through policy calls conditioned on energy efficiency and energy source."
      ],
      "contradictions_or_limits": [
        "Roadmap capacity is not awarded, constructed, energized, commissioned, or AI-specific IT load.",
        "Potential green-energy capacity depends on later projects and qualifying pathways."
      ],
      "fak_implications": [
        "Include jurisdictional capacity allocation, energy-efficiency conditions, and energy-source eligibility in site admission.",
        "Keep roadmap MW separate from operating facility and IT load."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "eu-ai-act-application-2026",
      "category": "policy_regulation",
      "entity": "European Union / AI Office",
      "topic": [
        "ai_regulation",
        "gpai",
        "enforcement",
        "transparency",
        "deployment_obligations"
      ],
      "published_at": "2026-07-31",
      "event_at": "2026-08-02",
      "source_title": "AI Act regulatory framework and application timeline",
      "source_url": "https://digital-strategy.ec.europa.eu/en/policies/regulatory-framework-ai",
      "source_kind": "official_policy_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "The EU AI Act became generally applicable on August 2, 2026; GPAI obligations had applied since August 2, 2025, and the AI Office and national authorities began implementation, supervision, and enforcement from August 2, 2026.",
      "quantified": {
        "gpai_obligations_applicable_at": "2025-08-02",
        "general_application_at": "2026-08-02",
        "high_risk_annex_i_transition_at": "2028-08-02"
      },
      "assumptions": [
        "Model serving in the EU can carry provider/deployer documentation, transparency, safety, incident, and governance obligations independent of hardware capacity."
      ],
      "contradictions_or_limits": [
        "Applicability depends on actor role, model/system classification, jurisdiction, and later amendments/guidance.",
        "This ledger is operational research, not legal advice or a substitute for the final regulation and counsel."
      ],
      "fak_implications": [
        "Represent jurisdiction, provider/deployer role, model class, documentation, and audit obligations in deployment policy.",
        "Version compliance logic by effective date and preserve the exact legal source."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "uec-spec-103-2026",
      "category": "standard",
      "entity": "Ultra Ethernet Consortium",
      "topic": [
        "network_standard",
        "ai_fabric",
        "interoperability",
        "congestion_control",
        "specification_lifecycle"
      ],
      "published_at": "2026-07-16",
      "event_at": "2026-07-16",
      "source_title": "Ultra Ethernet Specification history",
      "source_url": "https://ultraethernet.org/specification-history/",
      "source_kind": "official_standard",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "The Ultra Ethernet Consortium lists Specification 1.0.3, published July 16, 2026, as the current public Ultra Ethernet version after the initial 1.0 release on June 11, 2025 and corrective 1.0.1/1.0.2 updates.",
      "quantified": {
        "current_version": "1.0.3",
        "initial_public_release_at": "2025-06-11",
        "current_release_at": "2026-07-16"
      },
      "assumptions": [
        "AI-fabric standards and correction releases can change congestion-control and interoperability behavior across cluster generations."
      ],
      "contradictions_or_limits": [
        "A published specification is not compliance certification, deployment prevalence, multi-vendor interoperability, or measured application goodput.",
        "Existing clusters may run earlier revisions or proprietary extensions."
      ],
      "fak_implications": [
        "Record network specification/version and compliance evidence in cluster receipts.",
        "Benchmark the deployed implementation rather than assuming specification conformance implies performance."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "microsoft-phi4-reasoning-2025",
      "category": "frontier_lab",
      "entity": "Microsoft Research Phi",
      "topic": [
        "small_language_model",
        "reasoning_mode",
        "inference_time_compute",
        "open_weights"
      ],
      "published_at": "2025-04-30",
      "event_at": "2025-04-30",
      "source_title": "Phi-4-reasoning Technical Report",
      "source_url": "https://www.microsoft.com/en-us/research/publication/phi-4-reasoning-technical-report/",
      "source_kind": "technical_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Microsoft Research introduced Phi-4-reasoning as a 14B-parameter reasoning model fine-tuned from Phi-4 with curated prompts and o3-mini reasoning demonstrations, using detailed reasoning chains and inference-time compute.",
      "quantified": {
        "parameters": 14000000000
      },
      "assumptions": [
        "Compact open-weight reasoning models create a different memory and deployment envelope while retaining long, variable reasoning outputs."
      ],
      "contradictions_or_limits": [
        "Author benchmark results do not establish neutral production goodput, user demand, or reasoning-length distribution.",
        "The report does not disclose production fleet, request share, cluster scale, or per-query inference-time-compute distribution."
      ],
      "fak_implications": [
        "Separate model parameter footprint from output-token and reasoning-time cost.",
        "Benchmark compact reasoning models under mode- and quality-matched output distributions."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "huawei-pangu5-industrial-scale-2025",
      "category": "frontier_lab",
      "entity": "Huawei Cloud Pangu",
      "topic": [
        "model_family",
        "industry_models",
        "parameter_scale",
        "regional_hardware",
        "production_adoption"
      ],
      "published_at": "2025-03-26",
      "event_at": "2024-12-31",
      "source_title": "Huawei 2024 Annual Report",
      "source_url": "https://www.huawei.com/en/annual-report/2024",
      "source_kind": "official_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Huawei reported Pangu 5.0 models spanning more than 1B, 10B, 100B, and 1T parameters and deployment across more than 400 scenarios in over 30 industries.",
      "quantified": {
        "parameter_tiers": [
          1000000000,
          10000000000,
          100000000000,
          1000000000000
        ],
        "deployed_scenarios_gt": 400,
        "industries_gt": 30
      },
      "assumptions": [
        "Regional labs may operate tiered industry-specific model families on domestic cloud and accelerator stacks rather than one general-purpose endpoint."
      ],
      "contradictions_or_limits": [
        "Scenario count is vendor-reported adoption, not requests, users, revenue, accelerator inventory, or goodput.",
        "The annual report does not disclose training cluster, serving hardware mix, batch/cache policy, or model-tier traffic shares."
      ],
      "fak_implications": [
        "Track model tier, industry scenario, region, and accelerator stack as separate routing dimensions.",
        "Do not infer production load from scenario counts or parameter ceilings."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "lg-exaone4-hybrid-2025",
      "category": "frontier_lab",
      "entity": "LG AI Research",
      "topic": [
        "reasoning_mode",
        "on_device",
        "enterprise",
        "tool_use",
        "regional_language"
      ],
      "published_at": "2025-07-15",
      "event_at": "2025-07-15",
      "source_title": "Unveiling EXAONE 4.0, the next generation of hybrid AI",
      "source_url": "https://www.lgresearch.ai/blog/view?seq=576",
      "source_kind": "official_model_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "LG AI Research released EXAONE 4.0 as a hybrid reasoning/non-reasoning family with a 32B professional model and 1.2B on-device model, supporting Korean, English, Spanish, function calling, and MCP-oriented agent use.",
      "quantified": {
        "professional_parameters": 32000000000,
        "on_device_parameters": 1200000000,
        "supported_languages": 3
      },
      "assumptions": [
        "One family may route between server and device tiers and between direct and reasoning modes.",
        "Regional-language and enterprise tool workloads can have distinct tokenization and privacy constraints."
      ],
      "contradictions_or_limits": [
        "Performance and efficiency claims are vendor-reported; no physical cluster, production traffic, mode share, or device latency distribution is disclosed.",
        "Model downloads or derivative models do not establish active production use."
      ],
      "fak_implications": [
        "Represent device/server and reasoning/direct mode as explicit scheduler inputs.",
        "Measure Korean/Spanish tokenizer expansion and tool-call behavior separately from English."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aisingapore-sealion45-2026",
      "category": "frontier_lab",
      "entity": "AI Singapore SEA-LION",
      "topic": [
        "regional_language",
        "multimodal",
        "agentic",
        "speculative_decoding",
        "open_model"
      ],
      "published_at": "2026-05-20",
      "event_at": "2026-05-20",
      "source_title": "SEA-LION: Empowering Open Multilingual AI for Southeast Asia",
      "source_url": "https://aisingapore.org/aiproducts/sea-lion/",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "AI Singapore describes SEA-LION v4.5 as an open multilingual, multimodal, agentic model family for more than 11 Southeast Asian languages, with a custom speculative decoder claimed to deliver up to 6x inference efficiency.",
      "quantified": {
        "languages_gt": 11,
        "speculative_efficiency_claim_up_to": 6
      },
      "assumptions": [
        "Regional-language models require locale-specific token, safety, and cultural workload distributions.",
        "A custom draft/speculative path can be part of a sovereign open-model stack."
      ],
      "contradictions_or_limits": [
        "The 6x figure is an official project claim without a matched production envelope in this corpus.",
        "The page does not disclose acceptance distribution, model size, hardware, traffic, or user shares by language."
      ],
      "fak_implications": [
        "Benchmark speculative acceptance and verifier cost by language and prompt class.",
        "Include Southeast Asian multilingual and multimodal agent workloads in regional replay."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ntt-tsuzumi-lightweight-2024",
      "category": "frontier_lab",
      "entity": "NTT tsuzumi",
      "topic": [
        "regional_language",
        "small_language_model",
        "cpu_inference",
        "single_gpu",
        "enterprise_deployment"
      ],
      "published_at": "2024-03-26",
      "event_at": "2024-03-25",
      "source_title": "NTT large language model tsuzumi characteristics and deployment",
      "source_url": "https://www.ntt.com/bizon/tsuzumi.html",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "NTT describes tsuzumi service tiers at 0.6B and 7B parameters, with the ultra-light model intended for CPU inference and the 7B tier intended to run on one GPU for Japanese enterprise workloads.",
      "quantified": {
        "ultralight_parameters": 600000000,
        "light_parameters": 7000000000,
        "light_gpu_count": 1
      },
      "assumptions": [
        "Small regional-language models can target CPU/single-GPU private deployment and industry tuning rather than datacenter-scale general inference."
      ],
      "contradictions_or_limits": [
        "Size and fit claims do not establish latency, concurrency, quality, power, or production demand.",
        "The source does not disclose hardware SKU, quantization, context length, request distribution, or observed cost."
      ],
      "fak_implications": [
        "Include CPU and single-GPU Japanese enterprise tiers in deployment matrices.",
        "Measure private-deployment cost and tokenizer behavior rather than assuming parameter count determines efficiency."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mbzuai-k2think-v2-2026",
      "category": "frontier_lab",
      "entity": "MBZUAI / G42 Institute of Foundation Models",
      "topic": [
        "sovereign_ai",
        "reasoning_mode",
        "open_model",
        "training_transparency",
        "regional_lab"
      ],
      "published_at": "2026-01-27",
      "event_at": "2026-01-27",
      "source_title": "K2 Think V2: a fully sovereign reasoning model",
      "source_url": "https://mbzuai.ac.ae/news/k2-think-v2-a-fully-sovereign-reasoning-model/",
      "source_kind": "official_model_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "MBZUAI released K2 Think V2 as a 70B-parameter open general reasoning model described as fully sovereign and open from pretraining through post-training using Institute-curated data.",
      "quantified": {
        "parameters": 70000000000
      },
      "assumptions": [
        "Sovereign model programs can treat training-data provenance, code, checkpoints, and regional control as first-class product requirements."
      ],
      "contradictions_or_limits": [
        "Benchmark leadership is author-reported and does not disclose physical training cluster, production traffic, reasoning-token distribution, or serving cost.",
        "“Sovereign” is a governance/provenance claim, not capacity or locality evidence by itself."
      ],
      "fak_implications": [
        "Track training provenance, openness, jurisdiction, and deployment location separately.",
        "Benchmark reasoning output tails and serving footprint rather than inferring efficiency from parameter count."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "dynamo-planner-autoscaling-2026",
      "category": "serving_system",
      "entity": "NVIDIA Dynamo Planner",
      "topic": [
        "autoscaling",
        "prefill_decode",
        "traffic_prediction",
        "sla",
        "performance_model"
      ],
      "published_at": "2026-07-20",
      "event_at": "2026-07-20",
      "source_title": "Dynamo Planner Guide",
      "source_url": "https://docs.nvidia.com/dynamo/components/planner/planner-guide",
      "source_kind": "official_engineering_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "NVIDIA documents Dynamo Planner as an autoscaling controller that separately adjusts prefill and decode replica counts using traffic signals or prediction and engine performance models to target TTFT and inter-token-latency SLAs.",
      "quantified": {},
      "assumptions": [
        "Disaggregated serving requires coordinated phase-specific scaling rather than one monolithic replica count."
      ],
      "contradictions_or_limits": [
        "Documentation describes controller behavior but does not prove general SLO attainment, forecast accuracy, cold-start cost, or optimality across workloads.",
        "Performance-model error, provisioning delay, network bottlenecks, and failure handling remain envelope-specific."
      ],
      "fak_implications": [
        "Represent prefill and decode pools, forecast horizon, provisioning latency, and TTFT/ITL targets explicitly.",
        "Count controller and cold-start overhead in autoscaling evidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "llmd-autoscaling-guide-2026",
      "category": "serving_system",
      "entity": "llm-d / Google",
      "topic": [
        "autoscaling",
        "prefix_cache",
        "routing",
        "kubernetes",
        "hardware_heterogeneity"
      ],
      "published_at": "2026-08-06",
      "event_at": "2026-08-06",
      "source_title": "llm-d Well-Lit Path Guides",
      "source_url": "https://github.com/llm-d/llm-d/blob/main/guides/README.md",
      "source_kind": "official_engineering_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "llm-d documents benchmarked deployment recipes for prefix-cache-aware routing, disaggregation, KV offload, and proactive autoscaling, while warning that the manifests are starting points rather than configurations for every deployment.",
      "quantified": {},
      "assumptions": [
        "Kubernetes-native serving composes router, model-service, cache, and autoscaling controllers rather than one engine process."
      ],
      "contradictions_or_limits": [
        "Recipes and benchmarks do not establish production prevalence or universal gains.",
        "Hardware, workload, SLO, cache state, and operational maturity must match before comparison."
      ],
      "fak_implications": [
        "Treat serving features as composable but independently measurable controls.",
        "Retain configuration, version, hardware, and workload receipts for every result."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "operator-autoscaling-unit-2026",
      "category": "serving_system",
      "entity": "Operator-level LLM autoscaling study",
      "topic": [
        "autoscaling",
        "scaling_unit",
        "prefill_decode",
        "operator_placement",
        "trace_replay",
        "queue_model",
        "gpu_power",
        "slo"
      ],
      "published_at": "2026-08-13",
      "event_at": "2026-08-13",
      "source_title": "OpScale: Operator-level Provisioning and Autoscaling for LLM Serving",
      "source_url": "https://arxiv.org/html/2608.13499v1",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "OpScale is a roughly 17K-line nano-vLLM prototype that scales and places individual operator replicas. In production-trace replay on a 40-A100 cluster, it reports average allocations of 7.1 GPUs for Qwen2-7B and 10.8 GPUs for Qwen2-57B-A14B with 98.4% and 98.1% SLO attainment, respectively; separate comparisons against model-level provisioning report up to 36.3% fewer GPUs, 28% lower cluster power, or 44% higher input-token throughput within the paper's disclosed model, hardware, trace, sequence-length, and TTFT envelopes.",
      "quantified": {
        "prototype_implementation_python_lines_approx": 17000,
        "common_inference_backend": "nano-vLLM with the same model implementation, kernels, continuous-batching scheduler, KV-cache manager, and tensor-parallel backend for OpScale and ported baselines",
        "prototype_evaluation_models": [
          "Qwen2-7B dense",
          "Qwen2-57B-A14B MoE"
        ],
        "characterization_model_families": [
          "Qwen2-7B",
          "Qwen2-57B-A14B",
          "Llama3-8B",
          "Mixtral-8x7B",
          "Qwen2.5-VL-32B"
        ],
        "a100_cluster_gpus": 40,
        "a100_cluster_azure_vms": 5,
        "a100_gpus_per_vm": 8,
        "a100_memory_gb_per_gpu": 80,
        "a100_topology": "NVLink within each 8-GPU VM; high-speed InfiniBand across VMs",
        "gb200_sensitivity_cluster_gpus": 24,
        "gb200_topology": "same NVLink/NVL domain",
        "trace_replay_requests_per_model": 929000,
        "trace_replay_prompt_tokens_per_model": 1500000000,
        "displayed_dynamic_trace_window_seconds": 3600,
        "trace_replay_arrivals_and_lengths": "temporal request-arrival patterns and sequence lengths from the production LLM serving traces used by DynamoLLM; not generated from the analytical Poisson law",
        "opscale_control_interval_seconds": 1,
        "model_level_baseline_control_interval_seconds": 20,
        "dense_dynamic_p99_ttft_slo_seconds": 1.0,
        "moe_dynamic_p99_ttft_slo_seconds": 2.0,
        "dense_dynamic_average_gpu_count": 7.1,
        "dense_dynamic_baseline_average_gpu_counts": {
          "DynamoLLM": 11.2,
          "AIBrix_utilization_mode": 14.3,
          "Production_Stack_pending_prompt_tokens": 13.0
        },
        "dense_dynamic_slo_attainment_pct": 98.4,
        "dense_dynamic_baseline_slo_attainment_pct_range": [
          88,
          95
        ],
        "moe_dynamic_average_gpu_count": 10.8,
        "moe_dynamic_baseline_average_gpu_counts": {
          "DynamoLLM": 17.3,
          "AIBrix_utilization_mode": 23.1,
          "Production_Stack_pending_prompt_tokens": 18.0
        },
        "moe_dynamic_slo_attainment_pct": 98.1,
        "moe_dynamic_baseline_slo_attainment_pct_range": [
          84.2,
          97
        ],
        "moe_dynamic_peak_gpu_count_approx": 25,
        "moe_dynamic_baseline_peak_gpu_count_frequent_range_approx": [
          30,
          35
        ],
        "dense_power_reduction_vs_best_power_aware_baseline_pct": {
          "P50": 20,
          "P90": 16
        },
        "qwen2_7b_scale_up_latency_seconds": {
          "model_level": {
            "P99": 11.55,
            "P90": 11.04,
            "average": 10.68
          },
          "opscale_one_operator": {
            "P99": 0.1,
            "P90": 0.05,
            "average": 0.03
          },
          "opscale_50_pct_operators": {
            "P99": 0.42,
            "P90": 0.26,
            "average": 0.18
          },
          "opscale_all_operators": {
            "P99": 0.45,
            "P90": 0.38,
            "average": 0.33
          }
        },
        "cost_sweep_dense_qps": [
          80,
          120,
          160,
          200,
          240
        ],
        "cost_sweep_moe_qps": [
          25,
          50,
          75,
          100,
          125
        ],
        "cost_sweep_sequence_lengths_tokens": [
          1000,
          4000,
          8000
        ],
        "example_average_per_gpu_power_watts": {
          "OpScale": 282,
          "model_level": 245
        },
        "analytical_oracle_models": [
          "Qwen2-7B",
          "Qwen2-57B-A14B"
        ],
        "analytical_oracle_qps_range": [
          10,
          100
        ],
        "analytical_oracle_sequence_length_tokens_range": [
          128,
          65536
        ],
        "analytical_queue_assumption": "per-operator M/M/R queue with Poisson arrivals and exponential service times; Erlang-C waiting time",
        "online_greedy_resource_cost_within_pct_of_bruteforce_oracle": 8,
        "qwen2_7b_plan_time_ms": {
          "median": 2.6,
          "P99": 3.2,
          "provisioning": 2.5,
          "placement": 0.1
        },
        "qwen2_57b_a14b_plan_time_ms": 4.4,
        "execution_plane_overhead_pct": 0.3,
        "execution_plane_overhead_ms_per_500ms_prefill_approx": 1.5,
        "offline_57b_profile_time": "under one hour on one GB200 node",
        "model_prediction_relative_error_pct": {
          "operator_sensitivity_average": 7,
          "operator_sensitivity_P90": 15,
          "sm_contention_average": 5,
          "sm_contention_P90": 9.4,
          "queue_latency_average": 0.8,
          "queue_latency_P90": 1.9
        },
        "operator_transfer_overhead_vs_compute_pct": "below 5% for most operators and approximately 20% for SiLU Mul in the characterization",
        "op_level_resharding_overhead": "11x replica-scaling overhead; P99 1.15 seconds",
        "typical_profile_grid_not_achieved_runtime": {
          "batch_size_range": [
            1,
            256
          ],
          "sequence_length_tokens_range": [
            1,
            65536
          ],
          "sm_allocation_pct_range": [
            1,
            100
          ]
        },
        "configured_model_parallelism": "not disclosed per model; provisioning inherits tensor parallelism for intra-server and pipeline parallelism for cross-server deployment",
        "slo_constrained_gpu_reduction_pct_average_at_1k_vs_model_level": {
          "Qwen2_7B": 20.1,
          "Qwen2_57B_A14B": 35.7
        },
        "slo_constrained_gpu_reduction_pct_at_4k_vs_model_level": 36.3,
        "slo_constrained_gpu_reduction_pct_at_8k_vs_model_level": 22.1,
        "high_load_cluster_power_reduction_pct_range_vs_model_level": [
          14,
          28
        ],
        "fixed_budget_dense_input_tps_improvement_pct_range_vs_model_level": [
          3,
          38
        ],
        "fixed_budget_moe_input_tps_improvement_pct_up_to_at_40_gpus_vs_model_level": 44,
        "gb200_vs_a100_gpu_count_reduction_pct_at_40_qps": {
          "OpScale": 52,
          "monolith": 38
        },
        "a100_gpu_demand_reduction_vs_attn_ffn_pct_up_to": 33,
        "a100_max_request_throughput_gain_vs_attn_ffn_x": 1.7
      },
      "assumptions": [
        "The paper's prototype results assume one model owns a dedicated GPU set; multi-model, multi-tenant fairness and cross-model interference are future work.",
        "The dynamic prototype replays production-derived arrivals and sequence lengths, while the separate offline opportunity analysis uses a per-operator M/M/R approximation with Poisson arrivals and exponential service times.",
        "Operator replicas are split into monolithic base instances plus elastic components, placed by interference-aware best fit with intra-device, then intra-server NVLink/NVL, then cross-server InfiniBand locality; weighted shortest-queue dispatch selects among replicas.",
        "Warm standby capacity is counted as provisioned even while idle; headline savings therefore compare on-demand scale-out without a pre-reserved full-model replica."
      ],
      "contradictions_or_limits": [
        "This is a research prototype and trace replay, not a measured production deployment or evidence that operator-level autoscaling is prevalent at providers.",
        "The arXiv v1 landing page and TeX source link no OpScale code, configuration, trace, or result-data repository; nano-vLLM, kvcached, Dynamo, and Production Stack are referenced dependencies or baselines, not an OpScale artifact.",
        "No achieved active-batch or batch-size distribution is disclosed. The B=1-256 range is a typical offline profiling space, not the configured maximum or achieved active batch in the reported runs.",
        "No numeric queue depth, pending-prompt-token count, queue-wait distribution, or per-operator waiting-time distribution is reported; queue buildup is described qualitatively and the queue-driven baseline signal is named without its observed values.",
        "No exact achieved cluster GPU-utilization percentage or distribution is reported. The paper says per-GPU utilization is higher and uses GPU utilization as an AIBrix signal, but does not publish a readable fleet utilization value for the autoscaling runs.",
        "No realized per-operator replica-count series, total operator-replica count, replica placement map, per-model tensor/pipeline-parallel degrees, SM-allocation distribution, or migration count is disclosed; average and approximate peak GPU counts are not substitutes.",
        "No explicit goodput metric is reported. Static results are maximum sustainable input tokens per second at an SLO limit, and dynamic results report SLO attainment separately.",
        "Numeric TBT SLOs, output-token totals or distributions, request-rate time series values, failures, retries, cancellations, cold-start failures, recovery time, and workload cost in currency or GPU-hours are not disclosed.",
        "The granularity/topology comparison against Attn-FFN reports up to 33% lower GPU demand and 1.7x maximum request throughput on A100; separately, at the readable 40-QPS point, A100-to-GB200 GPU-count reductions are 52% for OpScale and 38% for the monolith. The figure does not disclose its model, sequence length, trace or arrival law, or SLO.",
        "The 36.3% GPU, 28% power, and 44% input-TPS maxima come from different sequence-length, load, baseline, hardware, or fixed-budget experiments and must not be combined as one universal gain."
      ],
      "fak_implications": [
        "Benchmark monolithic, phase-level, and operator-level scaling on the same replay with model, hardware, topology, control interval, QPS/length envelope, TTFT/TBT target, and warm-capacity accounting fixed.",
        "Record achieved active batch, queue depth and wait, per-operator replica counts and placement, GPU/SM utilization, scaling actions, migrations, failures/retries, and SLO-satisfied accepted output in addition to average GPU count and input TPS.",
        "Charge operator scale-out, placement, dispatch, cross-device transfer, profile/model error, and resharding overhead; prefer horizontal operator replication over resharding unless the measured envelope reverses the paper's 11x overhead result.",
        "Treat the paper as a WATCH/benchmark envelope for finer-grained autoscaling, not as evidence of provider production prevalence or a default fak-native serving architecture."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "taichi-hybrid-pd-2025",
      "category": "serving_system",
      "entity": "TaiChi serving study",
      "topic": [
        "prefill_decode",
        "aggregation",
        "disaggregation",
        "hybrid_scheduling",
        "goodput",
        "slo"
      ],
      "published_at": "2025-08-04",
      "event_at": "2025-08-04",
      "source_title": "Prefill-Decode Aggregation or Disaggregation? Unifying Both for Goodput-Optimized LLM Serving",
      "source_url": "https://arxiv.org/abs/2508.01989",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "TaiChi reports that aggregation can favor tight TTFT/relaxed TPOT, disaggregation can favor strict TPOT/relaxed TTFT, and a hybrid can outperform both under balanced SLOs, with up to 77% reported goodput improvement.",
      "quantified": {
        "reported_goodput_improvement_pct_up_to": 77
      },
      "assumptions": [
        "Prefill/decode architecture is an SLO- and workload-dependent control, not a universal binary choice."
      ],
      "contradictions_or_limits": [
        "The maximum gain is benchmark-specific and depends on baselines, hardware, model, request lengths, and SLO mix.",
        "Hybrid scheduling adds complexity and does not prove production prevalence."
      ],
      "fak_implications": [
        "Sweep TTFT and TPOT jointly when selecting aggregated, disaggregated, or hybrid serving.",
        "Reject architecture claims that report throughput without SLO-satisfied goodput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tokenscale-autoscaling-2025",
      "category": "serving_system",
      "entity": "TokenScale autoscaling study",
      "topic": [
        "autoscaling",
        "token_velocity",
        "prefill_decode",
        "bursts",
        "convertible_decoders"
      ],
      "published_at": "2025-12-03",
      "event_at": "2025-12-03",
      "source_title": "TokenScale: Timely and Accurate Autoscaling for Disaggregated LLM Serving with Token Velocity",
      "source_url": "https://arxiv.org/abs/2512.03416",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "TokenScale proposes token velocity as a cross-stage backpressure signal and convertible decoders as burst buffers; its production-trace experiments report SLO attainment rising from 50–88% to 80–96% and cost reductions of 4–14% against selected systems.",
      "quantified": {
        "baseline_slo_attainment_pct_range": [
          50,
          88
        ],
        "reported_slo_attainment_pct_range": [
          80,
          96
        ],
        "reported_cost_reduction_pct_range": [
          4,
          14
        ]
      },
      "assumptions": [
        "Request count and GPU utilization can lag token-level work and phase imbalance in disaggregated serving."
      ],
      "contradictions_or_limits": [
        "Results are an experimental replay, not a production deployment measurement.",
        "Token velocity and convertible decoders may behave differently under model churn, multimodality, failures, and heterogeneous interconnect."
      ],
      "fak_implications": [
        "Evaluate token-work and queue/backpressure metrics alongside requests and utilization.",
        "Measure decoder conversion delay and lost decode capacity during bursts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "heteroscale-production-2025",
      "category": "serving_system",
      "entity": "HeteroScale production study",
      "topic": [
        "autoscaling",
        "heterogeneous_hardware",
        "prefill_decode",
        "topology",
        "production_measurement"
      ],
      "published_at": "2025-08-27",
      "event_at": "2025-08-27",
      "source_title": "Taming the Chaos: Coordinated Autoscaling for Heterogeneous and Disaggregated LLM Inference",
      "source_url": "https://arxiv.org/abs/2508.19559",
      "source_kind": "preprint",
      "evidence_class": "production_measurement",
      "confidence": "medium_high",
      "claim": "HeteroScale reports deployment across tens of thousands of production GPUs, coordinating heterogeneous topology and prefill/decode scaling; it reports a 26.6 percentage-point increase in average GPU utilization while maintaining SLOs.",
      "quantified": {
        "production_gpu_scale": "tens_of_thousands",
        "reported_gpu_utilization_increase_percentage_points": 26.6
      },
      "assumptions": [
        "Large production fleets need topology-aware placement and coordinated phase scaling across heterogeneous hardware."
      ],
      "contradictions_or_limits": [
        "The paper does not make the underlying proprietary fleet trace and full economics independently reproducible.",
        "Reported gains depend on baseline, fleet composition, workload, network, and operational policy."
      ],
      "fak_implications": [
        "Treat hardware generation, network topology, and phase pool balance as autoscaler state.",
        "Label this as reported production measurement and require matched local reproduction before claiming analogous gains."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "lambda-microsoft-gpu-contract-2025",
      "category": "ai_cloud",
      "entity": "Lambda",
      "topic": [
        "neocloud",
        "customer_contract",
        "gpu_deployment",
        "customer_concentration",
        "funding"
      ],
      "published_at": "2025-11-03",
      "event_at": "2025-11-03",
      "source_title": "Lambda Announces Multibillion-Dollar Agreement With Microsoft to Deploy AI Infrastructure Powered by Tens of Thousands of NVIDIA GPUs",
      "source_url": "https://lambda.ai/blog/lambda-announces-multibillion-dollar-agreement-with-microsoft-to-deploy-ai-infrastructure-powered-by-tens-of-thousands-of-nvidia-gpus",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Lambda announced a multi-year, multibillion-dollar Microsoft agreement to deploy tens of thousands of NVIDIA GPUs including GB300 NVL72 systems; later financing was positioned to expand its AI-factory footprint.",
      "quantified": {
        "physical_nvidia_gpus": "tens_of_thousands"
      },
      "assumptions": [
        "Neocloud capacity can be financed against concentrated multi-year hyperscaler demand and deployed in named accelerator systems."
      ],
      "contradictions_or_limits": [
        "The announcement does not disclose exact contract value, GPU count, site, power, delivery schedule, acceptance, utilization, or revenue recognition.",
        "A contract and financing plan are not installed healthy capacity."
      ],
      "fak_implications": [
        "Track customer concentration, financed assets, physical delivery, acceptance, and useful goodput separately.",
        "Retain accelerator SKU and system boundary rather than converting to generic GPU equivalents."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "applied-digital-polaris-live-2026",
      "category": "datacenter_physical",
      "entity": "Applied Digital",
      "topic": [
        "site_lifecycle",
        "critical_it_load",
        "ready_for_service",
        "contracted_capacity",
        "construction"
      ],
      "published_at": "2026-07-01",
      "event_at": "2026-07-01",
      "source_title": "Applied Digital Delivers Second Building at Polaris Forge 1",
      "source_url": "https://ir.applieddigital.com/_assets/_7657ad86a3899c3d591ec79f9937f175/appliedblockchaininc/news/2026-07-01_Applied_Digital_Delivers_Second_Building_at_157.pdf",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Applied Digital said Building 2 Phase 1 at Polaris Forge 1 reached Ready for Service with 75 MW, bringing campus live capacity to 175 MW; full-build contracted critical IT load is 400 MW.",
      "quantified": {
        "new_ready_for_service_mw": 75,
        "campus_live_capacity_mw": 175,
        "full_build_contracted_critical_it_mw": 400
      },
      "assumptions": [
        "Ready-for-service and live critical IT load are stronger lifecycle evidence than lease, campus, or utility-power totals."
      ],
      "contradictions_or_limits": [
        "Company-reported Ready for Service does not disclose customer accelerator installation, utilization, PUE, uptime, or useful AI goodput.",
        "The remaining 225 MW of contracted full-build load is not yet live in this source."
      ],
      "fak_implications": [
        "Track site phase, live critical IT MW, contracted future MW, and customer compute acceptance separately.",
        "Do not treat total campus design or lease capacity as current service capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "lightmatter-m1000-photonic-platform-2025",
      "category": "supply_chain",
      "entity": "Lightmatter",
      "topic": [
        "photonic_interconnect",
        "scale_up",
        "bandwidth",
        "validation",
        "production_readiness"
      ],
      "published_at": "2025-09-05",
      "event_at": "2025-09-05",
      "source_title": "Seeing is Believing: A Technical Deep Dive into Lightmatter’s Hardware",
      "source_url": "https://lightmatter.co/blog/seeing-is-believing-a-technical-deep-dive-into-lightmatters-hardware/",
      "source_kind": "official_engineering_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Lightmatter described Passage M1000 as a production-ready 3D photonic interposer/server platform with a 4,000 mm² die complex, 34 chiplets, 1,024 SerDes lanes, 256 optical fibers, and up to 114 Tbps total bandwidth.",
      "quantified": {
        "silicon_die_complex_mm2": 4000,
        "integrated_chiplets": 34,
        "serdes_lanes": 1024,
        "optical_fibers": 256,
        "total_bandwidth_tbps_up_to": 114
      },
      "assumptions": [
        "Scale-up interconnect density and energy can gate package and rack scaling independently of compute silicon."
      ],
      "contradictions_or_limits": [
        "Production-ready and bandwidth figures are vendor claims; shipments, yield, customer qualification, system availability, power, error rates, and application goodput are not disclosed.",
        "Component bandwidth is not cluster goodput."
      ],
      "fak_implications": [
        "Track sample, validation, qualification, shipment, integration, and production states for photonic interconnect.",
        "Count protocol, topology, error, power, packaging, and application overhead before using peak bandwidth."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "marvell-celestial-ai-acquisition-2026",
      "category": "market_signal",
      "entity": "Marvell / Celestial AI",
      "topic": [
        "acquisition",
        "photonic_fabric",
        "interconnect",
        "revenue_milestones",
        "startup_lifecycle"
      ],
      "published_at": "2026-02-02",
      "event_at": "2026-02-02",
      "source_title": "Marvell Completes Acquisition of Celestial AI",
      "source_url": "https://investor.marvell.com/news-events/press-releases/detail/1005/marvell-completes-acquisition-of-celestial-ai",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Marvell completed its Celestial AI acquisition on February 2, 2026, integrating the team and photonic-fabric technology into its Data Center Group and projecting initial revenue contribution in the second half of fiscal 2028.",
      "quantified": {
        "cash_balance_reduction_usd": 1000000000,
        "new_diluted_shares_approx": 27000000,
        "annual_non_gaap_opex_increase_usd_approx": 50000000,
        "target_annualized_revenue_fy2028_q4_usd": 500000000,
        "target_annualized_revenue_fy2029_q4_usd": 1000000000
      },
      "assumptions": [
        "Optical-fabric startups may consolidate into incumbent semiconductor portfolios before high-volume deployment."
      ],
      "contradictions_or_limits": [
        "Completion and projected revenue milestones do not establish product shipments, qualification, interoperability, or deployed goodput.",
        "Integration success, roadmap continuity, and revenue projections require later evidence."
      ],
      "fak_implications": [
        "Track acquisition close, product integration, customer qualification, revenue milestones, and roadmap continuity separately.",
        "Do not interpret purchase price as deployed capacity or technical validation."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "fireworks-series-d-tokens-2026",
      "category": "ai_cloud",
      "entity": "Fireworks AI",
      "topic": [
        "inference_platform",
        "funding",
        "revenue_run_rate",
        "token_volume",
        "customer_specialization"
      ],
      "published_at": "2026-07-15",
      "event_at": "2026-07-15",
      "source_title": "Announcing our Series D and $1B ARR",
      "source_url": "https://fireworks.ai/blog/series-d-announcement",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Fireworks announced a $1.505B Series D at a $17.5B valuation, more than $1B annualized revenue run rate, and more than 40T tokens served daily, with over 95% from customer-specialized models.",
      "quantified": {
        "series_d_usd": 1505000000,
        "valuation_usd": 17500000000,
        "annualized_revenue_run_rate_usd_gt": 1000000000,
        "tokens_per_day_gt": 40000000000000,
        "customer_specialized_token_share_pct_gt": 95
      },
      "assumptions": [
        "Inference-platform traffic may be dominated by customer-specialized models rather than one public model popularity distribution."
      ],
      "contradictions_or_limits": [
        "All business and token-volume figures are company-reported; token counting, cache effects, model mix, customer concentration, quality, and profitability are undisclosed.",
        "Funding, valuation, ARR, and token volume are different evidence classes and do not prove margin or physical capacity."
      ],
      "fak_implications": [
        "Separate public-model, fine-tuned, and customer-specialized traffic in workload generation.",
        "Require token accounting, cache, model, hardware, and revenue definitions before comparing inference platforms."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openai-malone-departure-confirmed-2026",
      "category": "market_signal",
      "entity": "OpenAI",
      "topic": [
        "leadership",
        "datacenter_strategy",
        "rumor_resolution"
      ],
      "published_at": "2026-08-25",
      "event_at": "2026-08-24",
      "source_title": "OpenAI confirms its data centre chief left as it pivots to leasing",
      "source_url": "https://thenextweb.com/news/openai-head-of-data-centres-chris-malone-departure",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "An OpenAI spokesperson confirmed that data-center head Chris Malone was no longer with the company; reporting described the former role being split and some full-facility leasing work continuing under other leaders.",
      "quantified": {},
      "assumptions": [
        "Personnel confirmation can resolve the departure fragment without proving the cause or capacity effect."
      ],
      "contradictions_or_limits": [
        "The company did not provide a detailed primary statement on the cause, organizational rationale, or capacity impact.",
        "Leasing activity and team changes do not establish delivered or canceled capacity."
      ],
      "fak_implications": [
        "Resolve rumor fragments independently and preserve strategic interpretation as reported, not confirmed."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "anthropic-decart-reuters-corroboration-2026",
      "category": "market_signal",
      "entity": "Anthropic / Decart",
      "topic": [
        "acquisition_talks",
        "rumor_resolution",
        "inference_optimization"
      ],
      "published_at": "2026-08-12",
      "event_at": "2026-08-12",
      "source_title": "Anthropic in talks to buy Decart AI, source says",
      "source_url": "https://www.tradingview.com/news/reuters.com%2C2026%3Anewsml_L6N44A04I%3A0-anthropic-in-talks-to-buy-decart-ai-source-says/",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Reuters independently reported that Anthropic was in early talks to acquire Decart for about $6B; no final agreement or company confirmation was reported.",
      "quantified": {
        "reported_transaction_value_usd_approx": 6000000000
      },
      "assumptions": [
        "Independent reporting can corroborate that talks exist without confirming a transaction."
      ],
      "contradictions_or_limits": [
        "The source relies on unnamed sourcing; both parties did not confirm a signed or closed deal.",
        "Reported price, terms, staff/IP scope, and closing can change or fail."
      ],
      "fak_implications": [
        "Keep acquisition talks out of capacity, ownership, and roadmap assumptions until signed/closed evidence exists."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-memory-price-direction-corroboration-2026",
      "category": "market_signal",
      "entity": "NVIDIA / HBM suppliers",
      "topic": [
        "memory_cost",
        "pricing",
        "rumor_resolution",
        "gross_margin"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Nvidia resets its expectations on a profit metric due to 'extreme' memory-chip crunch",
      "source_url": "https://www.marketwatch.com/livecoverage/nvidia-earnings-stock-results-guidance-nvda-q2/card/nvidia-resets-its-expectations-on-a-profit-metric-due-to-extreme-memory-chip-crunch-KZTNMShlzs2dX78aZZTg",
      "source_kind": "credible_reporting",
      "evidence_class": "reported_observation",
      "confidence": "medium_high",
      "claim": "Live earnings-call coverage reported NVIDIA management describing extreme memory costs, lower future margin expectations, and planned product price increases early in the next fiscal year.",
      "quantified": {
        "reported_future_gross_margin_low_pct": 71,
        "reported_future_gross_margin_high_pct": 73
      },
      "assumptions": [
        "Memory inflation can pass through into system pricing and margins, but exact configuration-level increases remain contract- and BOM-specific."
      ],
      "contradictions_or_limits": [
        "This corroborates the direction of price pressure, not the earlier reported >15% magnitude or exact customer/configuration notice.",
        "Earnings-call reporting is not a published system price list."
      ],
      "fak_implications": [
        "Model memory-price sensitivity and keep exact system-price changes as scenarios until customer/product terms are confirmed."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "agentic-os-kernel-context-2025",
      "category": "serving_system",
      "entity": "Agentic OS research study",
      "topic": [
        "agentic_workload",
        "os_state",
        "kernel_context",
        "scheduling",
        "system_architecture"
      ],
      "published_at": "2025-04-11",
      "event_at": "2025-04-11",
      "source_title": "An Operating System for AI Agents",
      "source_url": "https://arxiv.org/abs/2504.06692",
      "source_kind": "preprint",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium",
      "claim": "This systems proposal argues that agent workloads need OS-like management for model calls, context/memory, tools, storage, access control, and multiple concurrent agents rather than treating each LLM request as an isolated stateless job.",
      "quantified": {},
      "assumptions": [
        "Agent execution composes model, tool, memory, storage, and policy state over long-lived workflows."
      ],
      "contradictions_or_limits": [
        "The proposal is not a production workload trace and does not quantify universal scheduler gains or user distributions.",
        "OS abstractions can add control overhead and must be justified against simpler runtimes."
      ],
      "fak_implications": [
        "Benchmark workflow state, tool/runtime latency, and policy checks end to end, not only token generation.",
        "Treat agent/session identity and durable context as scheduler state."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "parrot-serve-llm-programs-2024",
      "category": "serving_system",
      "entity": "Parrot serving study",
      "topic": [
        "llm_programs",
        "semantic_variables",
        "request_graph",
        "scheduling",
        "prefix_reuse"
      ],
      "published_at": "2024-05-21",
      "event_at": "2024-05-21",
      "source_title": "Parrot: Efficient Serving of LLM-based Applications with Semantic Variable",
      "source_url": "https://arxiv.org/abs/2405.19888",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "Parrot models an LLM application as a graph of semantic variables and dependent calls, enabling cross-request prefix sharing and graph-aware scheduling rather than optimizing independent requests only.",
      "quantified": {},
      "assumptions": [
        "Agentic and compound applications expose dependency graphs, shared prefixes, and critical paths that request-level queues hide."
      ],
      "contradictions_or_limits": [
        "Benchmarks are not broad production-agent traces and depend on application graph visibility and runtime integration.",
        "Graph construction and cross-request sharing add orchestration, privacy, and invalidation costs."
      ],
      "fak_implications": [
        "Represent workflow DAG, critical path, semantic prefix ownership, and tool dependencies in replay.",
        "Compare request-level and workflow-level scheduling under the same end-to-end objective."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "semantic-tool-result-caching-2025",
      "category": "serving_system",
      "entity": "Tool-result caching research",
      "topic": [
        "agentic_workload",
        "tool_calls",
        "semantic_cache",
        "redundancy",
        "workflow_latency"
      ],
      "published_at": "2025-10-17",
      "event_at": "2025-10-17",
      "source_title": "Semantic Caching for Tool-Augmented Language Model Agents",
      "source_url": "https://arxiv.org/abs/2510.15980",
      "source_kind": "preprint",
      "evidence_class": "synthetic_experiment",
      "confidence": "medium",
      "claim": "This study evaluates semantic reuse of tool results for tool-augmented agents, motivated by repeated or near-duplicate tool invocations that request-level LLM caches do not capture.",
      "quantified": {},
      "assumptions": [
        "Tool results can be a separate reuse distribution from prompt/KV prefixes, with freshness and side-effect constraints."
      ],
      "contradictions_or_limits": [
        "Experimental tasks do not establish production redundancy rates, freshness tolerances, or safe reuse across tenants.",
        "Caching mutating, personalized, or time-sensitive tools can be incorrect or unsafe."
      ],
      "fak_implications": [
        "Track tool identity, arguments, result digest, freshness, side effects, tenant, and authorization before reuse.",
        "Measure end-to-end saved latency/cost after semantic lookup and verification overhead."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "agentic-workflow-scheduling-2026",
      "category": "serving_system",
      "entity": "Agent workflow scheduling study",
      "topic": [
        "agentic_workload",
        "workflow_scheduling",
        "critical_path",
        "tool_latency",
        "goodput"
      ],
      "published_at": "2026-06-12",
      "event_at": "2026-06-12",
      "source_title": "Workflow-Aware Scheduling for LLM Agents",
      "source_url": "https://arxiv.org/abs/2606.11266",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "This study treats agent jobs as multi-stage workflows containing LLM and tool operations and optimizes workflow completion rather than per-request model latency.",
      "quantified": {},
      "assumptions": [
        "Agent throughput and latency depend on workflow critical paths, tool/runtime queues, fan-out, retries, and shared resources."
      ],
      "contradictions_or_limits": [
        "Benchmark workflows and schedulers may not represent production tool diversity, failures, user think time, or security policy.",
        "Improving workflow completion can trade off per-stage fairness or interactive latency."
      ],
      "fak_implications": [
        "Measure task completion, critical-path delay, useful tool outcomes, retries, and resource occupancy together.",
        "Do not optimize model-request latency while tool queues dominate the workflow."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "skywalker-cross-region-2025",
      "category": "serving_system",
      "entity": "SkyWalker cross-region study",
      "topic": [
        "multi_region",
        "diurnal",
        "prefix_locality",
        "routing",
        "geography"
      ],
      "published_at": "2025-05-30",
      "event_at": "2025-05-30",
      "source_title": "SkyWalker: A Locality-Aware Cross-Region Load Balancer for LLM Inference",
      "source_url": "https://arxiv.org/abs/2505.24095",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "SkyLB/SkyWalker maps WildChat demand from six countries into region-specific diurnal curves and evaluates cross-region routing that aggregates complementary peaks while preserving prefix locality; the system reports 1.12-2.06x higher throughput, 1.74-6.30x lower latency, and 25% lower serving cost than evaluated baselines.",
      "quantified": {
        "countries_analyzed": 6,
        "throughput_improvement_x_range": [
          1.12,
          2.06
        ],
        "latency_reduction_x_range": [
          1.74,
          6.3
        ],
        "serving_cost_reduction_fraction": 0.25,
        "round_robin_peak_kv_memory_imbalance_x": 2.64
      },
      "assumptions": [
        "Geographic demand can shift by local time, creating temporal capacity complementarity across regions.",
        "Cross-region balancing must trade WAN latency and policy constraints against utilization and prefix reuse.",
        "Session locality is represented by prefix-aware routing; it is not a disclosed production session-length distribution."
      ],
      "contradictions_or_limits": [
        "WildChat contains opt-in public chatbot traffic rather than a provider production billing/API population; the six-country mapping is an evaluation input, not a universal geography law.",
        "The paper reports system-level benchmark gains, not confidence intervals for country demand or tenant concentration.",
        "Benchmark gains depend on region mix, network latency, residency rules, reserved capacity, model/SLO envelope, and cache state."
      ],
      "fak_implications": [
        "Add region, timezone, session/user affinity, WAN latency, and data-locality constraints to replay.",
        "Compare local peak provisioning with cross-region pooled capacity under the same SLO and policy envelope."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "smetric-agent-scheduling-2026",
      "category": "serving_system",
      "entity": "SMetric agent scheduling study",
      "topic": [
        "agentic_workload",
        "session_locality",
        "kv_cache",
        "load_balance",
        "scheduling_metric"
      ],
      "published_at": "2026-07-09",
      "event_at": "2026-07-09",
      "source_title": "SMetric: Rethink LLM Scheduling for Serving Agents with Session Metrics",
      "source_url": "https://arxiv.org/abs/2607.08565",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "SMetric studies two real-world agent traces and finds that cache-affinity schedulers can overload a few instances while others remain idle; it proposes session-aware metrics that balance KV reuse against load.",
      "quantified": {
        "real_world_agent_traces": 2
      },
      "assumptions": [
        "Session locality and cache affinity create correlated, long-lived load that request-level queue length can misrepresent."
      ],
      "contradictions_or_limits": [
        "The traces and benchmark configurations do not establish universal production gains or tenant fairness.",
        "A different model, cache size, tool latency, session length, or hardware topology can change the reuse/load balance."
      ],
      "fak_implications": [
        "Schedule with session-level cached state and projected work, not cache hit alone.",
        "Measure idle replicas, hotspot queues, KV hit, throughput, and task completion together."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "continuum-cachettl-2026",
      "category": "serving_system",
      "entity": "Continuum multi-turn agent study",
      "topic": [
        "agentic_workload",
        "kv_cache_ttl",
        "tool_idle",
        "offload",
        "job_completion_time"
      ],
      "published_at": "2026-05-11",
      "event_at": "2026-05-11",
      "source_title": "Efficient and Robust Multi-Turn LLM Agent Scheduling with KV-Cache Retention",
      "source_url": "https://arxiv.org/abs/2511.02230",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "Continuum targets multi-turn agent job completion by assigning KV-cache TTLs during tool-call gaps based on reload cost and eviction-induced queueing, rather than pinning or evicting every session uniformly.",
      "quantified": {},
      "assumptions": [
        "Tool execution creates idle windows during which retained KV competes with active-request concurrency."
      ],
      "contradictions_or_limits": [
        "TTL benefits depend on tool-gap distribution, reload substrate, cache size, queue state, and session return probability.",
        "Benchmark evidence does not provide cross-product production tool-gap and memory-size distributions."
      ],
      "fak_implications": [
        "Replay tool-gap and session-return distributions; charge retained-memory time and reload cost.",
        "Compare pin, offload, evict, and TTL policies under task completion and fairness objectives."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sageserve-multiregion-2025",
      "category": "serving_system",
      "entity": "SageServe production-trace study",
      "topic": [
        "multi_region",
        "mixed_sla",
        "autoscaling",
        "model_placement",
        "spot_capacity"
      ],
      "published_at": "2025-02-20",
      "event_at": "2025-02-20",
      "source_title": "Serving Models, Fast and Slow: Optimizing Heterogeneous LLM Inferencing Workloads at Scale",
      "source_url": "https://arxiv.org/abs/2502.14617",
      "source_kind": "preprint",
      "evidence_class": "benchmark_measurement",
      "confidence": "medium_high",
      "claim": "SageServe evaluates mixed latency-sensitive and insensitive inference workloads using production traces with more than 8M requests, four open models, and three regions; it reports up to 25% GPU-hour savings while maintaining evaluated SLOs.",
      "quantified": {
        "production_trace_requests_gt": 8000000,
        "models": 4,
        "regions": 3,
        "reported_gpu_hour_savings_pct_up_to": 25,
        "reported_scaling_overhead_reduction_pct_up_to": 80
      },
      "assumptions": [
        "Global serving mixes SLO classes, models, regions, and spot/on-demand capacity at different timescales."
      ],
      "contradictions_or_limits": [
        "The optimization evaluation combines empirical and simulated components and does not disclose a universal geography or tenant distribution.",
        "Reported savings depend on cost model, spot availability, model placement time, forecast, and SLO mix."
      ],
      "fak_implications": [
        "Model latency-sensitive and batch/elastic classes separately across regions and provisioning types.",
        "Include model redeployment and VM scaling overhead before accepting capacity savings."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "schneider-gb300-reference-design-2025",
      "category": "datacenter_physical",
      "entity": "Schneider Electric / NVIDIA",
      "topic": [
        "reference_design",
        "liquid_cooling",
        "power_controls",
        "rack_density",
        "digital_twin"
      ],
      "published_at": "2025-09-18",
      "event_at": "2025-09-18",
      "source_title": "Schneider Electric Announces New Reference Designs, Featuring Integrated Power Management and Liquid Cooling Controls, Supporting NVIDIA Mission Control and NVIDIA GB300 NVL72",
      "source_url": "https://www.se.com/ww/en/about-us/newsroom/news/press-releases/schneider-electric-announces-new-reference-designs-featuring-integrated-power-management-and-liquid-cooling-controls-supporting-nvidia-mission-control-and-nvidia-gb300-nvl72-68ca8fdd5e6c6f9d3a096133/",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Schneider Electric and NVIDIA published a validated reference design for three GB300 NVL72 clusters in one data hall, supporting up to 142 kW per rack and 1,152 GPUs with liquid-to-liquid CDUs, high-temperature chillers, and integrated OT/IT power-and-cooling controls.",
      "quantified": {
        "max_rack_density_kw": 142,
        "clusters": 3,
        "gpus_up_to": 1152
      },
      "assumptions": [
        "Power, cooling, controls, IT space, and lifecycle software need co-design at rack-scale AI density."
      ],
      "contradictions_or_limits": [
        "A validated reference design is not a constructed, commissioned, occupied, or production-measured facility.",
        "The source does not provide site build time, installed cost, PUE/WUE, field reliability, utilization, or useful goodput."
      ],
      "fak_implications": [
        "Use reference designs to bound required physical systems, not as delivered-capacity receipts.",
        "Record rack density, cluster count, control interfaces, cooling loop, redundancy, and field acceptance separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "trane-dcda-apac-cdu-2026",
      "category": "supply_chain",
      "entity": "Trane Technologies",
      "topic": [
        "liquid_cooling",
        "cdu",
        "regional_supply",
        "delivery",
        "cluster_control"
      ],
      "published_at": "2026-01-22",
      "event_at": "2026-01-22",
      "source_title": "Trane Launches Advanced CDU to Enhance Liquid Cooling Efficiency in Asia-Pacific Data Centers",
      "source_url": "https://www.trane.com/commercial/asia-pacific/in/en/about-us/press-releases/Trane-Launches-Advanced-CDU-to-Enhance-Liquid-Cooling-Efficiency-in-Asia-Pacific-Data-Centers.html",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Trane launched an Asia-Pacific-developed DCDA CDU family in 400, 800, and 1,350 kW variants, configurable to 1,700 kW and cluster-controlled across up to 16 units; first China orders were expected in Q1 2026.",
      "quantified": {
        "standard_cooling_capacity_kw": [
          400,
          800,
          1350
        ],
        "custom_capacity_kw_up_to": 1700,
        "coordinated_units_up_to": 16,
        "claimed_pue_as_low_as": 1.1,
        "floor_space_savings_pct_up_to": 20
      },
      "assumptions": [
        "Regional cooling products have explicit unit, clustering, maintenance, certification, and delivery lifecycles."
      ],
      "contradictions_or_limits": [
        "Capacity, PUE, space, and control claims are vendor-reported and do not prove site-level efficiency or availability.",
        "Expected first delivery is not independently confirmed installation, commissioning, or customer operation."
      ],
      "fak_implications": [
        "Track ordered, delivered, installed, commissioned, and operating CDU capacity separately.",
        "Match coolant loop, redundancy, partial-load curve, controls, and service region to the site envelope."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "liquidstack-gigamodular-commercial-2026",
      "category": "supply_chain",
      "entity": "LiquidStack / Trane Technologies",
      "topic": [
        "liquid_cooling",
        "modular_cdu",
        "commercial_availability",
        "validation",
        "phased_delivery"
      ],
      "published_at": "2026-05-21",
      "event_at": "2026-05-21",
      "source_title": "LiquidStack GigaModular CDU Platform Is Now Commercially Available with Expanded 14 MW Capacity",
      "source_url": "https://www.trane.com/commercial/north-america/us/en/about-us/newsroom/press-releases/liquidstack-giga-modular-cdu.html",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "LiquidStack, a Trane company, announced commercial availability of its GigaModular CDU after multi-module integration and full-load testing, with modular building blocks scaling to 14 MW, ETL certification, and early customer orders.",
      "quantified": {
        "validated_capacity_mw_up_to": 14,
        "commercially_available": true
      },
      "assumptions": [
        "Modular cooling can align physical delivery with phased compute installation and avoid full upfront overprovisioning."
      ],
      "contradictions_or_limits": [
        "Commercial availability, testing, certification, and early orders do not establish shipment volume, site installation, uptime, efficiency, or accepted IT load.",
        "The 14 MW figure is cooling-system capacity, not critical IT MW or useful compute capacity."
      ],
      "fak_implications": [
        "Preserve cooling MW, IT MW, redundancy, test state, order, shipment, installation, and commissioning separately.",
        "Test system-level controls and partial-build efficiency rather than summing independent CDU nameplates."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "johnson-controls-q3-2026-backlog",
      "category": "supply_chain",
      "entity": "Johnson Controls",
      "topic": [
        "cooling",
        "building_systems",
        "backlog",
        "orders",
        "regional_delivery"
      ],
      "published_at": "2026-07-29",
      "event_at": "2026-06-30",
      "source_title": "Johnson Controls Reports Strong Q3 Results; Raises FY26 Guidance",
      "source_url": "https://investors.johnsoncontrols.com/news/news-details/2026/Johnson-Controls-Reports-Strong-Q3-Results-Raises-FY26-Guidance/default.aspx",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Johnson Controls reported $21.0B of Solutions and Services backlog, up 32% organically, with Americas backlog of $15.9B up 40% and demand supported by data centers and other mission-critical environments.",
      "quantified": {
        "total_backlog_usd": 21000000000,
        "backlog_yoy_organic_pct": 32,
        "americas_backlog_usd": 15900000000,
        "americas_backlog_yoy_pct": 40,
        "emea_backlog_usd": 3100000000,
        "apac_backlog_usd": 2000000000,
        "quarter_orders_yoy_organic_pct": 27
      },
      "assumptions": [
        "Cooling and building-system suppliers can become a delivery gate even when accelerator supply and site power exist."
      ],
      "contradictions_or_limits": [
        "Backlog includes Solutions and Services beyond datacenters and was restated to include certain equipment-only longer-cycle projects.",
        "Backlog is not shipped equipment, installed cooling, commissioned facility capacity, or live IT MW."
      ],
      "fak_implications": [
        "Track supplier backlog definition, product/site allocation, ship/install/commission dates, and live capacity.",
        "Do not allocate total company backlog to AI datacenters without disclosed segment/project detail."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "eaton-sc-transformer-factory-2025",
      "category": "supply_chain",
      "entity": "Eaton",
      "topic": [
        "transformers",
        "manufacturing_expansion",
        "factory",
        "production_start",
        "data_center_demand"
      ],
      "published_at": "2025-02-12",
      "event_at": "2025-02-12",
      "source_title": "Eaton invests in new South Carolina transformer manufacturing site to support U.S. electrical infrastructure",
      "source_url": "https://www.eaton.com/us/en-us/company/news-insights/news-releases/2025/eaton-invests-in-new-south-carolina-transformer-manufacturing.html",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Eaton announced a $340M investment in a new Jonesville, South Carolina factory for three-phase transformers serving utility, industrial, commercial, and datacenter demand, with production and hiring expected to begin in 2027.",
      "quantified": {
        "investment_usd": 340000000,
        "expected_production_start_year": 2027,
        "us_three_phase_transformer_factories_after_opening": 3
      },
      "assumptions": [
        "Transformer shortages and factory lead times can gate electrical delivery after datacenter demand is contracted."
      ],
      "contradictions_or_limits": [
        "Factory investment and planned production start are not current transformer output, orders shipped, substations commissioned, or live IT MW.",
        "The source does not disclose annual unit/MVA capacity, yields, customer allocation, or datacenter share."
      ],
      "fak_implications": [
        "Track factory construction, production start, output, allocation, shipment, site installation, and energization separately.",
        "Do not map investment dollars directly to transformer count or datacenter MW."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "hitachi-virginia-transformer-factory-2025",
      "category": "supply_chain",
      "entity": "Hitachi Energy",
      "topic": [
        "transformers",
        "manufacturing_expansion",
        "factory",
        "grid_equipment",
        "data_center_demand"
      ],
      "published_at": "2025-09-04",
      "event_at": "2025-09-04",
      "source_title": "Hitachi announces historic $1 billion USD manufacturing investment to power America’s energy future through production of critical grid infrastructure",
      "source_url": "https://www.hitachienergy.com/news-and-events/press-releases/2025/09/hitachi-announces-historic-1-billion-usd-manufacturing-investment-to-power-america-s-energy-future-through-production-of-critical-grid-infrastructure",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Hitachi Energy announced $1B of U.S. manufacturing investment, including $457M for a new large-power-transformer factory in South Boston, Virginia, aimed at transformer and high-voltage-equipment demand including AI datacenters.",
      "quantified": {
        "us_manufacturing_investment_usd": 1000000000,
        "virginia_large_power_transformer_factory_usd": 457000000
      },
      "assumptions": [
        "Large power transformers and high-voltage equipment form a separate supply chain from rack-level power and accelerators."
      ],
      "contradictions_or_limits": [
        "Announced factory investment is not completed construction, qualified output, shipment, project allocation, or commissioned grid capacity.",
        "The release does not provide annual MVA/unit output, lead-time reduction, yield, or datacenter allocation."
      ],
      "fak_implications": [
        "Track large-power-transformer factory/output separately from dry-type and distribution transformer capacity.",
        "Require project-level ship/install/energize evidence before admitting grid-ready site capacity."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "siemens-us-grid-manufacturing-2026",
      "category": "supply_chain",
      "entity": "Siemens Energy",
      "topic": [
        "transformers",
        "switchgear",
        "grid_components",
        "manufacturing_expansion",
        "services"
      ],
      "published_at": "2026-02-03",
      "event_at": "2026-02-03",
      "source_title": "Siemens Energy is investing $1 billion and creating highly skilled manufacturing jobs in the United States",
      "source_url": "https://www.siemens-energy.com/global/en/home/press-releases/siemens-energy-is-investing--1-billion-and-creating-highly-skill.html",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Siemens Energy announced $1B of U.S. manufacturing expansion across transformer production/service, large gas turbines, and a new Mississippi grid-component factory.",
      "quantified": {
        "us_manufacturing_investment_usd": 1000000000
      },
      "assumptions": [
        "Grid equipment and service capacity need brownfield and greenfield expansion to reduce project backlogs and maintenance bottlenecks."
      ],
      "contradictions_or_limits": [
        "Portfolio-wide investment does not reveal transformer/switchgear output, delivery dates, datacenter allocation, or commissioned capacity.",
        "Manufacturing expansion and standardized designs may shorten lead time but do not prove a specific project schedule."
      ],
      "fak_implications": [
        "Track manufacturing product split, service/refurbishment capacity, factory start, output, shipment, and project commissioning.",
        "Keep vendor speed-to-power claims as scenarios until site-level evidence exists."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "gevernova-india-grid-expansion-2025",
      "category": "supply_chain",
      "entity": "GE Vernova T&D India",
      "topic": [
        "transformers",
        "switchgear",
        "hvdc",
        "facts",
        "manufacturing_expansion",
        "backlog"
      ],
      "published_at": "2025-05-14",
      "event_at": "2025-05-14",
      "source_title": "GE Vernova to invest USD $16 million to expand manufacturing footprint in India to meet rising demand for advanced grid infrastructure",
      "source_url": "https://www.gevernova.com/news/press-releases/ge-vernova-invest-usd-16-million-expand-manufacturing-footprint-india-meet-rising",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "GE Vernova announced about $16M to add manufacturing/testing capacity in Chennai and Noida for HVDC, FACTS, transformers, switchgear, and related grid equipment; it said Electrification equipment backlog had more than tripled over the prior year.",
      "quantified": {
        "investment_usd_approx": 16000000,
        "reported_electrification_backlog_growth_multiple_gt": 3
      },
      "assumptions": [
        "Regional manufacturing and test capacity can feed both domestic and export grid projects, making electrical supply chains globally coupled."
      ],
      "contradictions_or_limits": [
        "Backlog growth and investment do not disclose product mix, factory throughput, export allocation, shipment, or commissioning.",
        "HVDC/FACTS and transformer/switchgear capacity cannot be summed as interchangeable units."
      ],
      "fak_implications": [
        "Track product-specific factory/testing capacity and regional export allocation.",
        "Do not convert supplier backlog growth into datacenter MW or interconnection completion."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "vast-bluefield-coreweave-2024",
      "category": "supply_chain",
      "entity": "VAST Data / NVIDIA / CoreWeave",
      "topic": [
        "storage",
        "dpu",
        "data_movement",
        "deployment",
        "multi_tenant"
      ],
      "published_at": "2024-03-12",
      "event_at": "2024-03-12",
      "source_title": "VAST Data Unveils New Data Center Architecture for the AI Factory",
      "source_url": "https://www.vastdata.com/press-releases/vast-nvidia-bluefield-architecture-for-ai-factory",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "VAST and NVIDIA described a BlueField-3 DPU architecture that embeds stateless storage/database services in every GPU server and said it was being tested and deployed first at CoreWeave, with claimed reductions in independent infrastructure footprint and power.",
      "quantified": {
        "claimed_vast_infrastructure_power_and_footprint_reduction_pct": 70,
        "claimed_net_energy_savings_pct_gt": 5,
        "target_scale_gpus": "hundreds_of_thousands"
      },
      "assumptions": [
        "Data services, DPUs, storage, and database processing can be integrated into GPU servers rather than isolated storage nodes."
      ],
      "contradictions_or_limits": [
        "Deployment and efficiency figures are vendor claims without neutral production workload, hardware, capacity, uptime, or cost disclosure.",
        "Target scale and being tested/deployed do not quantify installed prevalence or useful application goodput."
      ],
      "fak_implications": [
        "Track data path, DPU services, protocol, isolation, storage/network resource share, and measured end-to-end GPU utilization.",
        "Count storage/database/DPU power and failures in AI-factory receipts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "vast-series-f-storage-scale-2026",
      "category": "market_signal",
      "entity": "VAST Data",
      "topic": [
        "storage",
        "funding",
        "bookings",
        "carr",
        "gpu_environment_claim"
      ],
      "published_at": "2026-04-22",
      "event_at": "2026-04-22",
      "source_title": "VAST Data Valued at $30 Billion as AI Drives a New Infrastructure Stack",
      "source_url": "https://www.vastdata.com/press-releases/vast-series-f-financing-at-30-billion-valuation",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "VAST announced a Series F at a $30B valuation with approximately $1B of primary and secondary transaction value, more than $4B cumulative bookings, and more than $500M committed ARR, while claiming its platform supports environments spanning millions of GPUs.",
      "quantified": {
        "valuation_usd": 30000000000,
        "transaction_value_usd_approx": 1000000000,
        "cumulative_bookings_usd_gt": 4000000000,
        "committed_arr_usd_gt": 500000000,
        "claimed_supported_gpu_scale": "millions_globally"
      },
      "assumptions": [
        "Storage/data-platform economics and support continuity can become strategic at multi-cloud AI-factory scale."
      ],
      "contradictions_or_limits": [
        "Company-reported bookings, CARR, profitability, and GPU-environment claims are not audited here and do not reveal storage traffic, capacity, customer concentration, or deployed goodput.",
        "Funding and valuation are not product throughput or installed storage capacity."
      ],
      "fak_implications": [
        "Track data-platform customer concentration, workload, bytes/IOPS/bandwidth, GPU linkage, support, and profitability independently.",
        "Do not convert supported GPU claims or bookings into storage throughput."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "arista-7060xe7-16t-2026",
      "category": "supply_chain",
      "entity": "Arista Networks",
      "topic": [
        "networking",
        "1_6t",
        "ai_fabric",
        "liquid_cooling",
        "availability"
      ],
      "published_at": "2026-06-09",
      "event_at": "2026-06-09",
      "source_title": "Arista Introduces Next-Generation 1.6Terabit Portfolio for AI Fabrics",
      "source_url": "https://investors.arista.com/Communications/Press-Releases-and-Events/Press-Release-Detail/2026/Arista-Introduces-Next-Generation-1-6Terabit-Portfolio-for-AI-Fabrics/default.aspx",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Arista announced 7060XE7 1.6T Ethernet platforms with about 100 Tbps system bandwidth, air- and liquid-cooled configurations, low-power optics, congestion/load-balancing features, and availability windows from Q4 2026 through Q1 2027.",
      "quantified": {
        "system_bandwidth_tbps_approx": 100,
        "port_speed_tbps": 1.6,
        "lpo_power_reduction_pct_approx": 60,
        "air_cooled_64x16t_availability": "Q4 2026",
        "liquid_cooled_64x16t_availability": "Q1 2027",
        "air_cooled_128x800g_availability": "Q1 2027"
      },
      "assumptions": [
        "AI fabric product availability, cooling, optics, congestion, and retry features evolve as one rack-scale system."
      ],
      "contradictions_or_limits": [
        "Announced availability and specifications do not prove shipment volume, interoperability, field reliability, congestion behavior, or application goodput.",
        "Peak port/system bandwidth ignores protocol, topology, collectives, failures, and workload imbalance."
      ],
      "fak_implications": [
        "Record switch SKU, ports, optics, cooling, NOS, protocol features, availability, shipment, and deployed topology.",
        "Benchmark collective/application goodput and failure recovery, not peak bandwidth."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "arista-xpo-msa-2026",
      "category": "standard",
      "entity": "Arista XPO MSA",
      "topic": [
        "optics",
        "liquid_cooling",
        "msa",
        "rack_density",
        "specification_lifecycle"
      ],
      "published_at": "2026-03-12",
      "event_at": "2026-03-12",
      "source_title": "Arista Announces XPO High Density Liquid Cooled Pluggable Optics",
      "source_url": "https://www.arista.com/en/company/news/press-release/23697-pr-20260311",
      "source_kind": "official_standard",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Arista announced the XPO multi-source agreement for a 64-channel, liquid-cooled pluggable optics module with 12.8 Tbps capacity, 204.8 Tbps per OCP rack unit, and cold-plate cooling up to 400 W per module.",
      "quantified": {
        "module_bandwidth_tbps": 12.8,
        "rack_unit_front_panel_density_tbps": 204.8,
        "density_improvement_vs_1600g_osfp": 4,
        "module_cooling_watts_up_to": 400
      },
      "assumptions": [
        "Optical form factors, thermal integration, density, serviceability, and multi-vendor ecosystems are coupled at AI-fabric scale."
      ],
      "contradictions_or_limits": [
        "An MSA announcement and live demonstrations do not establish production qualification, shipment, interoperability, yield, field reliability, or application goodput.",
        "Density figures do not include switch/rack power, fiber routing, protocol, and topology overhead."
      ],
      "fak_implications": [
        "Track specification, participating vendors, qualification, shipment, deployment, and interoperability separately.",
        "Include optics cooling and serviceability in rack/site constraints."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "weka-neuralmesh-ga-2026",
      "category": "supply_chain",
      "entity": "WEKA",
      "topic": [
        "storage",
        "memory_extension",
        "general_availability",
        "agentic_workload",
        "data_platform"
      ],
      "published_at": "2026-03-16",
      "event_at": "2026-03-16",
      "source_title": "NeuralMesh AI Factory Data Platform for Agentic ROI",
      "source_url": "https://www.weka.io/news/neuralmesh-aidp",
      "source_kind": "official_product_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "WEKA announced general availability of NeuralMesh as an enterprise AI data platform spanning storage and memory-system functions for training, inference, and agentic workloads.",
      "quantified": {},
      "assumptions": [
        "Storage and memory-extension systems are increasingly packaged as one AI data plane for checkpoint, data, and agent-state movement."
      ],
      "contradictions_or_limits": [
        "General availability does not disclose production shipment volume, workload mix, bytes/IOPS/bandwidth, cache hit, failures, or independent application goodput.",
        "Vendor platform scope does not prove that storage or memory is the bottleneck for a given workload."
      ],
      "fak_implications": [
        "Record storage/memory tier, checkpoint and agent-state path, capacity, bandwidth, latency, topology, failure, and workload.",
        "Benchmark end-to-end training/inference/task completion with data movement included."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "samsung-gauss2-internal-deployment-2024",
      "category": "frontier_lab",
      "entity": "Samsung Electronics",
      "topic": [
        "model_architecture",
        "serving",
        "enterprise_workload",
        "adoption"
      ],
      "published_at": "2024-11-21",
      "event_at": "2024-11-21",
      "source_title": "Samsung Electronics Hosts Samsung Developer Conference Korea 2024, Unveils Its Improved Gen AI Model",
      "source_url": "https://news.samsung.com/global/samsung-electronics-hosts-samsung-developer-conference-korea-2024-unveils-its-improved-gen-ai-model",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Samsung said Gauss2 shipped in Compact, Balanced, and Supreme variants; Supreme used mixture-of-experts, while internal Gauss services had reached about 60% of DX software developers and call-center summarization use.",
      "quantified": {
        "model_variants": 3,
        "supported_languages_min": 9,
        "supported_languages_max": 14,
        "claimed_processing_speed_multiple_min": 1.5,
        "claimed_processing_speed_multiple_max": 3,
        "dx_software_developer_usage_share": 0.6,
        "code_i_monthly_usage_growth_multiple": 4
      },
      "assumptions": [
        "A first-party model portfolio spans constrained on-device execution, balanced cloud use, and a larger MoE service tier.",
        "Internal coding, office, and call-center workloads are distinct production demand classes rather than one generic chat distribution."
      ],
      "contradictions_or_limits": [
        "The 1.5-3x speed range is Samsung's comparison with unnamed leading open-source models, not a neutral matched serving benchmark.",
        "Developer share and usage growth do not reveal requests, tokens, concurrency, latency, hardware, or external-customer traffic.",
        "Planned product integration is not evidence of shipped consumer deployment."
      ],
      "fak_implications": [
        "Model workload envelopes should distinguish on-device, employee cloud, coding, and call-center paths.",
        "Treat internal adoption counts as evidence of workload existence, not a request or token distribution."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "skt-ax4-model-and-adot-deployment-2025",
      "category": "frontier_lab",
      "entity": "SK Telecom",
      "topic": [
        "model_release",
        "serving",
        "korean_language",
        "enterprise_workload"
      ],
      "published_at": "2025-07-24",
      "event_at": "2025-07-24",
      "source_title": "SK Telecom Unveils Proprietary Standard Large Language Model A.X 3.1",
      "source_url": "https://news.sktelecom.com/en/2035",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "SK Telecom described a dual model strategy: from-scratch A.X 3.1 and continual-pretrained A.X 4.0, whose standard and light variants were open-sourced and whose model had been applied to the A. call-summarization service since May 2025.",
      "quantified": {
        "claimed_korean_token_efficiency_gain_vs_gpt4o": 0.33,
        "kmmlu_score": 78.3,
        "click_score": 85.7
      },
      "assumptions": [
        "Korean-language tokenization efficiency can materially change capacity and cost relative to English-centric token accounting.",
        "One operator may serve from-scratch sovereign models and adapted open-weight models as separate product tiers."
      ],
      "contradictions_or_limits": [
        "The 33% token-efficiency result and benchmark scores are company-reported tests, not independently reproduced serving measurements.",
        "Application to call summarization establishes a production use case but gives no request volume, latency, hardware, batching, or user distribution.",
        "Open-source availability is not evidence of external adoption."
      ],
      "fak_implications": [
        "Track tokenizer-specific token expansion and model lineage when comparing regional serving cost.",
        "Separate model release, internal service application, and measured production traffic."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "kakao-kanana-family-report-2025",
      "category": "frontier_lab",
      "entity": "Kakao",
      "topic": [
        "model_release",
        "training",
        "open_source",
        "model_family"
      ],
      "published_at": "2025-02-27",
      "event_at": "2025-02-27",
      "source_title": "Kakao publishes Kanana language-model technical report",
      "source_url": "https://www.kakaocorp.com/page/detail/11481",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Kakao published a Kanana technical report spanning models from 2.1B to 32.5B parameters and released the 2.1B-parameter Kanana Nano model as open source.",
      "quantified": {
        "minimum_parameters": 2100000000,
        "maximum_parameters": 32500000000,
        "kanana_nano_parameters": 2100000000,
        "claimed_training_cost_reduction_min": 0.5
      },
      "assumptions": [
        "A product lab can maintain a multi-size model family spanning an on-device-scale open model and larger internally trained models."
      ],
      "contradictions_or_limits": [
        "The greater-than-50% training-cost reduction is Kakao's comparison with similar-size models, not neutral normalized cost accounting.",
        "A technical report and open model release do not establish production deployment, serving traffic, cluster size, or active-user demand."
      ],
      "fak_implications": [
        "Keep model-family release evidence separate from service adoption and workload-distribution evidence.",
        "Use the technical report as the source for training parameters rather than inferring production characteristics from model size."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sensetime-sensenova5-moE-context-2024",
      "category": "frontier_lab",
      "entity": "SenseTime",
      "topic": [
        "model_release",
        "mixture_of_experts",
        "context_length",
        "training_data",
        "deployment"
      ],
      "published_at": "2024-04-25",
      "event_at": "2024-04-23",
      "source_title": "SenseTime releases cloud-device-edge full-stack model matrix and upgrades SenseNova 5.0",
      "source_url": "https://www.sensetime.com/cn/news/51167729/",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "SenseTime said SenseNova 5.0 used a mixture-of-experts architecture, trained on more than 10TB of token data, supported roughly 200K effective inference context, and sat in a cloud-device-edge product matrix with named finance and automotive integrations.",
      "quantified": {
        "training_corpus_tb_tokens_min": 10,
        "effective_context_tokens_approx": 200000,
        "major_version_ordinal": 5
      },
      "assumptions": [
        "A model provider can partition one family across cloud, device, and edge deployment envelopes.",
        "Long-context and multimodal enterprise workloads include cross-document extraction, summarization, finance, and in-vehicle interaction."
      ],
      "contradictions_or_limits": [
        "The source's GPT-4 Turbo comparisons, training-volume wording, and 200K effective-context claim are vendor-reported rather than independently reproduced.",
        "Named integrations do not disclose traffic, hardware, batching, latency, availability, or user distributions.",
        "TB of tokens is the source's unit and is not normalized to an exact token count."
      ],
      "fak_implications": [
        "Preserve effective context separately from maximum advertised context and production prevalence.",
        "Represent cloud, device, and edge execution as distinct operating envelopes."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "xiaomi-mimo-v2-pro-api-2026",
      "category": "frontier_lab",
      "entity": "Xiaomi MiMo",
      "topic": [
        "model_release",
        "mixture_of_experts",
        "context_length",
        "api_pricing",
        "agent_workload"
      ],
      "published_at": "2026-06-29",
      "event_at": "2026-06-29",
      "source_title": "Xiaomi MiMo-V2-Pro: Flagship Foundation Model towards Maximum Reasoning Capability",
      "source_url": "https://mimo.mi.com/docs/en-US/news/previous-news/v2-pro-release",
      "source_kind": "official_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "Xiaomi released MiMo-V2-Pro as a greater-than-1T-total, 42B-active hybrid-attention model with a 1M-token context window and a public API whose price doubled above 256K input context.",
      "quantified": {
        "active_parameters": 42000000000,
        "context_tokens": 1000000,
        "global_to_sliding_attention_ratio": "1:7",
        "api_input_usd_per_million_tokens_up_to_256k": 1,
        "api_output_usd_per_million_tokens_up_to_256k": 3,
        "api_input_usd_per_million_tokens_256k_to_1m": 2,
        "api_output_usd_per_million_tokens_256k_to_1m": 6,
        "total_parameters_min_exclusive": 1000000000000
      },
      "assumptions": [
        "Agentic and coding demand can include contexts far beyond 256K, with a provider pricing a separate long-context tier.",
        "Sparse activation and hybrid attention are used to constrain inference cost at trillion-parameter scale."
      ],
      "contradictions_or_limits": [
        "Architecture, benchmark ranking, and internal-engineer comparisons are Xiaomi claims, not neutral matched measurements.",
        "API availability and price do not establish request mix, context prevalence, latency, batching, or capacity.",
        "The page's most-recent update date is later than the stated model release date."
      ],
      "fak_implications": [
        "Model long-context cost as a tiered workload rather than assuming flat per-token economics.",
        "Keep total parameters, active parameters, attention window, API availability, and production traffic separate."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "nvidia-nemotron3-ultra-training-2026",
      "category": "frontier_lab",
      "entity": "NVIDIA Nemotron",
      "topic": [
        "model_architecture",
        "training",
        "mixture_of_experts",
        "quantization",
        "long_context"
      ],
      "published_at": "2026-06-09",
      "event_at": "2026-06-09",
      "source_title": "Nemotron 3 Ultra: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning",
      "source_url": "https://research.nvidia.com/labs/nemotron/files/NVIDIA-Nemotron-3-Ultra-Technical-Report.pdf",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "NVIDIA documented Nemotron 3 Ultra as a 550B-total, 55B-active hybrid Mamba-attention MoE with 108 layers, 512 experts per MoE layer and top-22 routing, pretrained with NVFP4 on 20T text tokens before a 1M-token context extension.",
      "quantified": {
        "total_parameters": 550000000000,
        "active_parameters": 55000000000,
        "layers": 108,
        "model_dimension": 8192,
        "experts_per_layer": 512,
        "active_experts": 22,
        "mtp_layers": 2,
        "training_ablation_checkpoints_tokens": [
          5000000000000,
          10000000000000,
          16000000000000
        ],
        "bf16_ablation_continuation_tokens": 74000000000,
        "fresh_code_tokens": 173000000000,
        "pretraining_tokens": 20000000000000,
        "context_tokens": 1000000
      },
      "assumptions": [
        "Frontier first-party model development can use low-precision training, sparse expert activation, Mamba-attention hybrids, and multi-token prediction together.",
        "Training-health validation includes precision-switched continuation branches at multiple token checkpoints."
      ],
      "contradictions_or_limits": [
        "The report's benchmark results are author measurements; they do not establish production traffic, serving SLOs, or third-party deployment prevalence.",
        "Checkpoint ablations and fresh-code corpus size are not the same denominator as total pretraining tokens.",
        "Open weights and recipes do not imply independent reproducibility at the original compute scale."
      ],
      "fak_implications": [
        "Preserve active versus total parameters and low-precision training mode in model envelopes.",
        "Use precision-switched checkpoint continuations as a prior-art pattern for training-health witnesses, not as serving evidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mbzuai-k2-nanda-model-program-2025",
      "category": "frontier_lab",
      "entity": "MBZUAI / Inception / Cerebras / Petuum",
      "topic": [
        "model_release",
        "open_source",
        "regional_language",
        "training"
      ],
      "published_at": "2025-01-21",
      "event_at": "2024-12-31",
      "source_title": "Making MBZUAI the Stanford of the Middle East",
      "source_url": "https://mbzuai.ac.ae/news/making-mbzuai-the-stanford-of-the-middle-east/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "MBZUAI's 2024 review said its foundation-model program released NANDA with Inception and Cerebras, launched LLM360 for fully open training artifacts, and collaborated with Petuum on the 65B-parameter K2-65B model.",
      "quantified": {
        "k2_parameters": 65000000000,
        "specialized_models_released": 5
      },
      "assumptions": [
        "Regional-language model programs combine university, product-lab, systems-vendor, and open-research partners.",
        "Some openness programs include training code, data, checkpoints, and intermediate results rather than weights alone."
      ],
      "contradictions_or_limits": [
        "The annual review summarizes releases but does not expose cluster size, training duration, serving traffic, or production adoption.",
        "Partnership and release evidence is not proof that a model is deployed at scale.",
        "The source calls NANDA a leading Hindi model but does not provide a neutral benchmark or all model parameters."
      ],
      "fak_implications": [
        "Track artifact openness dimensions separately: weights, data, code, recipes, checkpoints, and intermediate states.",
        "Do not infer production serving from a research release or partnership."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "sdaia-allam-watsonx-deployment-2024",
      "category": "frontier_lab",
      "entity": "SDAIA / IBM",
      "topic": [
        "model_release",
        "deployment",
        "arabic_language",
        "enterprise_platform"
      ],
      "published_at": "2024-05-21",
      "event_at": "2024-05-21",
      "source_title": "Through partnership with IBM, Saudi Data and Artificial Intelligence Authority launches a groundbreaking Arabic AI model to the Middle East",
      "source_url": "https://mea.newsroom.ibm.com/sdaia-launches-allam-on-watsonx",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "SDAIA and IBM said the Saudi-developed Arabic ALLaM model became operational on watsonx.ai so enterprise and government clients could access, train, tune, and deploy it with platform governance controls.",
      "quantified": {},
      "assumptions": [
        "Sovereign-language models may be distributed through a multinational enterprise AI platform rather than only a national service.",
        "Government and enterprise use requires governance and deployment controls in addition to model weights."
      ],
      "contradictions_or_limits": [
        "Operational availability on watsonx is not a measurement of active customers, requests, tokens, latency, hardware, or successful deployments.",
        "The announcement does not give the served model's parameter count or training-compute envelope.",
        "Open-source with an optional acceptable-use policy is a licensing statement, not evidence of unrestricted use."
      ],
      "fak_implications": [
        "Represent model availability, client access, fine-tuning, and observed production use as separate lifecycle stages.",
        "Regional-language routing may need governance and residency attributes in addition to model capability."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "humain-chat-allam34b-launch-2025",
      "category": "frontier_lab",
      "entity": "HUMAIN / SDAIA",
      "topic": [
        "model_release",
        "consumer_service",
        "arabic_language",
        "deployment"
      ],
      "published_at": "2025-08-25",
      "event_at": "2025-08-25",
      "source_title": "Sovereign AI for Smarter Conversations: HUMAIN Chat",
      "source_url": "https://www.humain.ai/en/news/humain-chat-launch/",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "HUMAIN launched a public chat service powered by the Saudi-hosted ALLAM 34B Arabic-first model, saying model development incorporated more than 600 experts and 250 evaluators.",
      "quantified": {
        "model_parameters": 34000000000,
        "domain_experts_min": 600,
        "evaluators_min": 250,
        "client_surfaces": 3
      },
      "assumptions": [
        "Sovereign deployment can include a national consumer chat surface as well as enterprise model access.",
        "Arabic-first service quality depends on regional dialect, cultural context, and human-evaluation coverage."
      ],
      "contradictions_or_limits": [
        "A public launch proves a reachable product surface, not active-user count, request distribution, service reliability, hardware, or useful goodput.",
        "Claims about model leadership, dataset scale, and cultural fluency are vendor statements unless independently reproduced.",
        "Saudi hosting and operation do not identify the physical cluster or datacenter envelope.",
        "The current public page body is dynamically delivered and its static HTML exposes only title/description metadata; the launch details require browser-rendered or archived-page verification during refresh."
      ],
      "fak_implications": [
        "Separate public product availability from measured adoption and workload distributions.",
        "Preserve language, dialect, residency, and governance as workload-routing dimensions."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "01ai-yi34b-200k-long-context-2024",
      "category": "frontier_lab",
      "entity": "01.AI",
      "topic": [
        "model_release",
        "context_length",
        "long_context_training",
        "open_source"
      ],
      "published_at": "2024-03-08",
      "event_at": "2024-03-07",
      "source_title": "Yi-34B-200K model card",
      "source_url": "https://huggingface.co/01-ai/Yi-34B-200K",
      "source_kind": "official_model_card",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "01.AI's official Yi-34B-200K model card described a 34B-parameter bilingual Transformer with a 200K context window and a long-context enhancement that continued pretraining on a 5B-token mixture.",
      "quantified": {
        "parameters": 34000000000,
        "context_tokens": 200000,
        "long_context_continued_pretraining_tokens": 5000000000,
        "base_training_sequence_tokens": 4000,
        "base_inference_extension_tokens": 32000
      },
      "assumptions": [
        "One model family can expose a much larger long-context variant than its original training sequence through continued pretraining and extension techniques.",
        "Open model release supports independent serving experiments but does not describe provider traffic."
      ],
      "contradictions_or_limits": [
        "The model card's needle-test and benchmark rankings are vendor-reported and time-sensitive.",
        "A 200K maximum context does not reveal production context prevalence, latency, memory footprint, batching, or user demand.",
        "Hugging Face counters and likes are repository activity, not active model users or requests."
      ],
      "fak_implications": [
        "Track original training sequence, inference extension, long-context continued training, and advertised maximum as separate fields.",
        "Use open weights for reproducible envelope tests without treating repository popularity as production adoption."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "stepfun-step3-moe-afd-2025",
      "category": "frontier_lab",
      "entity": "StepFun",
      "topic": [
        "model_architecture",
        "mixture_of_experts",
        "multimodal",
        "serving_efficiency"
      ],
      "published_at": "2025-07-31",
      "event_at": "2025-07-31",
      "source_title": "Step3: Cost-Effective Multimodal Intelligence",
      "source_url": "https://stepfun.ai/research/en/step3",
      "source_kind": "official_technical_release",
      "evidence_class": "vendor_claim",
      "confidence": "medium_high",
      "claim": "StepFun released Step3 as a multimodal MoE with 321B total and 38B active parameters, pairing Multi-Matrix Factorization Attention with Attention-FFN Disaggregation to target lower decoding cost across flagship and lower-end accelerators.",
      "quantified": {
        "total_parameters": 321000000000,
        "active_parameters": 38000000000
      },
      "assumptions": [
        "Attention and expert computation can be assigned to different GPU fleets for heterogeneous serving.",
        "Sparse activation and attention-compute reduction are co-designed with the serving topology rather than treated as model-only changes."
      ],
      "contradictions_or_limits": [
        "The efficiency and accelerator-portability statements are vendor claims; the page does not expose production request traces or neutral matched benchmarks.",
        "Parameter counts and architecture do not identify the training cluster, service capacity, batching policy, latency SLO, or user distribution.",
        "Open release does not establish third-party adoption."
      ],
      "fak_implications": [
        "Evaluate attention/FFN disaggregation as a serving envelope with transfer and scheduling overhead included.",
        "Preserve total and active parameters plus accelerator class when comparing sparse-model serving."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "shanghai-ai-lab-internlm3-2025",
      "category": "frontier_lab",
      "entity": "Shanghai AI Laboratory / InternLM",
      "topic": [
        "model_release",
        "training_data",
        "reasoning",
        "open_source"
      ],
      "published_at": "2025-01-15",
      "event_at": "2025-01-15",
      "source_title": "Official release of InternLM series",
      "source_url": "https://github.com/InternLM/InternLM",
      "source_kind": "official_repository",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Shanghai AI Laboratory released InternLM3-8B-Instruct as an 8B general-purpose and reasoning model and said it was trained on 4T high-quality tokens, while the broader project supplied open model and deployment toolchains.",
      "quantified": {
        "parameters": 8000000000,
        "training_tokens": 4000000000000,
        "claimed_training_cost_reduction_min": 0.75
      },
      "assumptions": [
        "A regional lab may pair model weights with first-party training, evaluation, compression, and deployment toolchains.",
        "Data quality and training process can be traded against raw token volume at a fixed model scale."
      ],
      "contradictions_or_limits": [
        "The greater-than-75% training-cost reduction and model comparisons are project claims, not independently normalized cost accounting.",
        "Repository release does not reveal physical training hardware, production traffic, latency, batching, or user adoption.",
        "Supported deployment examples are capability evidence, not observed production deployments."
      ],
      "fak_implications": [
        "Track the model and its adjacent training/deployment stack without converting supported paths into adoption claims.",
        "Require matched accounting before using claimed training-cost reductions as a fak baseline."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openrouter-100t-geography-2026",
      "category": "workload_trace",
      "entity": "OpenRouter",
      "topic": [
        "production workload",
        "geography",
        "regional demand",
        "language mix"
      ],
      "published_at": "2026-01-15",
      "event_at": "2026-01-15",
      "source_title": "State of AI: An Empirical 100 Trillion Token Study with OpenRouter",
      "source_url": "https://arxiv.org/html/2601.10088v1",
      "source_kind": "paper with provider production data",
      "evidence_class": "measured provider production trace",
      "confidence": "high for the disclosed OpenRouter population; low for extrapolation to all providers",
      "claim": "OpenRouter analyzed more than 100T production tokens across tasks, geographies, and time; weekly spend was globally distributed, with North America below half for most observed weeks, Europe in the mid-teens to low twenties, and Asia rising from about 13% to 31%.",
      "quantified": {
        "tokens_gt": 100000000000000,
        "north_america_weekly_spend_share_max_for_most_periods": 0.5,
        "europe_weekly_spend_share_range_approx": [
          0.15,
          0.22
        ],
        "asia_weekly_spend_share_start_approx": 0.13,
        "asia_weekly_spend_share_end_approx": 0.31,
        "english_token_share_min": 0.8,
        "simplified_chinese_token_share_approx": 0.05
      },
      "assumptions": [
        "Regional spend share is a billing/economic demand measure, not request share, token share, user share, or physical serving-region load.",
        "The source supports time-varying regional benchmark weights rather than one permanent geography mix."
      ],
      "contradictions_or_limits": [
        "The paper does not disclose comparable request, user, tenant, timezone, or serving-region denominators by geography.",
        "OpenRouter traffic is developer- and marketplace-skewed; spend shares depend on model price mix and cannot be treated as raw compute demand.",
        "No confidence intervals or universal geographic distribution are reported."
      ],
      "fak_implications": [
        "Parameterize geography by metric and epoch; do not convert spend share into request or token share.",
        "Keep language, billing geography, request origin, serving region, sovereignty, WAN latency, and model-price mix as separate fields."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "schneider-us-electrical-expansion-2025",
      "category": "supply_chain",
      "entity": "Schneider Electric",
      "topic": [
        "electrical equipment",
        "switchgear",
        "circuit breakers",
        "manufacturing expansion"
      ],
      "published_at": "2025-03-25",
      "event_at": "2025-03-25",
      "source_title": "Schneider Electric Plans to Invest Over $700 million in the U.S., Supporting Energy & AI Sectors and Job Growth",
      "source_url": "https://www.se.com/us/en/about-us/newsroom/news/press-releases/schneider-electric-plans-to-invest-over-700-million-in-the-u-s-supporting-energy-ai-sectors-and-job-growth-67bdeb3ee4475a5955011b6a/",
      "source_kind": "vendor press release",
      "evidence_class": "announced manufacturing investment",
      "confidence": "high for announced plan; low for delivered output",
      "claim": "Schneider Electric announced more than $700M of U.S. investment through 2027 across medium-voltage products, circuit breakers, switchgear, power distribution, and test labs, with more than 1,000 planned jobs.",
      "quantified": {
        "planned_investment_usd_gt": 700000000,
        "planned_jobs_gt": 1000,
        "plan_end_year": 2027,
        "us_investment_decade_total_usd_gt": 1000000000
      },
      "assumptions": [
        "Manufacturing investment expands a potential supply envelope; it is not equipment output or datacenter energization."
      ],
      "contradictions_or_limits": [
        "No unit/MVA output, factory-slot allocation, customer order, shipment, acceptance, or commissioned-MW schedule is disclosed."
      ],
      "fak_implications": [
        "Track each facility from announced investment through construction, production, shipment, site acceptance, and energization."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "eaton-nebraska-switchgear-factory-2026",
      "category": "supply_chain",
      "entity": "Eaton",
      "topic": [
        "switchgear",
        "electrical equipment",
        "manufacturing expansion",
        "data center power"
      ],
      "published_at": "2026-04-08",
      "event_at": "2026-04-08",
      "source_title": "Eaton expands operations in Nebraska with new manufacturing facility to meet increasing switchgear demand driven by AI data center boom",
      "source_url": "https://www.eaton.com/us/en-us/company/news-insights/news-releases/2026/eaton-expands-operations-in-nebraska-with-new-manufacturing-facility.html",
      "source_kind": "vendor press release",
      "evidence_class": "announced factory with production target",
      "confidence": "high for announced plan; medium for dated production target; low for delivered output",
      "claim": "Eaton announced a >$30M, 370,000-square-foot Bellevue, Nebraska facility for air- and gas-insulated medium-voltage switchgear, targeting production in the first half of 2027 and more than 200 jobs.",
      "quantified": {
        "investment_usd_gt": 30000000,
        "facility_square_feet": 370000,
        "planned_jobs_gt": 200,
        "production_target": "first_half_2027",
        "eaton_global_manufacturing_investment_since_2023_usd_gt": 1500000000
      },
      "assumptions": [
        "A production target is later than a factory announcement but earlier than qualified output, shipment, and site acceptance."
      ],
      "contradictions_or_limits": [
        "No annual switchgear units, ratings, yield, customer allocation, order backlog, shipment date, or energized datacenter capacity is disclosed."
      ],
      "fak_implications": [
        "Model medium-voltage switchgear as a dated long-lead dependency rather than fungible inventory."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "pjm-large-load-forecast-verification-2025",
      "category": "policy_regulation",
      "entity": "PJM Interconnection",
      "topic": [
        "large loads",
        "data centers",
        "load forecast",
        "interconnection",
        "grid reliability"
      ],
      "published_at": "2025-10-17",
      "event_at": "2025-10-17",
      "source_title": "PJM Answers to FERC Questions on Large Load Forecasting",
      "source_url": "https://www.pjm.com/-/media/DotCom/documents/ferc/filings/2025/20251017-ad25-8-000.pdf",
      "source_kind": "RTO filing to federal regulator",
      "evidence_class": "operative process disclosure",
      "confidence": "high",
      "claim": "PJM disclosed that it had no RTO-administered load-interconnection process, relied on utilities and load-serving entities for large-load requests, and faced inconsistent validation and duplicate-request handling across jurisdictions.",
      "quantified": {
        "pjm_2024_2030_peak_growth_gw": 32,
        "pjm_2024_2030_data_center_growth_gw_approx": 30,
        "indiana_large_load_threshold_mw": 70,
        "new_jersey_large_load_threshold_mw": 100,
        "ohio_data_center_threshold_mw": 25
      },
      "assumptions": [
        "A forecasted or requested MW is not contracted, constructed, energized, or continuously served MW."
      ],
      "contradictions_or_limits": [
        "Customer-level requests and commercial probabilities initially reside with utilities; duplicate requests are generally not explicitly accounted for unless known within related territories.",
        "The filing documents heterogeneous rules and missing uniform verification rather than a clean interconnection-queue census."
      ],
      "fak_implications": [
        "Carry request, probability adjustment, financial commitment, duplicate-screening, ramp, contract, upgrade, and energization as separate lifecycle fields."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "bis-h200-china-case-review-2026",
      "category": "policy_regulation",
      "entity": "U.S. Bureau of Industry and Security",
      "topic": [
        "export controls",
        "advanced computing",
        "China",
        "H200",
        "MI325X"
      ],
      "published_at": "2026-01-13",
      "event_at": "2026-01-13",
      "source_title": "Department of Commerce Revises License Review Policy for Semiconductors Exported to China",
      "source_url": "https://www.bis.gov/press-release/department-commerce-revises-license-review-policy-semiconductors-exported-china",
      "source_kind": "official regulator press release and final-rule notice",
      "evidence_class": "operative rule change",
      "confidence": "high",
      "claim": "BIS changed license review for Nvidia H200, AMD MI325X, and similar chips exported to approved customers in China from the prior posture to case-by-case review subject to supply, compliance, customer-screening, and U.S. third-party testing requirements.",
      "quantified": {
        "effective_immediately": true,
        "named_chip_families": 2
      },
      "assumptions": [
        "Case-by-case eligibility is not license approval, shipment, installation, schedulable capacity, or useful goodput."
      ],
      "contradictions_or_limits": [
        "Applicants must show exports will not reduce global production capacity available to U.S. customers, purchasers have compliance/customer-screening procedures, and products undergo independent U.S. testing.",
        "The rule does not disclose approved customers, quantities, licenses granted, shipment dates, or installed capacity."
      ],
      "fak_implications": [
        "Version regional hardware envelopes by rule effective date and preserve license, shipment, installation, and runtime evidence separately."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "bis-uae-advanced-computing-2026",
      "category": "policy_regulation",
      "entity": "U.S. Bureau of Industry and Security / UAE",
      "topic": [
        "export controls",
        "advanced computing",
        "UAE",
        "license exception",
        "sovereign AI"
      ],
      "published_at": "2026-07-10",
      "event_at": "2026-07-10",
      "source_title": "Department of Commerce Eases Export Controls for UAE",
      "source_url": "https://www.bis.gov/press-release/department-commerce-eases-export-controls-uae",
      "source_kind": "official regulator press release and EAR change",
      "evidence_class": "operative rule change",
      "confidence": "high",
      "claim": "BIS moved the UAE from EAR Country Groups D:3/D:4 to A:5 and approved the UAE Government and certain companies to receive advanced-computing items, including AI chips and servers, license-free under the bilateral framework.",
      "quantified": {
        "effective_date": "2026-07-10",
        "country_groups_removed": [
          "D:3",
          "D:4"
        ],
        "country_group_added": "A:5"
      },
      "assumptions": [
        "Destination and approved-end-user eligibility is not evidence of a specific order, shipment, installation, healthy cluster, or schedulable goodput."
      ],
      "contradictions_or_limits": [
        "The announcement does not enumerate all approved companies, item quantities, shipment dates, facilities, or live capacity.",
        "License-free treatment remains bounded by the EAR, named eligibility, anti-diversion commitments, and the bilateral framework."
      ],
      "fak_implications": [
        "Version hardware-availability envelopes by destination, end user, item, license/exception, effective date, and actual delivery receipts."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "graphcore-softbank-acquisition-2024",
      "category": "market_signal",
      "entity": "Graphcore / SoftBank",
      "topic": [
        "acquisition",
        "alternative accelerator",
        "AI compute",
        "startup lifecycle"
      ],
      "published_at": "2024-07-11",
      "event_at": "2024-07-11",
      "source_title": "Graphcore joins SoftBank Group to build next generation of AI compute",
      "source_url": "https://www.graphcore.ai/posts/graphcore-joins-softbank-group-to-build-next-generation-of-ai-compute",
      "source_kind": "company acquisition announcement",
      "evidence_class": "completed acquisition",
      "confidence": "high",
      "claim": "Graphcore announced that SoftBank had acquired the company; Graphcore became a wholly owned subsidiary and continued under its existing name and UK headquarters.",
      "quantified": {},
      "assumptions": [
        "A completed acquisition is a company-lifecycle outcome, not evidence that IPU hardware is deployed, competitive, or schedulable for frontier workloads."
      ],
      "contradictions_or_limits": [
        "Purchase price and pre-acquisition financial condition were not disclosed in the company announcement.",
        "Post-acquisition hiring and roadmap claims do not establish shipped systems or useful goodput."
      ],
      "fak_implications": [
        "Track accelerator vendors through independence, strategic acquisition, retained roadmap, shipped systems, and engine/runtime evidence."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "ampere-softbank-completed-2025",
      "category": "market_signal",
      "entity": "Ampere Computing / SoftBank",
      "topic": [
        "acquisition",
        "CPU",
        "AI compute",
        "startup lifecycle",
        "financials"
      ],
      "published_at": "2025-11-26",
      "event_at": "2025-11-25",
      "source_title": "Completion of Acquisition of Ampere Computing Holdings LLC",
      "source_url": "https://group.softbank/en/news/press/20251126",
      "source_kind": "acquirer completion notice and transaction filing",
      "evidence_class": "completed acquisition with disclosed financial history",
      "confidence": "high",
      "claim": "SoftBank completed its $6.5B acquisition of Ampere, making it a wholly owned subsidiary; the original filing disclosed revenue falling from $151.8M in 2022 to $16.5M in 2024 alongside three years of operating losses.",
      "quantified": {
        "acquisition_price_usd": 6500000000,
        "revenue_2022_usd": 151822000,
        "revenue_2023_usd": 46704000,
        "revenue_2024_usd": 16460000,
        "operating_loss_2022_usd": 518290000,
        "operating_loss_2023_usd": 714689000,
        "operating_loss_2024_usd": 510623000,
        "net_assets_2024_usd": -1513315000
      },
      "assumptions": [
        "Acquisition value reflects strategic control and option value; it is not a neutral measure of product revenue, installed CPUs, or AI goodput."
      ],
      "contradictions_or_limits": [
        "The completion notice does not disclose customer concentration, unit shipments, roadmap delivery, or AI-workload share.",
        "Historic financials are company-wide accounting results, not product benchmark evidence."
      ],
      "fak_implications": [
        "Preserve financial distress, strategic acquisition, architecture roadmap, shipment, and workload fitness as separate dimensions."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "octoai-nvidia-acquisition-2024",
      "category": "market_signal",
      "entity": "OctoAI / NVIDIA",
      "topic": [
        "acquisition",
        "inference platform",
        "serving software",
        "startup lifecycle"
      ],
      "published_at": "2025-05-22",
      "event_at": "2024-09-30",
      "source_title": "Blackwell Breaks the 1,000 TPS/User Barrier With Meta’s Llama 4 Maverick",
      "source_url": "https://developer.nvidia.com/blog/blackwell-breaks-the-1000-tps-user-barrier-with-metas-llama-4-maverick/",
      "source_kind": "acquirer technical publication with personnel provenance",
      "evidence_class": "post-acquisition integration evidence",
      "confidence": "medium-high",
      "claim": "NVIDIA identifies former OctoAI product staff as having joined through NVIDIA’s 2024 acquisition of OctoAI; later NVIDIA material also identifies FlashInfer as originating partly at OctoAI and continuing in NVIDIA’s serving stack.",
      "quantified": {},
      "assumptions": [
        "Personnel and open-source technology integration are post-acquisition evidence, not proof that the former hosted OctoAI service continued unchanged."
      ],
      "contradictions_or_limits": [
        "NVIDIA did not publish transaction terms or a standalone acquisition announcement in this source.",
        "The source establishes team/technology provenance but not OctoAI customer migration, service shutdown dates, revenue, or retained contracts."
      ],
      "fak_implications": [
        "Track whether acquired serving technology persists as a product, library, team, or kernel contribution rather than treating the startup as simply active or failed."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "agentsysbench-2026",
      "category": "workload_trace",
      "entity": "AgentSysBench",
      "topic": [
        "agentic workload",
        "workflow systems",
        "sandbox state",
        "tool caching",
        "production trace"
      ],
      "published_at": "2026-08-20",
      "event_at": "2026-08-20",
      "source_title": "AgentSysBench: A Workload Characterization and Benchmark Suite for System-Level Evaluation of Agentic AI Applications",
      "source_url": "https://arxiv.org/abs/2608.15127",
      "source_kind": "benchmark paper with controlled experiments and production trace sidecar",
      "evidence_class": "measured system workload plus production trace analysis",
      "confidence": "high within the evaluated applications and trace; medium for broader agent populations",
      "claim": "AgentSysBench profiles ten agentic applications and finds stateful non-LLM work dominates latency in half, sandbox working sets reach 28 GB per session, task latency differs by up to 32x, production state sits idle for minutes to hours, and redundant tool calls create measurable cache opportunity.",
      "quantified": {
        "applications": 10,
        "applications_non_llm_latency_dominant": 5,
        "sandbox_working_set_gb_peak_per_session": 28,
        "intra_application_task_latency_divergence_x_max": 32,
        "task_disaggregated_latency_reduction_fraction_range": [
          0.29,
          0.4
        ],
        "communication_aware_speedup_x_max": 4.5,
        "state_offload_memory_reduction_x": 4.6,
        "tool_cache_ttl_minutes": 10,
        "redundant_search_calls_eliminated_fraction": 0.352,
        "aggregate_search_latency_saved_fraction": 0.193
      },
      "assumptions": [
        "Agent execution must be modeled as a component DAG with heterogeneous GPU, CPU, memory, network, sandbox, retrieval, and external-tool stages.",
        "Production idle state lasting minutes to hours makes request-scoped teardown an inaccurate default."
      ],
      "contradictions_or_limits": [
        "Ten applications are a benchmark suite, not a census of all agents; the production trace sidecar does not disclose universal session, tenant, geography, or retry distributions.",
        "Reported optimization gains depend on the evaluated application mix, task graph, load levels, resource pool, cache TTL, and tool behavior."
      ],
      "fak_implications": [
        "Benchmark end-to-end useful completion with non-LLM latency, sandbox memory, idle residency, state transfer, retries, and tool calls included.",
        "Schedule by task/component affinity rather than treating every agent step as one homogeneous LLM request."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "orca-iteration-scheduling-2022",
      "category": "serving_system",
      "entity": "Orca",
      "topic": [
        "iteration-level scheduling",
        "selective batching",
        "distributed inference",
        "queueing"
      ],
      "published_at": "2022-07-11",
      "event_at": "2022-07-11",
      "source_title": "Orca: A Distributed Serving System for Transformer-Based Generative Models",
      "source_url": "https://www.usenix.org/conference/osdi22/presentation/yu",
      "source_kind": "peer-reviewed systems paper",
      "evidence_class": "controlled benchmark",
      "confidence": "high within evaluated hardware/models/workloads",
      "claim": "Orca replaces request-level static batching with iteration-level scheduling and selective batching so completed requests leave and new requests join between decoding iterations; it reports up to 36.9x throughput over FasterTransformer at the same latency on its largest evaluated model.",
      "quantified": {
        "reported_throughput_improvement_x_up_to": 36.9,
        "largest_evaluated_model_parameters_billion": 175
      },
      "assumptions": [
        "Iteration-level admission helps when request lengths differ and decoding iterations expose regular scheduling points."
      ],
      "contradictions_or_limits": [
        "The maximum is against a 2022 FasterTransformer baseline and is not a modern vLLM/SGLang/Dynamo comparison.",
        "Gain depends on model, tensor/pipeline parallelism, arrival rate, sequence lengths, latency target, and batching implementation."
      ],
      "fak_implications": [
        "Carry admission cadence, active sequence count, batch composition, completed-sequence removal, and per-iteration overhead in serving receipts."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "vllm-pagedattention-2023",
      "category": "serving_system",
      "entity": "vLLM / PagedAttention",
      "topic": [
        "KV cache",
        "memory management",
        "continuous batching",
        "serving throughput"
      ],
      "published_at": "2023-10-24",
      "event_at": "2023-10-24",
      "source_title": "Efficient Memory Management for Large Language Model Serving with PagedAttention",
      "source_url": "https://arxiv.org/abs/2309.06180",
      "source_kind": "peer-reviewed systems paper",
      "evidence_class": "controlled benchmark with production workload trace",
      "confidence": "high within evaluated models/hardware/traces",
      "claim": "vLLM introduces PagedAttention and block-based KV memory management to reduce fragmentation and enable flexible KV sharing, reporting 2-4x throughput over FasterTransformer and Orca at comparable latency.",
      "quantified": {
        "reported_throughput_improvement_x_range": [
          2,
          4
        ],
        "evaluated_gpu_types": [
          "NVIDIA A100 40GB",
          "NVIDIA A10G 24GB"
        ]
      },
      "assumptions": [
        "KV paging helps when variable sequence lengths and memory fragmentation constrain batch concurrency."
      ],
      "contradictions_or_limits": [
        "The reported range is against 2023 implementations and depends on block size, model, GPU memory, workload trace, sampling, and latency envelope.",
        "Paged KV memory does not remove compute, transfer, admission fairness, or multi-tenant responsibility constraints."
      ],
      "fak_implications": [
        "Track useful KV bytes, reserved bytes, fragmentation, block-table overhead, sharing, eviction, and batch concurrency separately."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "sarathi-chunked-prefill-2024",
      "category": "serving_system",
      "entity": "Sarathi-Serve",
      "topic": [
        "chunked prefill",
        "stall-free batching",
        "goodput",
        "prefill decode interference"
      ],
      "published_at": "2024-07-10",
      "event_at": "2024-07-10",
      "source_title": "Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve",
      "source_url": "https://www.usenix.org/conference/osdi24/presentation/agrawal",
      "source_kind": "peer-reviewed systems paper",
      "evidence_class": "controlled benchmark",
      "confidence": "high within evaluated models/hardware/SLOs",
      "claim": "Sarathi-Serve uses chunked prefills and stall-free batching to reduce prefill/decode interference, reporting up to 2.6x higher serving capacity than vLLM on a single A100 for Mistral-7B and up to 6.3x higher capacity than Orca for Falcon-180B on 64 A100 GPUs.",
      "quantified": {
        "mistral7b_single_a100_capacity_improvement_vs_vllm_x_up_to": 2.6,
        "falcon180b_64xa100_capacity_improvement_vs_orca_x_up_to": 6.3,
        "falcon180b_64xa100_capacity_improvement_vs_vllm_x_up_to": 4.3
      },
      "assumptions": [
        "Chunk size trades TTFT against decode stalls and must be selected under an explicit workload and SLO."
      ],
      "contradictions_or_limits": [
        "Capacity gains span different model/cluster envelopes and must not be merged into one universal factor.",
        "The prototype and baselines are versioned research implementations; transfer, scheduler, and fairness overheads remain workload-specific."
      ],
      "fak_implications": [
        "Benchmark chunk-size sweeps with TTFT, TPOT, throughput, fairness, and prefill/decode interference under the same trace."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "distserve-pd-disaggregation-2024",
      "category": "serving_system",
      "entity": "DistServe",
      "topic": [
        "prefill decode disaggregation",
        "goodput",
        "resource allocation",
        "SLO"
      ],
      "published_at": "2024-07-10",
      "event_at": "2024-07-10",
      "source_title": "DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving",
      "source_url": "https://www.usenix.org/conference/osdi24/presentation/zhong-yinmin",
      "source_kind": "peer-reviewed systems paper",
      "evidence_class": "controlled benchmark",
      "confidence": "high within evaluated models/hardware/workloads/SLOs",
      "claim": "DistServe places prefill and decode on separate GPU groups and optimizes parallelism and allocation independently; across its evaluated workloads it serves up to 7.4x more requests or meets 12.6x tighter SLOs than the compared systems.",
      "quantified": {
        "reported_request_rate_improvement_x_up_to": 7.4,
        "reported_slo_tightening_x_up_to": 12.6
      },
      "assumptions": [
        "Disaggregation helps when prefill/decode interference and independent scaling outweigh KV-transfer and stranded-capacity costs."
      ],
      "contradictions_or_limits": [
        "The paper reports benchmark maxima, not universal break-even thresholds; results depend on input/output geometry, TTFT/TPOT targets, model, GPU topology, network, and allocation granularity.",
        "Later hybrid-PD evidence shows aggregation or hybrid placement can win other SLO/workload envelopes."
      ],
      "fak_implications": [
        "Measure KV-transfer bytes/time, topology, prefill/decode utilization, stranded capacity, queueing, SLO attainment, and end-to-end goodput before selecting disaggregation."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "modine-us-cooling-expansion-franklin-2025",
      "category": "supply_chain",
      "entity": "Modine / Airedale",
      "topic": [
        "data center cooling",
        "manufacturing expansion",
        "factory opening",
        "lifecycle"
      ],
      "published_at": "2025-11-17",
      "event_at": "2025-11-17",
      "source_title": "Modine Expands Data Center Cooling Capacity with Opening of New Facility in Franklin, Wisconsin",
      "source_url": "https://www.prnewswire.com/news-releases/modine-expands-data-center-cooling-capacity-with-opening-of-new-facility-in-franklin-wisconsin-302614862.html",
      "source_kind": "company-authored release distributed by PR Newswire",
      "evidence_class": "opened manufacturing facility",
      "confidence": "high for opened facility and announced staffing; low for output delivered to customer sites",
      "claim": "Modine opened a 155,000-square-foot Franklin, Wisconsin manufacturing facility within its $100M four-site U.S. Airedale cooling expansion, targeting more than 300 new jobs by March 2026 and about 430 employees within three years.",
      "quantified": {
        "facility_square_feet": 155000,
        "multi_site_investment_usd": 100000000,
        "jobs_target_by_march_2026_gt": 300,
        "employees_expected_within_three_years_approx": 430,
        "us_sites_in_expansion": 4
      },
      "assumptions": [
        "An opened factory is stronger lifecycle evidence than an announced investment but remains upstream of qualified output, customer shipment, site acceptance, commissioning, and live IT load."
      ],
      "contradictions_or_limits": [
        "No annual unit, tonnage, thermal-MW, yield, backlog allocation, shipment, customer-site acceptance, or commissioned cooling capacity is disclosed.",
        "Job targets and floor area are not cooling throughput."
      ],
      "fak_implications": [
        "Track cooling factories from announcement through opening, staffing, qualified production, shipment, acceptance, commissioning, and supported live IT MW."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "modine-airedale-4b-capacity-lta-2026",
      "category": "market_signal",
      "entity": "Modine / Airedale",
      "topic": [
        "cooling backlog",
        "capacity agreement",
        "customer concentration",
        "manufacturing expansion"
      ],
      "published_at": "2026-05-26",
      "event_at": "2026-05-26",
      "source_title": "Modine Announces Landmark $4 Billion Long-Term Capacity Agreement through 2029 with Strategic Data Center Customer for Airedale by Modine Cooling Solutions",
      "source_url": "https://www.prnewswire.com/news-releases/modine-announces-landmark-4-billion-long-term-capacity-agreement-through-2029-with-strategic-data-center-customer-for-airedale-by-modine-cooling-solutions-302779610.html",
      "source_kind": "company-authored release distributed by PR Newswire",
      "evidence_class": "contracted supply-capacity reservation",
      "confidence": "high for disclosed agreement; low for future product delivery",
      "claim": "Modine agreed to reserve capacity for more than $4B of Airedale cooling products for one strategic data-center customer during 2027-2029 and received $165M upfront to support capacity investment and related expenditures.",
      "quantified": {
        "reserved_product_value_usd_gt": 4000000000,
        "delivery_start_year": 2027,
        "delivery_end_year": 2029,
        "upfront_customer_cash_usd": 165000000,
        "named_customers": 0
      },
      "assumptions": [
        "A long-term capacity agreement and upfront payment are stronger than non-binding demand but remain earlier than manufactured, shipped, accepted, commissioned, or operating cooling capacity."
      ],
      "contradictions_or_limits": [
        "The customer, product mix, annual schedule, unit/thermal-MW quantities, cancellation/shortfall terms, factory allocation, and site destinations are undisclosed.",
        "Contract value is price-based and cannot be converted directly into cooling MW or live IT MW."
      ],
      "fak_implications": [
        "Carry customer concentration, reserved factory capacity, upfront funding, delivery schedule, product mix, shipment, site acceptance, and commissioned thermal load separately."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "schneider-prefab-pod-shipping-2025",
      "category": "supply_chain",
      "entity": "Schneider Electric",
      "topic": [
        "prefabricated modular data center",
        "liquid cooling",
        "shipping product",
        "high-density racks"
      ],
      "published_at": "2025-11-06",
      "event_at": "2025-11-06",
      "source_title": "Schneider Electric Launches New Data Center Solutions to Meet Challenges of High-Density AI and Accelerated Compute Applications",
      "source_url": "https://www.se.com/ww/en/about-us/newsroom/news/press-releases/schneider-electric-launches-new-data-center-solutions-to-meet-challenges-of-high-density-ai-and-accelerated-compute-applications-68432e8baaaf82b041044f06/",
      "source_kind": "vendor product and shipping announcement",
      "evidence_class": "shipping modular product",
      "confidence": "high for product availability; low for installed fleet",
      "claim": "Schneider Electric said its prefabricated EcoStruxure pod data center was shipping pre-designed and pre-assembled, supporting high-density pods rated to 1MW and above with liquid cooling, busway, cabling, containment, InRow, or rear-door heat-exchanger options.",
      "quantified": {
        "pod_supported_power_mw_min": 1
      },
      "assumptions": [
        "Shipping product availability is later than a reference design but does not establish customer delivery, site commissioning, accepted thermal performance, or live IT MW."
      ],
      "contradictions_or_limits": [
        "No shipped-unit count, customer, factory output, site, acceptance date, commissioning receipt, density distribution, or live load is disclosed.",
        "The 1MW+ rating is a supported design envelope, not a measured operating population."
      ],
      "fak_implications": [
        "Separate reference design, order, factory integration, shipment, site assembly, commissioning, and accepted/live capacity for modular data halls."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "burstgpt-azure-openai-2026",
      "category": "workload_trace",
      "entity": "Azure OpenAI / BurstGPT",
      "topic": [
        "production_workload_trace",
        "arrival_periodicity",
        "burstiness",
        "token_lengths",
        "failure_classes"
      ],
      "published_at": "2026-08-25",
      "event_at": "2026-08-25",
      "source_title": "BurstGPT: A Real-World Workload Dataset to Optimize LLM Serving Systems",
      "source_url": "https://arxiv.org/abs/2401.17644v1",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "The original BurstGPT v1 paper and released Azure OpenAI traces expose production request timestamps/aggregate arrivals and per-request input/output token counts for two separate products. They characterize burstiness and empirical token/request patterns but do not fit a named Zipf, power-law, Pareto, lognormal, Poisson, Hawkes, or MMPP family to the production trace.",
      "quantified": "Azure OpenAI GPT services are represented by a 121-day trace, and a separate research API by a 36-day trace. The released request-level fields permit empirical interarrival derivation within each trace and preserve input/output token counts. No production family parameter, goodness-of-fit statistic, or rejection test is reported in arXiv v1 (2024-01-31).",
      "assumptions": "Keep the 121-day and 36-day populations separate, preserve request-level granularity, and derive interarrivals only from the released timestamps while recording any filtering or aggregation. Treat this record as the original v1/2024 trace and paper, not later BurstGPT v1.1 analysis or generator material.",
      "contradictions_or_limits": "Aggregate arrivals are not per-user/client arrivals, token histograms are not fitted distribution families, and either finite window is insufficient to prove stationarity. Later BurstGPT versions/material add synthetic Zipf request-length sampling; that is generator configuration, not a production fit, and is outside this v1 record. No untested family is rejected by the absence of a reported fit.",
      "fak_implications": "Replay the two official traces separately and report burst metrics plus empirical input/output-token distributions. Any Zipf, Poisson, Hawkes, MMPP, Pareto, or lognormal scenario must be labeled synthetic or independently fitted; do not attribute later v1.1 generator assumptions to the original Azure production trace.",
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "splitwise-azure-traces-2023",
      "category": "workload_trace",
      "entity": "Microsoft Azure / Splitwise",
      "topic": [
        "production_workload_trace",
        "conversation_service",
        "coding_service",
        "token_distributions",
        "request_rates"
      ],
      "published_at": "2024-03-13",
      "event_at": "2023-11-11",
      "source_title": "Splitwise: Efficient Generative LLM Inference Using Phase Splitting",
      "source_url": "https://arxiv.org/abs/2311.18677",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "Splitwise characterizes one day from two production Azure LLM services—Conversation, similar to ChatGPT, and Coding, similar to GPT-4—with a few thousand requests each and empirical input-token, output-token, and request-rate distributions, without disclosing model weights.",
      "quantified": {
        "production_services": 2,
        "trace_days": 1,
        "requests_per_service": "a few thousand",
        "conversation_service_comparison": "similar to ChatGPT",
        "coding_service_comparison": "similar to GPT-4"
      },
      "assumptions": [
        "Conversation and coding traffic should be replayed as separate empirical workload classes with their own input/output token distributions and rates.",
        "Phase-splitting conclusions must retain the profiled model, hardware, topology, and simulated scheduling envelope."
      ],
      "contradictions_or_limits": [
        "The traces cover one day and only a few thousand requests per service.",
        "They do not establish universal tenant, geography, seasonality, or session distributions.",
        "Simulation and profiling results are not proof of a deployed production cluster, and the model weights are not disclosed."
      ],
      "fak_implications": [
        "Keep conversation and coding traces distinct when evaluating prefill/decode balance and phase placement.",
        "Require deployed-cluster receipts before treating a simulation or profile-based gain as production goodput."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "huawei-atlas-900-a3-superpod-2025",
      "category": "hardware_supply",
      "entity": "Huawei Atlas 900 A3 SuperPoD",
      "topic": [
        "ascend_accelerator_pod",
        "scale_up_topology",
        "hbm_capacity",
        "memory_bandwidth",
        "reference_specification"
      ],
      "published_at": "2025-04-28",
      "event_at": "2025-04-28",
      "source_title": "Huawei Atlas 900 A3 SuperPoD",
      "source_url": "https://e.huawei.com/cn/products/computing/ascend/atlas-900-a3-superpod",
      "source_kind": "vendor_product_page",
      "evidence_class": "vendor_specification",
      "confidence": "medium",
      "claim": "Huawei specifies an Atlas 900 A3 SuperPoD reference topology with 384 Ascend NPUs across 96 cabinets, 300 TB of HBM, 48 PFLOPS dense BF16, 16.1 PB/s aggregate HBM bandwidth, and 7.2 PB/s scale-up bandwidth.",
      "quantified": {
        "ascend_npus": 384,
        "hbm_tb": 300,
        "dense_bf16_pflops": 48,
        "hbm_bandwidth_pb_per_s": 16.1,
        "scale_up_bandwidth_pb_per_s": 7.2,
        "cabinets": 96
      },
      "assumptions": [
        "Ascend hardware should be represented as a topology and memory/fabric envelope, not converted into an NVIDIA-equivalent chip count.",
        "Reference specifications are upstream of installed, healthy, schedulable, and useful-goodput-producing capacity."
      ],
      "contradictions_or_limits": [
        "This is a vendor reference topology and specification, not a neutral benchmark or proof of a built, installed, healthy, schedulable production system.",
        "The source does not disclose power or cooling requirements."
      ],
      "fak_implications": [
        "Add Ascend-native hardware/topology envelopes to scheduling and communication benchmark matrices.",
        "Require deployment lifecycle, power/cooling, software, health, and quality-constrained goodput receipts before counting production capacity."
      ],
      "rumor": {
        "is_rumor": false,
        "status": "not_applicable"
      }
    },
    {
      "id": "baichuan2-training-2023",
      "category": "frontier_lab",
      "entity": "Baichuan 2",
      "topic": [
        "training_cluster",
        "training_tokens",
        "model_scale"
      ],
      "published_at": "2023-09-19",
      "event_at": "2023-09-19",
      "source_title": "Baichuan 2: Open Large-scale Language Models",
      "source_url": "https://arxiv.org/abs/2309.10305",
      "source_kind": "research_paper",
      "evidence_class": "model_training_report",
      "confidence": "high",
      "claim": "Baichuan 2 reports open 7B and 13B base/chat models trained on 2.6T tokens with 4,096-token maximum sequences; the 7B run used 1,200 NVIDIA A800 GPUs and the 13B run used 1,024 A800 GPUs.",
      "quantified": {
        "model_parameters_billion": [
          7,
          13
        ],
        "training_tokens_trillion": 2.6,
        "maximum_sequence_tokens": 4096,
        "seven_b_training_a800_gpus": 1200,
        "thirteen_b_training_a800_gpus": 1024
      },
      "assumptions": [
        "The disclosed GPU counts describe two model-specific training runs, not one additive fleet."
      ],
      "contradictions_or_limits": [
        "Author-reported training envelope, not production serving traffic, installed fleet capacity, utilization, or user distribution.",
        "The two GPU counts are run-specific and must not be summed into deployed capacity."
      ],
      "fak_implications": [
        "Training-cluster records should bind accelerator counts to a named model, run, token budget, and sequence envelope."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "iflytek-open-platform-2025",
      "category": "frontier_lab",
      "entity": "iFLYTEK Spark / Open Platform",
      "topic": [
        "developer_adoption",
        "api_growth",
        "agent_ecosystem"
      ],
      "published_at": "2026-03-28",
      "event_at": "2025-06-30",
      "source_title": "iFLYTEK 2025 Securities Offering Prospectus",
      "source_url": "https://static.cninfo.com.cn/finalpage/2026-03-28/1225045471.PDF",
      "source_kind": "regulatory_filing",
      "evidence_class": "company_operating_disclosure",
      "confidence": "high",
      "claim": "As of June 2025, iFLYTEK disclosed more than 8.7M AI developer teams, more than 3.42M production-grade applications, 1.52M large-model developers, 4.3x year-to-date growth in average daily large-model API calls, and 85% year-to-date growth in agent count.",
      "quantified": {
        "ai_developer_teams_min": 8700000,
        "production_applications_min": 3420000,
        "large_model_developers": 1520000,
        "large_model_api_average_daily_calls_growth_x": 4.3,
        "agent_count_growth_fraction": 0.85
      },
      "assumptions": [
        "The filing reports ecosystem/adoption denominators, not an inference workload trace."
      ],
      "contradictions_or_limits": [
        "Developer teams, applications, developers, API-call growth, and agent count are different denominators and cannot be substituted for each other.",
        "No absolute API calls, DAU/MAU, token volume, geography, session distribution, hardware count, or neutral performance is disclosed."
      ],
      "fak_implications": [
        "Adoption ledgers should type teams, apps, developers, agents, and API growth separately from users, requests, and tokens."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meituan-longcat-flash-2025",
      "category": "frontier_lab",
      "entity": "Meituan LongCat-Flash",
      "topic": [
        "mixture_of_experts",
        "training_cluster",
        "inference_throughput"
      ],
      "published_at": "2025-09-01",
      "event_at": "2025-09-01",
      "source_title": "LongCat-Flash Technical Report",
      "source_url": "https://arxiv.org/abs/2509.01322",
      "source_kind": "research_paper",
      "evidence_class": "model_training_report",
      "confidence": "high",
      "claim": "Meituan reports a 560B-total-parameter MoE activating 18.6B–31.3B parameters by context, pretrained over 20T tokens with 2.5D pipeline parallelism over 4,096 H800 GPUs; 15,999 chips accumulated 7.5M GPU-hours, with an estimated $6M training cost.",
      "quantified": {
        "total_parameters_billion": 560,
        "activated_parameters_billion_min": 18.6,
        "activated_parameters_billion_max": 31.3,
        "pretraining_tokens_trillion_min": 20,
        "pipeline_parallel_h800_gpus": 4096,
        "training_chips": 15999,
        "training_gpu_hours": 7500000,
        "estimated_training_cost_usd": 6000000,
        "single_h800_tokens_per_second_min": 9600,
        "h800_batch_32_total_tokens_per_second_approx": 100000,
        "throughput_batch_size": 32
      },
      "assumptions": [
        "Dollar cost uses the paper’s then-prevailing H800 rental price rather than an invoice."
      ],
      "contradictions_or_limits": [
        "Author-reported model/run and benchmark envelope; GPU and chip counts have source-specific scopes and are not additive fleet capacity.",
        "Batch-32 throughput is not interactive latency or production prevalence, and the cost is an estimate rather than audited spend."
      ],
      "fak_implications": [
        "MoE capacity models should track total versus activated parameters, batch size, accelerator generation, and estimated versus paid cost."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meituan-dora-production-2026",
      "category": "serving_system",
      "entity": "Meituan DORA",
      "topic": [
        "reinforcement_learning",
        "asynchronous_rollout",
        "kv_cache_migration"
      ],
      "published_at": "2026-04-29",
      "event_at": "2026-04-29",
      "source_title": "DORA: A Scalable Asynchronous Reinforcement Learning System for Language Model Training",
      "source_url": "https://arxiv.org/abs/2604.26256",
      "source_kind": "research_paper",
      "evidence_class": "production_deployment",
      "confidence": "high",
      "claim": "DORA combines multi-version streaming training, centralized load-balancing orchestration, and direct same-version KV-cache migration; it reports up to 2.12x end-to-end and 8.2x rollout-stage speedup on open benchmarks, plus up to 6.2x rollout speedup in production on thousands of accelerators while training an approximately 500B-parameter MoE.",
      "quantified": {
        "benchmark_end_to_end_speedup_max": 2.12,
        "benchmark_rollout_speedup_max": 8.2,
        "production_rollout_speedup_max": 6.2,
        "production_accelerators_min": 1000,
        "production_model_parameters_billion_approx": 500
      },
      "assumptions": [
        "Production accelerator count is disclosed only as “thousands,” so the numeric floor records the weakest literal bound."
      ],
      "contradictions_or_limits": [
        "Reported maxima are bounded and non-stackable.",
        "The production disclosure omits accelerator type, exact count, cluster topology, utilization, availability, and neutral reproduction."
      ],
      "fak_implications": [
        "RL capacity models need trajectory-tail, policy-version, migration, and stage-specific throughput accounting rather than synchronous batch assumptions alone."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meituan-mtserve-2026",
      "category": "serving_system",
      "entity": "Meituan MTServe",
      "topic": [
        "generative_recommendation",
        "hierarchical_cache",
        "stateful_serving"
      ],
      "published_at": "2026-04-24",
      "event_at": "2026-04-24",
      "source_title": "MTServe: Efficient Serving for Generative Recommendation Models with Hierarchical Caches",
      "source_url": "https://arxiv.org/abs/2604.22881",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "MTServe targets generative recommendation requests with unique user/item embeddings that break ordinary shared-prefix KV reuse; request- and prefix-level hierarchical caches evaluated on real-world production datasets delivered up to 4.93x throughput and up to 4.19x lower latency versus cited state-of-the-art baselines.",
      "quantified": {
        "throughput_speedup_max": 4.93,
        "latency_reduction_factor_max": 4.19,
        "cache_levels": 2
      },
      "assumptions": [
        "Per-request user/item state changes prefix-equivalence and cache-reuse geometry."
      ],
      "contradictions_or_limits": [
        "The source withholds absolute trace population/rate, cache-hit distributions, hardware topology, and deployment prevalence.",
        "Maxima are not universal, and per-user/item state is not evidence for a Zipf law."
      ],
      "fak_implications": [
        "Cache routing must distinguish request-level state reuse from token-prefix reuse and key locality by typed user/item state."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meituan-mtgenrec-2025",
      "category": "serving_system",
      "entity": "Meituan MTGenRec",
      "topic": [
        "generative_recommendation",
        "distributed_training",
        "sequence_balancing"
      ],
      "published_at": "2025-05-19",
      "event_at": "2025-05-19",
      "source_title": "MTGenRec: An Efficient Distributed Training System for Generative Recommendation Models in Meituan",
      "source_url": "https://arxiv.org/abs/2505.12663",
      "source_kind": "research_paper",
      "evidence_class": "production_measurement",
      "confidence": "high",
      "claim": "MTGenRec evaluated generative-recommendation training on 200M real user sequences produced over one week at Meituan and 128 A100 GPUs; it reports near-linear scaling, 2.44x lower training time than TorchRec, 1.75x throughput from dynamic sequence balancing, and 53% throughput improvement from feature-ID deduplication.",
      "quantified": {
        "real_user_sequences": 200000000,
        "trace_days": 7,
        "evaluation_a100_gpus": 128,
        "training_time_reduction_factor": 2.44,
        "sequence_balancing_throughput_speedup": 1.75,
        "feature_id_deduplication_throughput_improvement_fraction": 0.53
      },
      "assumptions": [
        "The 200M rows are user sequences used for training, not 200M unique users or sessions."
      ],
      "contradictions_or_limits": [
        "One-week author/internal benchmark; user sequences are not unique users, sessions, or live requests.",
        "The source does not establish a universal sequence-length or popularity family or a live serving rate."
      ],
      "fak_implications": [
        "Recommendation training planners should account for sequence-length imbalance and duplicate feature communication separately from request traffic."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "builder-ai-insolvency-2025",
      "category": "market_signal",
      "entity": "Builder.ai",
      "topic": [
        "insolvency",
        "ai_startup",
        "counterparty_risk"
      ],
      "published_at": "2025-05-20",
      "event_at": "2025-05-20",
      "source_title": "Builder.ai enters insolvency proceedings",
      "source_url": "https://www.ft.com/content/9fdb4e63-a047-48ef-b3e2-3c454ad0c9f4",
      "source_kind": "news_report",
      "evidence_class": "insolvency_report",
      "confidence": "high",
      "claim": "Builder.ai entered insolvency proceedings after a cash crunch, showing that large fundraising and AI-product adoption claims do not guarantee operating continuity.",
      "quantified": {},
      "assumptions": [
        "The record concerns company continuity, not a frontier-model training cluster or public cloud region."
      ],
      "contradictions_or_limits": [
        "Insolvency does not itself quantify customer workload loss, GPU capacity, debt recovery, or final liquidation outcome.",
        "Company and creditor allegations require separate adjudication from the insolvency lifecycle fact."
      ],
      "fak_implications": [
        "Startup/provider ledgers should type insolvency separately from acquisition, shutdown, service migration, and technical failure."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "equinix-metal-winddown-2024",
      "category": "hyperscaler",
      "entity": "Equinix Metal",
      "topic": [
        "service_winddown",
        "bare_metal_cloud",
        "asset_impairment"
      ],
      "published_at": "2025-02-21",
      "event_at": "2024-12-31",
      "source_title": "Equinix 2024 Form 10-K",
      "source_url": "https://www.sec.gov/Archives/edgar/data/1101239/000162828025005126/eqix-20241231.htm",
      "source_kind": "regulatory_filing",
      "evidence_class": "official_financial_disclosure",
      "confidence": "high",
      "claim": "Equinix disclosed that it would no longer commercially offer Metal and would wind down supporting operations; its 2024 Form 10-K recorded $233M of impairment charges related to the Metal wind-down and projected cash flows, a service-line retreat distinct from shutting Equinix colocation data centers.",
      "quantified": {
        "metal_winddown_impairment_usd": 233000000
      },
      "assumptions": [
        "The wind-down applies to the Metal service and associated assets, not the company’s entire data-center footprint."
      ],
      "contradictions_or_limits": [
        "An accounting impairment is not equivalent to removed IT MW, and the filing does not provide a workload or customer migration distribution.",
        "Wind-down timing and residual support require date-specific follow-up."
      ],
      "fak_implications": [
        "Cloud capacity maps should distinguish service/product withdrawal from physical site closure and retain migration deadlines."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "intel-germany-poland-cancel-ohio-slow-2025",
      "category": "supply_chain",
      "entity": "Intel manufacturing expansion",
      "topic": [
        "fab_cancellation",
        "construction_delay",
        "semiconductor_capacity"
      ],
      "published_at": "2025-07-24",
      "event_at": "2025-07-24",
      "source_title": "Intel Reports Second-Quarter 2025 Financial Results",
      "source_url": "https://www.intc.com/news-events/press-releases/detail/1745/intel-reports-second-quarter-2025-financial-results",
      "source_kind": "official_earnings",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Intel said it would not move forward with planned projects in Germany and Poland and would slow Ohio construction to match market demand, reducing previously announced manufacturing expansion without proving loss of already-live wafer capacity.",
      "quantified": {
        "cancelled_projects": 2,
        "slowed_projects": 1
      },
      "assumptions": [
        "Germany and Poland are discontinued planned projects; Ohio is slowed, not cancelled."
      ],
      "contradictions_or_limits": [
        "The disclosure does not provide a single comparable wafer-start, accelerator-output, or AI-specific capacity decrement.",
        "Planned investment and subsidies are not equivalent to qualified shipped output."
      ],
      "fak_implications": [
        "Supply forecasts must carry project state transitions and exclude cancelled or delayed capacity from near-term usable supply."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "google-franklin-township-withdrawal-2025",
      "category": "datacenter_physical",
      "entity": "Google Franklin Township data-center proposal",
      "topic": [
        "project_withdrawal",
        "site_pipeline",
        "permitting"
      ],
      "published_at": "2025-09-22",
      "event_at": "2025-09-22",
      "source_title": "Google backs down from proposed data center after months of community pushback",
      "source_url": "https://www.wfyi.org/wfyi-news/2025-09-22/indianapolis-council-google-data-center-vote-withdrawl",
      "source_kind": "local_public_media",
      "evidence_class": "named_project_withdrawal",
      "confidence": "medium",
      "claim": "Google withdrew its rezoning proposal for a data-center campus on hundreds of acres in Franklin Township, Indianapolis, before construction, removing one named pipeline project rather than closing an operating site.",
      "quantified": {},
      "assumptions": [
        "Withdrawal date is tied to the local land-use/permitting process reported by the source."
      ],
      "contradictions_or_limits": [
        "A withdrawn proposal is not lost live MW, installed accelerators, or a company-wide capex reversal.",
        "The source does not establish whether demand or capital shifted to another site."
      ],
      "fak_implications": [
        "Site ledgers should retain withdrawn and relocated proposals so announced capacity is not double-counted as built supply."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "us-datacenter-cancellation-cohort-2025",
      "category": "datacenter_physical",
      "entity": "United States data-center cancellation cohort",
      "topic": [
        "project_cancellation",
        "pipeline_attrition",
        "public_records"
      ],
      "published_at": "2026-01-12",
      "event_at": "2025-12-31",
      "source_title": "Local Pushback, Canceled Data Centers Surged in 2025",
      "source_url": "https://heatmap.news/politics/data-center-cancellations-2025",
      "source_kind": "investigative_report",
      "evidence_class": "public_record_review",
      "confidence": "medium",
      "claim": "Heatmap’s review found 25 U.S. data-center projects canceled following local opposition in 2025, 21 in the second half, providing a bounded opposition-linked attrition cohort rather than a direct estimate of lost live capacity.",
      "quantified": {
        "cancelled_projects": 25,
        "second_half_cancelled_projects": 21,
        "cohort_year": 2025
      },
      "assumptions": [
        "The reported count is a discovered lower bound from public records and reporting, not a complete census."
      ],
      "contradictions_or_limits": [
        "Projects differ in size, stage, tenant, cause, and whether capacity was relocated; counts cannot be summed as homogeneous MW.",
        "Cancellation of a proposal or expansion is not shutdown of operating IT load."
      ],
      "fak_implications": [
        "Capacity models need stage-weighted cancellation hazards and should not assign equal MW or probability to every announced project."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "ferc-susquehanna-colocation-rejection-2024",
      "category": "policy_regulation",
      "entity": "FERC / Susquehanna-Amazon co-location",
      "topic": [
        "colocation",
        "interconnection_agreement",
        "regulatory_rejection"
      ],
      "published_at": "2024-11-01",
      "event_at": "2024-11-01",
      "source_title": "Order Rejecting Amended Interconnection Service Agreement, ER24-2172-000",
      "source_url": "https://www.ferc.gov/sites/default/files/2024-11/20241101-3061_ER24-2172-000.pdf",
      "source_kind": "regulator_order",
      "evidence_class": "official_regulatory_decision",
      "confidence": "high",
      "claim": "FERC rejected without prejudice an amended PJM interconnection service agreement that would have increased co-located load at the Susquehanna nuclear site from 300 MW to 480 MW, finding the applicants had not met their burden under the proposed non-conforming arrangement.",
      "quantified": {
        "existing_colocated_load_mw": 300,
        "proposed_colocated_load_mw": 480,
        "proposed_increment_mw": 180
      },
      "assumptions": [
        "The order concerns the amended interconnection agreement and allocation/treatment of an expanded behind-the-meter load."
      ],
      "contradictions_or_limits": [
        "Rejection without prejudice is not a permanent ban, site shutdown, or finding that Amazon cannot source power elsewhere.",
        "The proposed 180 MW increment is not installed or live IT capacity and must not be counted as delivered load."
      ],
      "fak_implications": [
        "Power ledgers must carry regulator approval state and distinguish co-location proposals from energized service."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "aep-ohio-data-center-tariff-2025",
      "category": "policy_regulation",
      "entity": "AEP Ohio data-center tariff",
      "topic": [
        "large_load_tariff",
        "minimum_demand",
        "collateral"
      ],
      "published_at": "2025-07-09",
      "event_at": "2025-07-09",
      "source_title": "Data Center Tariff",
      "source_url": "https://www.aepohio.com/company/about/rates/data-center-tariff/",
      "source_kind": "utility_tariff",
      "evidence_class": "approved_tariff",
      "confidence": "high",
      "claim": "AEP Ohio’s approved Schedule DCT applies enhanced study, contract, minimum-demand, collateral, and exit obligations to new data-center loads above 25 MW; the initial term is a ramp of up to four years plus eight years, with billing demand generally bounded by an 85% floor.",
      "quantified": {
        "threshold_kw": 25000,
        "maximum_ramp_years": 4,
        "post_ramp_contract_years": 8,
        "maximum_initial_term_years": 12,
        "minimum_demand_fraction_max": 0.85,
        "collateral_fraction_of_term_minimum_charges": 0.5,
        "prior_requests_mw": 30000
      },
      "assumptions": [
        "The tariff converts requests into stronger financial commitments but does not prove energization or actual consumption."
      ],
      "contradictions_or_limits": [
        "Minimum billing demand is a financial obligation, not measured utilization, energy consumption, or live IT load.",
        "Infrastructure timing remains contingent on utility and PJM studies and construction; contracted capacity can come online progressively."
      ],
      "fak_implications": [
        "Queue forecasts should weight binding collateralized tariff contracts differently from uncommitted load-study requests."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "dominion-gs5-large-load-tariff-2025",
      "category": "policy_regulation",
      "entity": "Dominion Energy Virginia GS-5",
      "topic": [
        "large_load_tariff",
        "minimum_demand",
        "collateral"
      ],
      "published_at": "2025-11-25",
      "event_at": "2025-11-25",
      "source_title": "Virginia SCC Final Order, PUR-2025-00058",
      "source_url": "https://www.scc.virginia.gov/docketsearch/DOCS/89g601!.PDF",
      "source_kind": "regulator_order",
      "evidence_class": "official_regulatory_decision",
      "confidence": "high",
      "claim": "Virginia’s SCC approved Dominion’s GS-5 class effective January 1, 2027 for contiguous-site loads of at least 25 MW and at least 75% load factor, with 14-year contracts, up to four years of ramp, $1.5M/MW collateral before credit reductions, and minimum demand charges of 85% for distribution/transmission and 60% for generation.",
      "quantified": {
        "effective_at": "2027-01-01",
        "threshold_mw": 25,
        "minimum_load_factor": 0.75,
        "contract_years": 14,
        "maximum_ramp_years": 4,
        "minimum_annual_ramp_fraction": 0.2,
        "collateral_usd_per_mw": 1500000,
        "maximum_collateral_credit_reduction_fraction": 0.7,
        "minimum_distribution_demand_fraction": 0.85,
        "minimum_transmission_demand_fraction": 0.85,
        "minimum_generation_demand_fraction": 0.6
      },
      "assumptions": [
        "GS-5 is a customer-class and cost-allocation framework, not one named data-center service agreement."
      ],
      "contradictions_or_limits": [
        "Contracted and minimum-billed demand are not actual load, utilization, or energized capacity.",
        "The order separately directs a transparent large-load interconnection-queue proceeding; tariff eligibility does not guarantee queue completion."
      ],
      "fak_implications": [
        "Capacity ledgers should separate tariff commitment, queue readiness, collateral, cost recovery, energization, and actual load."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-laidley-lpsc-power-approval-2025",
      "category": "datacenter_physical",
      "entity": "Meta Laidley / Entergy Louisiana",
      "topic": [
        "named_site_power",
        "generation_approval",
        "transmission_approval"
      ],
      "published_at": "2025-08-20",
      "event_at": "2025-08-20",
      "source_title": "LPSC Business and Executive Session Transcript, Docket U-37425",
      "source_url": "https://www.lpsc.louisiana.gov/docs/transcripts/August-20-2025-BE.pdf",
      "source_kind": "regulator_transcript",
      "evidence_class": "official_regulatory_approval",
      "confidence": "high",
      "claim": "Louisiana regulators approved the settlement path for Entergy resources serving Meta subsidiary Laidley’s Richland Parish hyperscale project: three combined-cycle turbines totaling 2,262 MW, a new 500-kV line, substations, and upgrades.",
      "quantified": {
        "approved_generators": 3,
        "approved_generation_mw": 2262,
        "generator_capacity_mw_each": 754,
        "transmission_voltage_kv": 500,
        "commission_vote_for": 4,
        "commission_vote_against": 1
      },
      "assumptions": [
        "The approval authorizes generation and transmission development for a named customer project; construction and energization follow separately."
      ],
      "contradictions_or_limits": [
        "Approved generation nameplate is not live IT MW, campus compute power, delivered annual energy, or schedulable accelerator capacity.",
        "The record does not prove completion, commissioning, availability, fuel delivery, or Meta’s eventual realized load."
      ],
      "fak_implications": [
        "Named-site power records should preserve regulator approval, generation nameplate, transmission scope, construction, commissioning, and delivered-load states separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "micron-hbm3e-volume-production-2024",
      "category": "supply_chain",
      "entity": "Micron HBM3E",
      "topic": [
        "hbm",
        "volume_production",
        "customer_shipment"
      ],
      "published_at": "2024-02-26",
      "event_at": "2024-02-26",
      "source_title": "Micron Commences Volume Production of Industry-Leading HBM3E Solution",
      "source_url": "https://investors.micron.com/news-releases/news-release-details/micron-commences-volume-production-industry-leading-hbm3e",
      "source_kind": "company_press_release",
      "evidence_class": "volume_production_announcement",
      "confidence": "high",
      "claim": "Micron said its 24 GB 8-high HBM3E had entered volume production and would ship with NVIDIA H200 accelerators beginning in calendar Q2 2024, moving this product from sample/qualification language into production tied to a named customer platform.",
      "quantified": {
        "stack_capacity_gb": 24,
        "stack_height_dies": 8,
        "named_customer_ship_start": "2024-Q2"
      },
      "assumptions": [
        "Volume production and named-platform shipment timing are supplier disclosures, not audited unit counts or end-customer deployment."
      ],
      "contradictions_or_limits": [
        "No wafer starts, good-stack yield, shipped units, customer allocation, price, or deployed accelerator count is disclosed.",
        "Production start is upstream of system shipment, site acceptance, and schedulable GPU capacity."
      ],
      "fak_implications": [
        "HBM supply ledgers should distinguish engineering samples, qualification, volume production, named-platform shipment, and deployed accelerator inventory."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "micron-hbm4-high-volume-shipments-2026",
      "category": "supply_chain",
      "entity": "Micron HBM4",
      "topic": [
        "hbm4",
        "high_volume_shipments",
        "customer_qualification"
      ],
      "published_at": "2026-06-24",
      "event_at": "2026-06-24",
      "source_title": "Micron Technology Reports Results for the Third Quarter of Fiscal 2026",
      "source_url": "https://investors.micron.com/news-releases/news-release-details/micron-technology-inc-reports-results-third-quarter-fiscal-2026",
      "source_kind": "official_earnings",
      "evidence_class": "official_shipment_disclosure",
      "confidence": "high",
      "claim": "Micron reported that HBM4 had entered high-volume shipments for customer platforms, a stronger lifecycle state than qualification samples, while HBM supply remained sold out under customer agreements.",
      "quantified": {},
      "assumptions": [
        "High-volume shipment is supplier output to customer platforms, not a disclosed module count or complete system deployment."
      ],
      "contradictions_or_limits": [
        "The earnings release does not disclose HBM4 units, bit volume, yield, customer split, platform acceptance, or deployed cluster capacity.",
        "Sold-out supply and agreements indicate allocation pressure but not shipment quantity or useful goodput."
      ],
      "fak_implications": [
        "Procurement forecasts should weight high-volume shipment above samples but still require unit, allocation, and downstream system evidence."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "broadcom-tomahawk6-production-shipments-2025",
      "category": "supply_chain",
      "entity": "Broadcom Tomahawk 6",
      "topic": [
        "ethernet_switch",
        "production_volume_shipments",
        "networking"
      ],
      "published_at": "2025-06-03",
      "event_at": "2025-06-03",
      "source_title": "Broadcom Ships Tomahawk 6, Industry’s First 102.4 Tbps Ethernet Switch",
      "source_url": "https://www.broadcom.com/company/news/product-releases/130211",
      "source_kind": "company_press_release",
      "evidence_class": "production_volume_shipment",
      "confidence": "high",
      "claim": "Broadcom said Tomahawk 6, a 102.4 Tbps Ethernet switch with 200G SerDes, was shipping in production volume, crossing from sampling into merchant-silicon delivery for scale-up and scale-out AI networks.",
      "quantified": {
        "switch_capacity_tbps": 102.4,
        "serdes_gbps": 200
      },
      "assumptions": [
        "Production-volume shipment refers to merchant switch silicon, not assembled switch-system deployments."
      ],
      "contradictions_or_limits": [
        "No shipped-unit count, customer allocation, port utilization, optics availability, fabric size, or production cluster prevalence is disclosed.",
        "Switch-silicon shipment does not prove compatible optics, cables, systems, or end-to-end network goodput."
      ],
      "fak_implications": [
        "Network BOM ledgers should couple switch ASIC delivery to optics, system, cabling, topology, and deployment receipts."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tsmc-cowos-l-production-qualification-2024",
      "category": "supply_chain",
      "entity": "TSMC CoWoS-L",
      "topic": [
        "advanced_packaging",
        "cowos",
        "customer_qualification"
      ],
      "published_at": "2024-04-18",
      "event_at": "2024-04-18",
      "source_title": "TSMC 1Q24 Earnings Conference Transcript",
      "source_url": "https://investor.tsmc.com/sites/ir/financial-report/2024/TSMC%201Q24%20Transcript.pdf",
      "source_kind": "official_earnings",
      "evidence_class": "official_production_disclosure",
      "confidence": "high",
      "claim": "TSMC said CoWoS-L had entered production and customer qualification, while it was expanding advanced-packaging capacity to address AI demand; this is a qualified process state rather than a disclosed package-volume or accelerator-shipment count.",
      "quantified": {},
      "assumptions": [
        "Production and customer qualification describe process/customer readiness, not universal qualification across every design."
      ],
      "contradictions_or_limits": [
        "No package output, yield, qualified-customer count, allocation, queue time, or deployed accelerator total is disclosed.",
        "Capacity expansion and qualification do not equal shipped good packages or live systems."
      ],
      "fak_implications": [
        "Packaging ledgers should type process production, design qualification, good-package output, customer shipment, and system deployment separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tsmc-arizona-n4-high-volume-production-2025",
      "category": "supply_chain",
      "entity": "TSMC Arizona Fab 21 N4",
      "topic": [
        "wafer_fab",
        "high_volume_production",
        "yield"
      ],
      "published_at": "2025-01-16",
      "event_at": "2024-Q4",
      "source_title": "TSMC 4Q24 Earnings Conference Transcript",
      "source_url": "https://investor.tsmc.com/sites/ir/financial-report/2024/TSMC%204Q24%20Transcript.pdf",
      "source_kind": "official_earnings",
      "evidence_class": "official_high_volume_production",
      "confidence": "high",
      "claim": "TSMC said its first Arizona fab entered high-volume production on N4 in 4Q24, with yield comparable to Taiwan fabs, converting the site from construction/ramp into qualified leading-edge wafer output.",
      "quantified": {
        "process_node_nm_class": 4,
        "high_volume_production_start": "2024-Q4"
      },
      "assumptions": [
        "Comparable yield is the company’s statement and does not disclose the underlying yield percentage."
      ],
      "contradictions_or_limits": [
        "No wafer-start count, product/customer mix, good-die output, AI share, packaging destination, or accelerator shipment is disclosed.",
        "Leading-edge wafer production is upstream of packaging, HBM integration, board/system assembly, and deployed compute."
      ],
      "fak_implications": [
        "Domestic-fab capacity models should begin credit at qualified high-volume output, then separately track product mix, good dies, packaging, and system shipment."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "speculative-decoding-foundation-2022",
      "category": "serving_system",
      "entity": "Speculative decoding",
      "topic": [
        "speculative_decoding",
        "lossless_sampling",
        "draft_verification"
      ],
      "published_at": "2022-11-30",
      "event_at": "2022-11-30",
      "source_title": "Fast Inference from Transformers via Speculative Decoding",
      "source_url": "https://arxiv.org/abs/2211.17192",
      "source_kind": "research_paper",
      "evidence_class": "controlled_benchmark",
      "confidence": "high",
      "claim": "The foundational speculative-decoding method uses a faster approximation model to draft several tokens and a target model to verify them in parallel with a rejection-sampling correction that preserves the target output distribution; T5-XXL experiments reported roughly 2x-3x acceleration over standard T5X decoding.",
      "quantified": {
        "speedup_min": 2,
        "speedup_max": 3
      },
      "assumptions": [
        "Speedup depends on draft cost, draft-target agreement, verification cost, hardware, model pair, sequence geometry, and load."
      ],
      "contradictions_or_limits": [
        "The T5-XXL/T5X benchmark is not a production acceptance distribution or universal LLM-serving multiplier.",
        "Rejected drafts still consume draft and verification work; the paper does not supply provider retry/failure denominators."
      ],
      "fak_implications": [
        "Decode accounting should track drafted, verified, accepted, rejected, fallback, and emitted tokens separately."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "specinfer-tree-speculation-2023",
      "category": "serving_system",
      "entity": "SpecInfer",
      "topic": [
        "speculative_decoding",
        "token_tree",
        "distributed_inference"
      ],
      "published_at": "2023-05-16",
      "event_at": "2023-05-16",
      "source_title": "SpecInfer: Accelerating Large Language Model Serving with Tree-based Speculative Inference and Verification",
      "source_url": "https://arxiv.org/abs/2305.09781",
      "source_kind": "research_paper",
      "evidence_class": "controlled_benchmark",
      "confidence": "high",
      "claim": "SpecInfer drafts and verifies token trees to use spare parallel resources; it reported 1.5x-2.8x gains for distributed inference and 2.6x-3.5x for offloading inference, with tree speculation 1.2x-1.5x faster than sequence-based speculation in its evaluated envelopes.",
      "quantified": {
        "distributed_speedup_min": 1.5,
        "distributed_speedup_max": 2.8,
        "offloading_speedup_min": 2.6,
        "offloading_speedup_max": 3.5,
        "tree_vs_sequence_speedup_min": 1.2,
        "tree_vs_sequence_speedup_max": 1.5
      },
      "assumptions": [
        "Tree width and depth trade extra candidate work for higher probability of accepting useful continuations."
      ],
      "contradictions_or_limits": [
        "Distributed and offloading results use different bottlenecks and are not interchangeable or stackable.",
        "The paper does not establish production prevalence, request-level acceptance histograms, or retry distributions."
      ],
      "fak_implications": [
        "Schedulers should cost candidate-tree nodes and verifier work, not count only emitted tokens or nominal draft length."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "medusa-multiple-heads-2024",
      "category": "serving_system",
      "entity": "Medusa",
      "topic": [
        "parallel_decoding",
        "multiple_heads",
        "tree_verification"
      ],
      "published_at": "2024-01-19",
      "event_at": "2024-01-19",
      "source_title": "Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads",
      "source_url": "https://arxiv.org/abs/2401.10774",
      "source_kind": "research_paper",
      "evidence_class": "controlled_benchmark",
      "confidence": "high",
      "claim": "Medusa adds multiple decoding heads to one backbone and verifies a candidate tree without maintaining a separate draft model; Medusa-1 reported over 2.2x speedup with the backbone frozen, while jointly trained Medusa-2 reported 2.3x-3.6x.",
      "quantified": {
        "medusa1_speedup_min": 2.2,
        "medusa2_speedup_min": 2.3,
        "medusa2_speedup_max": 3.6
      },
      "assumptions": [
        "Head accuracy, candidate tree, acceptance rule, backbone model, task, and hardware determine realized gains."
      ],
      "contradictions_or_limits": [
        "Medusa-1 and Medusa-2 use different training and quality-preservation contracts and must not be conflated.",
        "Benchmark maxima are not request-population acceptance distributions or evidence of deployment share."
      ],
      "fak_implications": [
        "Serving manifests should record whether speculation uses a separate drafter, attached heads, joint training, or relaxed acceptance."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "eagle-feature-speculation-2024",
      "category": "serving_system",
      "entity": "EAGLE",
      "topic": [
        "speculative_decoding",
        "feature_prediction",
        "acceptance_length"
      ],
      "published_at": "2024-01-26",
      "event_at": "2024-01-26",
      "source_title": "EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty",
      "source_url": "https://arxiv.org/abs/2401.15077",
      "source_kind": "research_paper",
      "evidence_class": "controlled_benchmark",
      "confidence": "high",
      "claim": "EAGLE predicts second-to-top-layer features plus an advanced token sequence while preserving the target distribution; across its LLaMA/Vicuna evaluations it reported roughly 1.5x-2.8x wall-clock speedup, with acceptance length varying materially by task, model, temperature, and draft position.",
      "quantified": {
        "wall_clock_speedup_min": 1.5,
        "wall_clock_speedup_max": 2.8
      },
      "assumptions": [
        "Acceptance length is tokens emitted per draft-verification cycle, including method-specific verification semantics."
      ],
      "contradictions_or_limits": [
        "The speedup range spans named paper evaluations and is not a request-population acceptance distribution or universal multiplier.",
        "Acceptance varies by task, model, temperature, token position, batch/load, hardware, and implementation."
      ],
      "fak_implications": [
        "Acceptance should be stored as a conditional position/task/model distribution, not one global scalar."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "magicdec-long-context-batching-2024",
      "category": "serving_system",
      "entity": "MagicDec",
      "topic": [
        "speculative_decoding",
        "long_context",
        "batching"
      ],
      "published_at": "2024-08-20",
      "event_at": "2024-08-20",
      "source_title": "MagicDec: Breaking the Latency-Throughput Tradeoff for Long Context Generation with Speculative Decoding",
      "source_url": "https://arxiv.org/abs/2408.11049",
      "source_kind": "research_paper",
      "evidence_class": "controlled_benchmark",
      "confidence": "high",
      "claim": "MagicDec uses sparse-KV draft models for long-context serving and shows that speculative decoding can help at batches 32-256 when KV work changes the bottleneck; it reported up to 2x for LLaMA-2-7B-32K and 1.84x for LLaMA-3.1-8B on eight A100 GPUs.",
      "quantified": {
        "batch_size_min": 32,
        "batch_size_max": 256,
        "a100_gpus": 8,
        "llama2_7b_32k_speedup_max": 2,
        "llama31_8b_speedup_max": 1.84
      },
      "assumptions": [
        "Large-batch benefits arise in moderate/long-sequence regimes where KV-cache costs and draft strategy alter the memory/compute balance."
      ],
      "contradictions_or_limits": [
        "This does not mean speculative decoding improves every large batch; short contexts, low agreement, or expensive drafting can erase gains.",
        "Controlled benchmarks do not provide production acceptance, retry, load, or request-mix distributions."
      ],
      "fak_implications": [
        "Batching policy should condition speculative admission on context length, effective batch, draft cost, KV footprint, and measured acceptance."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "webarena-browser-workloads-2023",
      "category": "workload_model",
      "entity": "WebArena",
      "topic": [
        "browser_agent",
        "long_horizon",
        "web_interaction"
      ],
      "published_at": "2023-07-25",
      "event_at": "2023-07-25",
      "source_title": "WebArena: A Realistic Web Environment for Building Autonomous Agents",
      "source_url": "https://arxiv.org/abs/2307.13854",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_workload",
      "confidence": "high",
      "claim": "WebArena provides 812 long-horizon tasks over five self-hosted websites spanning four application domains—e-commerce, social discussion, collaborative software development, and content management—with functional end-state evaluation; the original best GPT-4 agent achieved 14.41% versus 78.24% human success.",
      "quantified": {
        "tasks": 812,
        "websites": 5,
        "application_domains": 4,
        "best_gpt4_success_fraction": 0.1441,
        "human_success_fraction": 0.7824
      },
      "assumptions": [
        "Browser trajectories include observation rendering, page navigation, actions, external knowledge, and environment state."
      ],
      "contradictions_or_limits": [
        "The 812 tasks are curated benchmark instances, not production browser sessions, users, arrival rates, or action-count distributions.",
        "Self-hosted sites and evaluator resets differ from live-web availability, authentication, safety, and mutation costs."
      ],
      "fak_implications": [
        "Browser workloads need page/observation bytes, action count, website state, reset cost, and end-state evaluation—not only model tokens."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "workarena-enterprise-workloads-2024",
      "category": "workload_model",
      "entity": "WorkArena / BrowserGym",
      "topic": [
        "enterprise_agent",
        "browser_observation",
        "knowledge_work"
      ],
      "published_at": "2024-03-12",
      "event_at": "2024-03-12",
      "source_title": "WorkArena: How Capable are Web Agents at Solving Common Knowledge Work Tasks?",
      "source_url": "https://arxiv.org/abs/2403.07718",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_workload",
      "confidence": "high",
      "claim": "WorkArena defines 33 ServiceNow knowledge-work tasks and 19,912 unique instances covering lists, forms, knowledge search, service catalogs, dashboards, and menus; rendered flat HTML can range from 40k to 500k tokens even after basic cleaning.",
      "quantified": {
        "task_types": 33,
        "unique_instances": 19912,
        "cleaned_html_tokens_min": 40000,
        "cleaned_html_tokens_max": 500000
      },
      "assumptions": [
        "Enterprise browser state can dominate context size and includes dynamic UI, nested frames, shadow DOM, and proprietary elements."
      ],
      "contradictions_or_limits": [
        "The combinatorial instances are benchmark-generated on developer instances, not 19,912 production users or sessions.",
        "HTML token range is page representation size, not prompt size after accessibility filtering, compression, or visual-only interaction."
      ],
      "fak_implications": [
        "Agent memory and context planners should budget observation representation separately from conversational history and tool outputs."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "tau-bench-customer-service-2024",
      "category": "workload_model",
      "entity": "tau-bench",
      "topic": [
        "customer_service_agent",
        "api_tools",
        "multi_turn"
      ],
      "published_at": "2024-06-17",
      "event_at": "2024-06-17",
      "source_title": "tau-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains",
      "source_url": "https://arxiv.org/abs/2406.12045",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_workload",
      "confidence": "high",
      "claim": "tau-bench evaluates multi-turn retail and airline customer-service conversations in which a simulated user interacts with an agent that must follow policy and call domain APIs, scoring the resulting database state and repeated-trial reliability; reported agents succeeded on under 50% of tasks and retail pass^8 was under 25%.",
      "quantified": {
        "domains": 2,
        "single_trial_success_fraction_max_exclusive": 0.5,
        "retail_pass8_fraction_max_exclusive": 0.25,
        "reliability_trials": 8
      },
      "assumptions": [
        "Tool-call correctness, policy compliance, user clarification, database reads/writes, and repeated-trial consistency are separate workload dimensions."
      ],
      "contradictions_or_limits": [
        "Simulated users and curated tasks are not production call-center traffic, turn-count, tool-frequency, escalation, or retry distributions.",
        "pass^k is repeated-task reliability, not independent user success probability under arbitrary production mix."
      ],
      "fak_implications": [
        "Customer-service serving should price multi-turn state, read/write tool calls, policy checks, user-simulation variance, and repeated reliability."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "gaia-general-assistant-2023",
      "category": "workload_model",
      "entity": "GAIA",
      "topic": [
        "general_assistant",
        "web_browsing",
        "multimodal_tool_use"
      ],
      "published_at": "2023-11-21",
      "event_at": "2023-11-21",
      "source_title": "GAIA: A Benchmark for General AI Assistants",
      "source_url": "https://arxiv.org/abs/2311.12983",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_workload",
      "confidence": "high",
      "claim": "GAIA contains 466 human-authored questions across three difficulty levels requiring combinations of reasoning, web browsing, file reading, coding, and multimodality; 355 questions were tagged for web browsing, while humans scored 92% and GPT-4 with plugins 15% in the paper.",
      "quantified": {
        "questions": 466,
        "difficulty_levels": 3,
        "web_browsing_questions": 355,
        "coding_questions": 154,
        "multimodality_questions": 138,
        "diverse_filetype_questions": 129,
        "human_success_fraction": 0.92,
        "gpt4_plugins_success_fraction": 0.15
      },
      "assumptions": [
        "Capability tags overlap; their counts must not be summed into unique questions."
      ],
      "contradictions_or_limits": [
        "GAIA measures exact-answer question completion, not interactive production session lengths, tool-call counts, latency, or user demand.",
        "Open-web tasks can decay or change, and human-reported solution paths need not match agent paths."
      ],
      "fak_implications": [
        "General-assistant routing needs per-task capability composition and open-world retrieval costs, not one assistant-average context or tool budget."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "osworld-desktop-workloads-2024",
      "category": "workload_model",
      "entity": "OSWorld",
      "topic": [
        "computer_use_agent",
        "desktop_gui",
        "multimodal_interaction"
      ],
      "published_at": "2024-04-11",
      "event_at": "2024-04-11",
      "source_title": "OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments",
      "source_url": "https://arxiv.org/abs/2404.07972",
      "source_kind": "research_paper",
      "evidence_class": "benchmark_workload",
      "confidence": "high",
      "claim": "OSWorld defines 369 real-computer tasks across web and desktop applications, OS file I/O, and multi-application workflows on Ubuntu, Windows, and macOS, with task-specific initial-state setup and execution-based evaluators; humans achieved over 72.36% while the best evaluated model achieved 12.24%.",
      "quantified": {
        "tasks": 369,
        "operating_systems": 3,
        "human_success_fraction_min": 0.7236,
        "best_model_success_fraction": 0.1224
      },
      "assumptions": [
        "Desktop-agent cost includes screenshots or accessibility state, GUI/CLI actions, application startup, files, evaluators, and environment restoration."
      ],
      "contradictions_or_limits": [
        "The 369 curated tasks are not a production desktop-session distribution or evidence of application popularity.",
        "Success rates are tied to the paper’s agents, versions, environments, and evaluator behavior."
      ],
      "fak_implications": [
        "Computer-use planners should account for multimodal observations, action latency, application state, setup/reset, and cross-app dependencies."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "lmsys-chat-1m-2023",
      "category": "workload_trace",
      "entity": "LMSYS-Chat-1M",
      "topic": [
        "conversation_trace",
        "multilingual",
        "model_mix"
      ],
      "published_at": "2023-09-21",
      "event_at": "2023-09-21",
      "source_title": "LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset",
      "source_url": "https://arxiv.org/abs/2309.11998",
      "source_kind": "research_paper",
      "evidence_class": "public_conversation_trace",
      "confidence": "high",
      "claim": "LMSYS-Chat-1M contains one million public-user conversations with 25 state-of-the-art LLMs collected from the Vicuna demo and Chatbot Arena between April and August 2023; the released records include conversation turns, model identity, timestamp, language, and anonymized user identifiers.",
      "quantified": {
        "conversations": 1000000,
        "models": 25,
        "collection_start": "2023-04",
        "collection_end": "2023-08"
      },
      "assumptions": [
        "Conversation count, message/turn count, anonymized user ID, language, timestamp, and model response are distinct fields and denominators."
      ],
      "contradictions_or_limits": [
        "Public opt-in demo/Arena traffic is not provider-wide production, billing, enterprise/API, or geography-representative traffic.",
        "An anonymized user identifier is not a verified person, tenant, household, or stable cross-platform identity."
      ],
      "fak_implications": [
        "Conversation planners should preserve turn and model geometry while treating public-chat sampling bias separately from production arrivals."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "wildchat-1m-2024",
      "category": "workload_trace",
      "entity": "WildChat",
      "topic": [
        "conversation_trace",
        "geography",
        "language",
        "toxicity"
      ],
      "published_at": "2024-05-02",
      "event_at": "2024-05-02",
      "source_title": "WildChat: 1M ChatGPT Interaction Logs in the Wild",
      "source_url": "https://arxiv.org/abs/2405.01470",
      "source_kind": "research_paper",
      "evidence_class": "public_conversation_trace",
      "confidence": "high",
      "claim": "WildChat releases one million conversations comprising about 2.5 million user and ChatGPT turns, collected from more than 68,000 anonymized users, with timestamps, inferred country and language metadata, and in-the-wild content including toxic and jailbreak interactions.",
      "quantified": {
        "conversations": 1000000,
        "turns_approx": 2500000,
        "anonymized_users_min": 68000
      },
      "assumptions": [
        "User counts derive from anonymized identifiers; inferred country and language are attributes with classification/error boundaries."
      ],
      "contradictions_or_limits": [
        "The public free-access chatbot population is self-selected and not a ChatGPT-wide or provider billing/API census.",
        "IP-derived identity/country does not equal unique person, citizenship, residence, tenant, or stable user across address changes."
      ],
      "fak_implications": [
        "Geographic workload models can use country-local timestamps as bounded samples but must not promote them to universal regional demand weights."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "chatbot-arena-preference-2024",
      "category": "workload_trace",
      "entity": "Chatbot Arena",
      "topic": [
        "human_preference",
        "conversation_trace",
        "model_routing"
      ],
      "published_at": "2024-03-07",
      "event_at": "2024-03-07",
      "source_title": "Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference",
      "source_url": "https://arxiv.org/abs/2403.04132",
      "source_kind": "research_paper",
      "evidence_class": "public_preference_trace",
      "confidence": "high",
      "claim": "Chatbot Arena reports an open pairwise-evaluation platform with more than 240,000 votes from over 90,000 users across more than 50 models by the paper snapshot, linking conversation prompts and randomized model pairs to user preference outcomes.",
      "quantified": {
        "votes_min": 240000,
        "users_min": 90000,
        "models_min": 50
      },
      "assumptions": [
        "Votes, users, conversations, prompts, model appearances, and pairwise comparisons are different denominators."
      ],
      "contradictions_or_limits": [
        "Arena users and votes are self-selected and not production request share, spend, token share, user retention, or model demand.",
        "Repeated votes and changing model rosters prevent treating vote count as unique users or stationary traffic."
      ],
      "fak_implications": [
        "Preference-routing evidence should preserve pair, prompt, timestamp, voter identity boundary, and repeated-observation structure."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "openassistant-conversations-2023",
      "category": "workload_trace",
      "entity": "OpenAssistant Conversations",
      "topic": [
        "conversation_tree",
        "multilingual",
        "human_annotation"
      ],
      "published_at": "2023-04-14",
      "event_at": "2023-04-14",
      "source_title": "OpenAssistant Conversations — Democratizing Large Language Model Alignment",
      "source_url": "https://arxiv.org/abs/2304.07327",
      "source_kind": "research_paper",
      "evidence_class": "crowdsourced_conversation_dataset",
      "confidence": "high",
      "claim": "OpenAssistant Conversations released 161,443 messages in 35 languages, organized into 66,497 conversation trees with 461,292 quality ratings from more than 13,500 volunteers, exposing branching dialogue and annotation geometry rather than one linear request stream.",
      "quantified": {
        "messages": 161443,
        "languages": 35,
        "conversation_trees": 66497,
        "quality_ratings": 461292,
        "volunteers_min": 13500
      },
      "assumptions": [
        "A conversation tree can branch into multiple paths; trees, messages, paths, annotations, and volunteers are distinct counts."
      ],
      "contradictions_or_limits": [
        "Crowdsourced alignment collection is not organic production traffic, arrival timing, user retention, provider geography, or service demand.",
        "Volunteer and language composition reflects project recruitment and annotation workflow."
      ],
      "fak_implications": [
        "Conversation-state models should support branching histories and multiple ratings without flattening them into independent sessions."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "vllm-spec-decode-metrics-2026",
      "category": "serving_system",
      "entity": "vLLM",
      "topic": [
        "speculative_decoding",
        "production_observability",
        "accepted_draft_length",
        "token_accounting"
      ],
      "published_at": "2026-06-14",
      "event_at": "2026-06-14",
      "source_title": "vLLM speculative-decoding metrics implementation",
      "source_url": "https://github.com/vllm-project/vllm/blob/4ef4492e9b7a5a7ba295da783d456d45db5eb9d6/vllm/v1/spec_decode/metrics.py",
      "source_kind": "official_repository",
      "evidence_class": "instrumentation_surface",
      "confidence": "high",
      "claim": "vLLM V1 records speculative-decoding statistics from each engine-step update: drafted-token count, accepted-token count, emitted-token count, draft acceptance rate, a request-level histogram of consecutive accepted draft lengths, and acceptance counts by draft position. The histogram is updated once per request that has speculative tokens in that step; the implementation also keeps drafted, accepted, emitted, and iteration totals as separate counters.",
      "quantified": {
        "request_histogram_bucket_semantics": "number of consecutive accepted draft tokens in one request update",
        "aggregate_counters": [
          "num_draft_tokens",
          "num_accepted_tokens",
          "num_emitted_tokens",
          "num_spec_decode_iterations"
        ],
        "position_index_origin": 0
      },
      "assumptions": [
        "The pinned repository implementation is evidence that instrumentation is available in that source revision, not that an operator enabled export or retained observations.",
        "A request-level accepted-length histogram and fleet aggregate counters have different denominators and must not be merged.",
        "Accepted draft tokens, drafted tokens, emitted tokens, and speculative iterations are distinct quantities."
      ],
      "contradictions_or_limits": [
        "The source publishes no representative production distribution, deployment prevalence, tenant mix, or fleet goodput outcome.",
        "The code surface does not establish that metrics are enabled, scraped, stored, or representative in any production fleet.",
        "No directly verified retry, client-side fallback, or failure-recovery distribution is exposed by this source.",
        "Availability of these counters does not convert benchmark speedup into production prevalence or production goodput."
      ],
      "fak_implications": [
        "Preserve request histogram samples separately from monotonic fleet counters and attach engine/config/version labels before aggregation.",
        "Compute acceptance diagnostics only from denominator-compatible fields; do not substitute emitted tokens for accepted drafts or iterations for requests.",
        "Treat missing collection state and retry/fallback observations as explicit unknowns."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "sglang-spec-decode-metrics-adaptive-2026",
      "category": "serving_system",
      "entity": "SGLang",
      "topic": [
        "speculative_decoding",
        "production_observability",
        "accepted_draft_length",
        "adaptive_speculation"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "SGLang speculative-decoding metrics and adaptive parameter implementation",
      "source_url": "https://github.com/sgl-project/sglang/blob/5b7fc61306138582e0fb3536047fcdbd12aedd57/python/sglang/srt/speculative/adaptive_spec_params.py",
      "source_kind": "official_repository",
      "evidence_class": "instrumentation_and_control_surface",
      "confidence": "high",
      "claim": "SGLang exposes a speculative accepted-length histogram and separate cumulative accepted-token and forward-iteration totals in its metrics collector, while its adaptive speculative-parameter implementation accepts runtime outcome statistics such as accepted lengths and batch context to choose subsequent draft parameters. This verifies an available telemetry/control seam, not a published production outcome.",
      "quantified": {
        "accepted_length_histogram": "sglang:spec_accept_length",
        "cumulative_inputs": [
          "spec_num_total_accepted_tokens",
          "spec_num_total_forward_ct"
        ],
        "adaptive_output_examples": [
          "speculative_num_steps",
          "speculative_eagle_topk"
        ]
      },
      "assumptions": [
        "The histogram, cumulative totals, and adaptive controller operate at different aggregation boundaries and remain separately typed.",
        "Controller inputs and selected parameters are implementation surfaces, not evidence of controller benefit in a production workload.",
        "The pinned repository revision is a current primary-source snapshot; operators can configure or disable the relevant features."
      ],
      "contradictions_or_limits": [
        "The source does not publish a representative request-level production distribution, fleet deployment rate, or production goodput delta.",
        "Instrumentation available does not mean enabled, collected, exported, retained, or representative.",
        "Adaptive controller input and parameter output do not establish a published controller outcome, stability envelope, or causal speedup.",
        "No directly verified retry, client-side fallback, or fallback-frequency telemetry was found in this bounded source review."
      ],
      "fak_implications": [
        "Record adaptive-controller inputs, chosen parameters, and later request outcomes as separate events so a controller can be evaluated rather than presumed effective.",
        "Keep accepted-length histograms distinct from cumulative accepted-token and iteration counters.",
        "Require an explicit collection/enabled-state witness before treating the surface as observed workload evidence."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "tensorrt-llm-speculative-benchmark-reporting-2026",
      "category": "serving_system",
      "entity": "NVIDIA TensorRT-LLM",
      "topic": [
        "speculative_decoding",
        "benchmark_reporting",
        "acceptance_rate",
        "token_accounting"
      ],
      "published_at": "2026-08-19",
      "event_at": "2026-08-19",
      "source_title": "TensorRT-LLM benchmark reporting dataclasses",
      "source_url": "https://github.com/NVIDIA/TensorRT-LLM/blob/e6caf6f09bfaf383776a9f88bb58b45940331e43/tensorrt_llm/bench/dataclasses/reporting.py",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_instrumentation_surface",
      "confidence": "high",
      "claim": "TensorRT-LLM benchmark reporting defines speculative-decoding fields for total draft tokens, total accepted draft tokens, acceptance rate, acceptance length, and acceptance probability alongside request and output-token totals. This is a benchmark/report schema surface and preserves drafted, accepted, and output-token quantities as separate fields.",
      "quantified": {
        "speculative_fields": [
          "total_draft_tokens",
          "total_accepted_draft_tokens",
          "acceptance_rate",
          "acceptance_length",
          "acceptance_probability"
        ],
        "non_speculative_denominators": [
          "num_requests",
          "total_output_tokens"
        ]
      },
      "assumptions": [
        "Fields in the benchmark report schema are available instrumentation, not proof that a production service exports them.",
        "Drafted, accepted, and output tokens are different accounting quantities; acceptance length and acceptance probability are derived statistics, not token counters."
      ],
      "contradictions_or_limits": [
        "The source is benchmark reporting, not a representative production workload distribution or fleet aggregate telemetry publication.",
        "No production prevalence, enabled-state, scrape/retention state, tenant mix, fallback frequency, retry distribution, or client-side outcome distribution is published here.",
        "Benchmark speedup or acceptance summaries cannot be promoted to production prevalence or goodput without production workload and accounting evidence."
      ],
      "fak_implications": [
        "Tag benchmark-only speculative fields so they cannot be silently joined with production request telemetry.",
        "Require request/fleet denominators and output-token accounting before comparing acceptance across engines.",
        "Retain the unresolved production retry/fallback/client-side distribution gap."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "ercot-batch-zero-puct-delay-2026",
      "category": "policy_regulation",
      "entity": "ERCOT Batch Zero large-load process",
      "topic": [
        "large_load_interconnection",
        "batch_zero",
        "queue_verification",
        "load_forecast",
        "regulatory_timeline"
      ],
      "published_at": "2026-08-20",
      "event_at": "2026-08-20",
      "source_title": "PUCT Project No. 59142 Order Granting ERCOT Request for Good Cause Exception",
      "source_url": "https://interchange.puc.texas.gov/Documents/59142_53_1675960.PDF",
      "source_kind": "regulator_filing_and_order",
      "evidence_class": "official_regulatory_decision",
      "confidence": "high",
      "claim": "ERCOT asked the Public Utility Commission of Texas to extend the first-classification deadline for its transition Batch Zero of large-load interconnection requests. The Commission granted the good-cause exception, moving the classification deadline from September 1, 2026 to October 1, 2026 while leaving the October 1, 2027 batch-study deadline unchanged. ERCOT said the extra month was needed to complete a manual initial screening process; identify duplicate, terminated, withdrawn, and completed requests; identify lower-priority projects; validate project status and readiness with transmission service providers; and reflect the resulting status changes in the Batch Zero forecast incorporated into ERCOT’s Long-Term Load Forecast before the October 1, 2026 forecasting deadline.",
      "quantified": {
        "previous_classification_deadline": "2026-09-01",
        "extended_classification_deadline": "2026-10-01",
        "batch_study_deadline_unchanged": "2027-10-01",
        "forecast_update_deadline": "2026-10-01",
        "extension_days": 30
      },
      "assumptions": [
        "The PUCT Interchange filing and order are the authoritative witnessed surfaces for the requested and granted deadline change.",
        "Batch Zero is a transition classification for requests already under study before the new batch process; it is not proof that any request completed a study, received interconnection approval, energized, or became live load."
      ],
      "contradictions_or_limits": [
        "The decision changes a classification deadline; it does not cancel or deny projects and does not stop construction.",
        "Initial screening, transmission-service-provider validation, Batch Zero classification, Long-Term Load Forecast inclusion, completed batch study, approved interconnection, energization, and measured demand are separate lifecycle states.",
        "Requested nameplate megawatts are not verified project capacity, coincident peak, annual energy, IT megawatts, or actual demand.",
        "The source does not establish the suggested approximately 474 GW queue total, a completed audit outcome, project-level verification results, or a published transmission-planning sensitivity. Those claims remain omitted pending direct official witnesses."
      ],
      "fak_implications": [
        "Grid-capacity ledgers should preserve request, screening, validated status, batch classification, forecast, study, approval, energization, and live-load states separately.",
        "Planning inputs should retain forecast version and deadline provenance because verification can revise the load set before transmission studies consume it.",
        "Do not treat a classification extension or forecast update as built or energized transmission capacity."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "new-york-executive-order-62-data-center-permit-moratorium-2026",
      "category": "policy_regulation",
      "entity": "New York Executive Order 62",
      "topic": [
        "data_center_permitting",
        "temporary_moratorium",
        "state_environmental_review",
        "grid_load_requests",
        "geis"
      ],
      "published_at": "2026-07-14",
      "event_at": "2026-07-14",
      "source_title": "Executive Order No. 62: Establishing a Temporary Moratorium on Data Centers in New York While the State Develops a Sustainable Data Center Policy",
      "source_url": "https://www.governor.ny.gov/executive-order/no-62-establishing-temporary-moratorium-data-centers-new-york-while-state-develops",
      "source_kind": "official_policy_release",
      "evidence_class": "official_regulatory_decision",
      "confidence": "high",
      "claim": "New York Executive Order 62 directed the Department of Environmental Conservation (DEC) to hold in abeyance discretionary state permit, approval, and license applications for construction or expansion of covered data centers that were pending and had not been deemed complete before the order. It excluded local-government permits and other permissions. The order cited nearly 12 GW of data-center load requests in the NYISO interconnection queue as of May 2026, with more than 8 GW entering during 2025. Its covered definition reaches facilities whose operations consume, or are designed to consume, at least 50 MW, while excluding facilities used primarily for manufacturing, research, education, or medical care. The order also starts future policy work: a DEC-led report and Generic Environmental Impact Statement (GEIS), initial agency information within 60 days, stakeholder recommendations within 90 days, and a final report and GEIS within twelve months.",
      "quantified": {
        "nyiso_data_center_load_requests_as_of_may_2026_gw": "nearly 12",
        "load_requests_entered_during_2025_gw": "more than 8",
        "covered_facility_threshold_mw": 50,
        "initial_agency_information_days": 60,
        "stakeholder_recommendations_days": 90,
        "final_report_and_geis_months": 12
      },
      "assumptions": [
        "The governor-hosted executive-order text is the authoritative witness for the order, its scope, definitions, exclusions, and directed future processes.",
        "The cited NYISO figures are queue requests reported by the order, not verified projects, coincident demand, energized capacity, or measured load."
      ],
      "contradictions_or_limits": [
        "This is not a blanket construction ban: it applies to specified discretionary state applications pending and not deemed complete before the order, and it excludes local-government permissions.",
        "The order does not establish a simple fixed one-year expiry. Its report, GEIS, 60-day, 90-day, and twelve-month processes are future work, and the order describes conditions for the moratorium rather than a single automatic expiration date.",
        "A covered facility is defined by a 50 MW consumption or design threshold and excludes facilities used primarily for manufacturing, research, education, or medical care.",
        "Queue-request megawatts are not built, approved, energized, or live demand."
      ],
      "fak_implications": [
        "Policy ledgers should preserve state versus local authority, application-completeness status, facility threshold, and use-based exclusions rather than collapsing the order into a blanket ban.",
        "Capacity models should keep NYISO queue requests separate from verified, permitted, constructed, energized, and measured demand.",
        "Treat the GEIS, report, and 60/90-day and twelve-month deliverables as pending milestones until directly witnessed as completed."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "ercot-batch-zero-process-release-2026",
      "category": "policy_regulation",
      "entity": "ERCOT Batch Zero large-load process",
      "topic": [
        "large_load_interconnection",
        "batch_zero",
        "tracked_requests",
        "data_centers",
        "transmission_planning"
      ],
      "published_at": "2026-06-18",
      "event_at": "2026-06-18",
      "source_title": "PUCT Approves ERCOT's Batch Zero Process for Connecting Large Electricity Users While Protecting System Reliability for Texans",
      "source_url": "https://www.ercot.com/news/release/06182026-puct-approves-ercots",
      "source_kind": "official_release",
      "evidence_class": "official_regulatory_approval",
      "confidence": "high",
      "claim": "ERCOT reported that the PUCT approved its Batch Zero process for large-user connection requests. ERCOT was tracking more than 438,000 MW of large-load requests, nearly 89% attributed to data centers. Qualified projects of at least 75 MW that were already under study were grouped into the transition Batch Zero. The June release expected applicant classification in August 2026 and a final statewide transmission plan for the batch in Fall 2027; the classification timing was subsequently delayed by the later PUCT record in this corpus. ERCOT expressly cautioned that not all interconnection requests result in built projects.",
      "quantified": {
        "tracked_large_load_requests_mw": "more than 438000",
        "data_center_share_percent": "nearly 89",
        "qualified_project_minimum_mw": 75,
        "classification_expected": "2026-08",
        "statewide_final_transmission_plan_expected": "Fall 2027"
      },
      "assumptions": [
        "The ERCOT release is the authoritative witness for the June 18 approval announcement, tracked-request total, stated data-center share, qualification threshold, and announced planning milestones.",
        "The more-than-438,000 MW figure is tracked request nameplate, not a verified or livable demand forecast."
      ],
      "contradictions_or_limits": [
        "The August 2026 classification expectation was later superseded by the PUCT-granted October 1, 2026 deadline recorded separately.",
        "Tracked requests, qualified Batch Zero membership, classification, forecast inclusion, study completion, approval, construction, energization, and live load are separate states.",
        "ERCOT states that not all requests become built projects; the tracked total is not verified demand, coincident peak, annual energy, or operational load.",
        "The Fall 2027 statewide plan was future work at publication and does not prove facilities were constructed or energized."
      ],
      "fak_implications": [
        "Use the tracked request total only as an upper-stage queue measure with source date and denominator, never as livable demand.",
        "Preserve the at-least-75 MW qualification rule and transition-batch status when comparing large-load queue figures.",
        "Keep announced classification and transmission-plan milestones linked to later deadline and completion evidence."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "ercot-geren-response-transmission-load-sensitivities-2026",
      "category": "supply_chain",
      "entity": "ERCOT regional transmission planning",
      "topic": [
        "transmission_planning",
        "load_forecast",
        "forecast_sensitivity",
        "765_kv",
        "batch_zero"
      ],
      "published_at": "2026-08-20",
      "event_at": "2026-08-12",
      "source_title": "ERCOT Response to Representative Geren Regarding Permian Basin Reliability Plan and 765 kV Transmission Facilities",
      "source_url": "https://www.ercot.com/files/docs/2026/08/20/Rep.-Geren-Response-2026-08-12.pdf",
      "source_kind": "official_report",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "In an August 12 response published August 20, ERCOT distinguished successive planning assumptions and forecasts. The 2024 Regional Transmission Plan used an approximately 150 GW 2030 load level; the 2025 plan used an approximately 159 GW 2031 summer peak. ERCOT also studied a lower approximately 125 GW 2030 sensitivity and said the full Strategic Texas Electric Plan (STEP) 765-kV additions were needed under both the 125 GW and 150 GW cases. Its most recent econometric base-plus-mid-size forecast was approximately 110 GW in 2032, but excluded large loads being evaluated through the Batch process; those loads must be added to obtain ERCOT’s total forecast. ERCOT expected the final forecast including Batch-process loads by the end of 2026.",
      "quantified": {
        "rtp_2024_load_assumption_2030_gw": "approximately 150",
        "rtp_2025_summer_peak_2031_gw": "approximately 159",
        "lower_sensitivity_2030_gw": "approximately 125",
        "econometric_base_plus_mid_size_forecast_2032_gw": "approximately 110",
        "final_forecast_including_batch_loads_expected": "end of 2026",
        "step_voltage_kv": 765
      },
      "assumptions": [
        "The ERCOT-hosted response dated August 12 and posted August 20 is the authoritative witness for the stated planning cases, forecast boundaries, and dependency claims.",
        "The 150 GW, 159 GW, 125 GW, and 110 GW values retain their distinct years, methods, and roles rather than being treated as interchangeable forecasts or actual demand."
      ],
      "contradictions_or_limits": [
        "The approximately 150 GW and 159 GW values are Regional Transmission Plan assumptions; the approximately 125 GW value is a sensitivity; and the approximately 110 GW value is an econometric base-plus-mid-size forecast that excludes Batch-process large loads. None is actual measured load.",
        "The final total forecast including Batch-process loads remained future work expected by the end of 2026.",
        "A finding that the full STEP 765-kV plan is needed in studied cases is a planning result, not proof that transmission facilities were permitted, constructed, energized, or available.",
        "The response does not validate every tracked large-load request or convert queue nameplate into livable demand."
      ],
      "fak_implications": [
        "Forecast ledgers should retain forecast family, study year, target year, included load classes, and sensitivity status for every grid-demand number.",
        "Transmission-capacity models should separate recommended or planned facilities from permitted, constructed, energized, and operational assets.",
        "Batch-process loads should be added only through a dated final forecast or project-level evidence, not by combining raw queue nameplate with econometric demand."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserbench-steel-lifecycle-2026",
      "category": "workload_trace",
      "entity": "Steel BrowserBench / Steel",
      "topic": [
        "remote_browser",
        "browser_lifecycle_benchmark",
        "execution_envelope",
        "latency",
        "reliability"
      ],
      "published_at": "2026-01-15",
      "event_at": "2025-11-06",
      "source_title": "Steel BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020: README, runner, package lock, and results/steel.jsonl",
      "source_url": "https://github.com/steel-dev/browserbench/blob/847e4ed604764ce8b887265709ddb2d1c3c5f020/results/steel.jsonl",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "At pinned BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020, the committed steel result file records 5,000 sequential remote-browser lifecycle attempts during 2025-11-06T05:37:07.009Z to 2025-11-06T06:52:26.679Z. The repository README reports 5,000 successful samples (100%) with mean create, CDP connect, DOMContentLoaded navigation, release, and four-stage total latencies of 181.57, 174.64, 490.29, 47.62, and 894.13 ms. All 5,000 committed rows are successful and contain all four stage timings.",
      "quantified": {
        "repository_commit": "847e4ed604764ce8b887265709ddb2d1c3c5f020",
        "commit_authored_at": "2026-01-15T00:20:49-05:00",
        "runner_client_region": "AWS EC2 us-east-1",
        "navigation_target": "https://google.com/",
        "warmups_per_provider_excluded": 10,
        "provider_sdk_version": "steel-sdk 0.14.0",
        "test_window_utc": "2025-11-06T05:37:07.009Z to 2025-11-06T06:52:26.679Z",
        "attempts": 5000,
        "successful_lifecycle_samples": 5000,
        "success_rate": "100%",
        "failed_attempts": 0,
        "mean_session_creation_ms": 181.57,
        "mean_cdp_connect_ms": 174.64,
        "mean_domcontentloaded_navigation_ms": 490.29,
        "mean_session_release_ms": 47.62,
        "mean_total_create_connect_goto_release_ms": 894.13,
        "total_median_ms": 867.0,
        "total_p95_ms": 1090.0,
        "total_p99_ms": 1340.05
      },
      "assumptions": [
        "The repository tree, README, runner, package lock, and committed JSONL result at commit 847e4ed604764ce8b887265709ddb2d1c3c5f020 are treated as one pinned benchmark artifact.",
        "The README summary is treated as a reproducible description of the committed sample, not as live provider telemetry.",
        "Stage means and total quantiles retain the runner definitions and provider/test-window boundary; no cross-provider normalization beyond the repository methodology is inferred."
      ],
      "contradictions_or_limits": [
        "This is a benchmark lifecycle sample from the committed BrowserBench corpus, not production telemetry or a production agent trajectory.",
        "Create/start, browser readiness, CDP connection, first-page navigation, task completion, and release are distinct boundaries; the runner measures create, Playwright connectOverCDP, page.goto(..., waitUntil=\"domcontentloaded\"), and release, but does not establish a separate browser-ready time or task-complete time.",
        "The run count is attempts and the success count is successful lifecycle samples; neither is a count of users, user sessions, agent tasks, actions, observations, or tool calls.",
        "The sequential runner, Kernel retry-on-429 behavior, and Browserbase approximately three-second cycle floor do not measure provider concurrency, queue depth, or steady-state throughput. Attempt rate must not be relabeled as any of those quantities.",
        "Mean stage latency is not p50, p95, p99, or a failure distribution. The README total table supplies total median/p95/p99 only for this sample and does not expose stage-tail or retry distributions.",
        "One provider SDK version, test window, AWS us-east-1 client, target page, and repository commit do not establish universal provider performance.",
        "A hosted remote-browser lifecycle is not a desktop-computer trajectory and does not measure action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, or retry/failure/tail distributions for production browser or desktop agents."
      ],
      "fak_implications": [
        "Represent a remote-browser execution envelope as typed stage boundaries plus attempt/success accounting; do not collapse lifecycle latency into agent task latency.",
        "Preserve provider, SDK version, repository pin, test window, client region, target page, warm-up policy, throttling/retry behavior, and missing-stage values before using the sample for scheduling or timeout experiments.",
        "Use these bounded benchmark values only as lifecycle priors. Production browser/desktop capacity planning still requires trajectory-level action, tool-call, observation-byte, reset, session-duration, escalation, retry, failure, tail, and user-population measurements."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserbench-kernel-lifecycle-2026",
      "category": "workload_trace",
      "entity": "Steel BrowserBench / Kernel",
      "topic": [
        "remote_browser",
        "browser_lifecycle_benchmark",
        "execution_envelope",
        "latency",
        "reliability"
      ],
      "published_at": "2026-01-15",
      "event_at": "2026-01-11",
      "source_title": "Steel BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020: README, runner, package lock, and results/kernel.jsonl",
      "source_url": "https://github.com/steel-dev/browserbench/blob/847e4ed604764ce8b887265709ddb2d1c3c5f020/results/kernel.jsonl",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "At pinned BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020, the committed kernel result file records 5,000 sequential remote-browser lifecycle attempts during 2026-01-10T21:51:06.716Z to 2026-01-11T06:11:00.966Z. The repository README reports 5,000 successful samples (100%) with mean create, CDP connect, DOMContentLoaded navigation, release, and four-stage total latencies of 36.64, 288.30, 434.15, 34.77, and 793.84 ms. All 5,000 rows are marked successful; one successful row has a null release timing, so only 4,999 successful rows have all four stage values and support a complete stage-sum total.",
      "quantified": {
        "repository_commit": "847e4ed604764ce8b887265709ddb2d1c3c5f020",
        "commit_authored_at": "2026-01-15T00:20:49-05:00",
        "runner_client_region": "AWS EC2 us-east-1",
        "navigation_target": "https://google.com/",
        "warmups_per_provider_excluded": 10,
        "provider_sdk_version": "@onkernel/sdk 0.18.0",
        "test_window_utc": "2026-01-10T21:51:06.716Z to 2026-01-11T06:11:00.966Z",
        "attempts": 5000,
        "successful_lifecycle_samples": 5000,
        "success_rate": "100%",
        "failed_attempts": 0,
        "mean_session_creation_ms": 36.64,
        "mean_cdp_connect_ms": 288.3,
        "mean_domcontentloaded_navigation_ms": 434.15,
        "mean_session_release_ms": 34.77,
        "mean_total_create_connect_goto_release_ms": 793.84,
        "total_median_ms": 776.0,
        "total_p95_ms": 1006.0,
        "total_p99_ms": 1105.0
      },
      "assumptions": [
        "The repository tree, README, runner, package lock, and committed JSONL result at commit 847e4ed604764ce8b887265709ddb2d1c3c5f020 are treated as one pinned benchmark artifact.",
        "The README summary is treated as a reproducible description of the committed sample, not as live provider telemetry.",
        "Stage means and total quantiles retain the runner definitions and provider/test-window boundary; no cross-provider normalization beyond the repository methodology is inferred."
      ],
      "contradictions_or_limits": [
        "This is a benchmark lifecycle sample from the committed BrowserBench corpus, not production telemetry or a production agent trajectory.",
        "Create/start, browser readiness, CDP connection, first-page navigation, task completion, and release are distinct boundaries; the runner measures create, Playwright connectOverCDP, page.goto(..., waitUntil=\"domcontentloaded\"), and release, but does not establish a separate browser-ready time or task-complete time.",
        "The run count is attempts and the success count is successful lifecycle samples; neither is a count of users, user sessions, agent tasks, actions, observations, or tool calls.",
        "The sequential runner, Kernel retry-on-429 behavior, and Browserbase approximately three-second cycle floor do not measure provider concurrency, queue depth, or steady-state throughput. Attempt rate must not be relabeled as any of those quantities.",
        "Mean stage latency is not p50, p95, p99, or a failure distribution. The README total table supplies total median/p95/p99 only for this sample and does not expose stage-tail or retry distributions.",
        "One provider SDK version, test window, AWS us-east-1 client, target page, and repository commit do not establish universal provider performance.",
        "A hosted remote-browser lifecycle is not a desktop-computer trajectory and does not measure action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, or retry/failure/tail distributions for production browser or desktop agents."
      ],
      "fak_implications": [
        "Represent a remote-browser execution envelope as typed stage boundaries plus attempt/success accounting; do not collapse lifecycle latency into agent task latency.",
        "Preserve provider, SDK version, repository pin, test window, client region, target page, warm-up policy, throttling/retry behavior, and missing-stage values before using the sample for scheduling or timeout experiments.",
        "Use these bounded benchmark values only as lifecycle priors. Production browser/desktop capacity planning still requires trajectory-level action, tool-call, observation-byte, reset, session-duration, escalation, retry, failure, tail, and user-population measurements."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserbench-browserbase-lifecycle-2026",
      "category": "workload_trace",
      "entity": "Steel BrowserBench / Browserbase",
      "topic": [
        "remote_browser",
        "browser_lifecycle_benchmark",
        "execution_envelope",
        "latency",
        "reliability"
      ],
      "published_at": "2026-01-15",
      "event_at": "2026-01-13",
      "source_title": "Steel BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020: README, runner, package lock, and results/browserbase.jsonl",
      "source_url": "https://github.com/steel-dev/browserbench/blob/847e4ed604764ce8b887265709ddb2d1c3c5f020/results/browserbase.jsonl",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "At pinned BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020, the committed browserbase result file records 5,000 sequential remote-browser lifecycle attempts during 2026-01-12T23:08:19.110Z to 2026-01-13T07:29:07.295Z. The repository README reports 5,000 successful samples (100%) with mean create, CDP connect, DOMContentLoaded navigation, release, and four-stage total latencies of 212.83, 1,794.42, 745.08, 214.55, and 2,966.87 ms. All 5,000 committed rows are successful and contain all four stage timings; the runner enforces an approximately three-second floor per full Browserbase cycle to avoid rate-limit bursts.",
      "quantified": {
        "repository_commit": "847e4ed604764ce8b887265709ddb2d1c3c5f020",
        "commit_authored_at": "2026-01-15T00:20:49-05:00",
        "runner_client_region": "AWS EC2 us-east-1",
        "navigation_target": "https://google.com/",
        "warmups_per_provider_excluded": 10,
        "provider_sdk_version": "@browserbasehq/sdk 2.6.0",
        "test_window_utc": "2026-01-12T23:08:19.110Z to 2026-01-13T07:29:07.295Z",
        "attempts": 5000,
        "successful_lifecycle_samples": 5000,
        "success_rate": "100%",
        "failed_attempts": 0,
        "mean_session_creation_ms": 212.83,
        "mean_cdp_connect_ms": 1794.42,
        "mean_domcontentloaded_navigation_ms": 745.08,
        "mean_session_release_ms": 214.55,
        "mean_total_create_connect_goto_release_ms": 2966.87,
        "total_median_ms": 2888.0,
        "total_p95_ms": 3886.0,
        "total_p99_ms": 4309.12
      },
      "assumptions": [
        "The repository tree, README, runner, package lock, and committed JSONL result at commit 847e4ed604764ce8b887265709ddb2d1c3c5f020 are treated as one pinned benchmark artifact.",
        "The README summary is treated as a reproducible description of the committed sample, not as live provider telemetry.",
        "Stage means and total quantiles retain the runner definitions and provider/test-window boundary; no cross-provider normalization beyond the repository methodology is inferred."
      ],
      "contradictions_or_limits": [
        "This is a benchmark lifecycle sample from the committed BrowserBench corpus, not production telemetry or a production agent trajectory.",
        "Create/start, browser readiness, CDP connection, first-page navigation, task completion, and release are distinct boundaries; the runner measures create, Playwright connectOverCDP, page.goto(..., waitUntil=\"domcontentloaded\"), and release, but does not establish a separate browser-ready time or task-complete time.",
        "The run count is attempts and the success count is successful lifecycle samples; neither is a count of users, user sessions, agent tasks, actions, observations, or tool calls.",
        "The sequential runner, Kernel retry-on-429 behavior, and Browserbase approximately three-second cycle floor do not measure provider concurrency, queue depth, or steady-state throughput. Attempt rate must not be relabeled as any of those quantities.",
        "Mean stage latency is not p50, p95, p99, or a failure distribution. The README total table supplies total median/p95/p99 only for this sample and does not expose stage-tail or retry distributions.",
        "One provider SDK version, test window, AWS us-east-1 client, target page, and repository commit do not establish universal provider performance.",
        "A hosted remote-browser lifecycle is not a desktop-computer trajectory and does not measure action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, or retry/failure/tail distributions for production browser or desktop agents."
      ],
      "fak_implications": [
        "Represent a remote-browser execution envelope as typed stage boundaries plus attempt/success accounting; do not collapse lifecycle latency into agent task latency.",
        "Preserve provider, SDK version, repository pin, test window, client region, target page, warm-up policy, throttling/retry behavior, and missing-stage values before using the sample for scheduling or timeout experiments.",
        "Use these bounded benchmark values only as lifecycle priors. Production browser/desktop capacity planning still requires trajectory-level action, tool-call, observation-byte, reset, session-duration, escalation, retry, failure, tail, and user-population measurements."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserbench-hyperbrowser-lifecycle-2026",
      "category": "workload_trace",
      "entity": "Steel BrowserBench / Hyperbrowser",
      "topic": [
        "remote_browser",
        "browser_lifecycle_benchmark",
        "execution_envelope",
        "latency",
        "reliability"
      ],
      "published_at": "2026-01-15",
      "event_at": "2025-11-06",
      "source_title": "Steel BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020: README, runner, package lock, and results/hyperbrowser.jsonl",
      "source_url": "https://github.com/steel-dev/browserbench/blob/847e4ed604764ce8b887265709ddb2d1c3c5f020/results/hyperbrowser.jsonl",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "At pinned BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020, the committed hyperbrowser result file records 5,000 sequential remote-browser lifecycle attempts during 2025-11-05T16:34:26.511Z to 2025-11-06T10:44:52.875Z. The repository README reports 5,000 successful samples (100%) with mean create, CDP connect, DOMContentLoaded navigation, release, and four-stage total latencies of 1,731.63, 347.60, 377.89, 1,199.98, and 3,657.11 ms. All 5,000 committed rows are successful and contain all four stage timings.",
      "quantified": {
        "repository_commit": "847e4ed604764ce8b887265709ddb2d1c3c5f020",
        "commit_authored_at": "2026-01-15T00:20:49-05:00",
        "runner_client_region": "AWS EC2 us-east-1",
        "navigation_target": "https://google.com/",
        "warmups_per_provider_excluded": 10,
        "provider_sdk_version": "@hyperbrowser/sdk 0.71.0",
        "test_window_utc": "2025-11-05T16:34:26.511Z to 2025-11-06T10:44:52.875Z",
        "attempts": 5000,
        "successful_lifecycle_samples": 5000,
        "success_rate": "100%",
        "failed_attempts": 0,
        "mean_session_creation_ms": 1731.63,
        "mean_cdp_connect_ms": 347.6,
        "mean_domcontentloaded_navigation_ms": 377.89,
        "mean_session_release_ms": 1199.98,
        "mean_total_create_connect_goto_release_ms": 3657.11,
        "total_median_ms": 3665.5,
        "total_p95_ms": 5338.0,
        "total_p99_ms": 6695.05
      },
      "assumptions": [
        "The repository tree, README, runner, package lock, and committed JSONL result at commit 847e4ed604764ce8b887265709ddb2d1c3c5f020 are treated as one pinned benchmark artifact.",
        "The README summary is treated as a reproducible description of the committed sample, not as live provider telemetry.",
        "Stage means and total quantiles retain the runner definitions and provider/test-window boundary; no cross-provider normalization beyond the repository methodology is inferred."
      ],
      "contradictions_or_limits": [
        "This is a benchmark lifecycle sample from the committed BrowserBench corpus, not production telemetry or a production agent trajectory.",
        "Create/start, browser readiness, CDP connection, first-page navigation, task completion, and release are distinct boundaries; the runner measures create, Playwright connectOverCDP, page.goto(..., waitUntil=\"domcontentloaded\"), and release, but does not establish a separate browser-ready time or task-complete time.",
        "The run count is attempts and the success count is successful lifecycle samples; neither is a count of users, user sessions, agent tasks, actions, observations, or tool calls.",
        "The sequential runner, Kernel retry-on-429 behavior, and Browserbase approximately three-second cycle floor do not measure provider concurrency, queue depth, or steady-state throughput. Attempt rate must not be relabeled as any of those quantities.",
        "Mean stage latency is not p50, p95, p99, or a failure distribution. The README total table supplies total median/p95/p99 only for this sample and does not expose stage-tail or retry distributions.",
        "One provider SDK version, test window, AWS us-east-1 client, target page, and repository commit do not establish universal provider performance.",
        "A hosted remote-browser lifecycle is not a desktop-computer trajectory and does not measure action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, or retry/failure/tail distributions for production browser or desktop agents."
      ],
      "fak_implications": [
        "Represent a remote-browser execution envelope as typed stage boundaries plus attempt/success accounting; do not collapse lifecycle latency into agent task latency.",
        "Preserve provider, SDK version, repository pin, test window, client region, target page, warm-up policy, throttling/retry behavior, and missing-stage values before using the sample for scheduling or timeout experiments.",
        "Use these bounded benchmark values only as lifecycle priors. Production browser/desktop capacity planning still requires trajectory-level action, tool-call, observation-byte, reset, session-duration, escalation, retry, failure, tail, and user-population measurements."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserbench-anchorbrowser-lifecycle-2026",
      "category": "workload_trace",
      "entity": "Steel BrowserBench / AnchorBrowser",
      "topic": [
        "remote_browser",
        "browser_lifecycle_benchmark",
        "execution_envelope",
        "latency",
        "reliability"
      ],
      "published_at": "2026-01-15",
      "event_at": "2025-11-06",
      "source_title": "Steel BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020: README, runner, package lock, and results/anchorbrowser.jsonl",
      "source_url": "https://github.com/steel-dev/browserbench/blob/847e4ed604764ce8b887265709ddb2d1c3c5f020/results/anchorbrowser.jsonl",
      "source_kind": "official_repository",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "claim": "At pinned BrowserBench commit 847e4ed604764ce8b887265709ddb2d1c3c5f020, the committed anchorbrowser result file records 5,000 sequential remote-browser lifecycle attempts during 2025-11-05T17:35:22.143Z to 2025-11-06T09:05:04.543Z. The repository README reports 4,867 successful samples (97.34%) with mean create, CDP connect, DOMContentLoaded navigation, release, and four-stage total latencies of 3,796.55, 184.29, 1,259.66, 2,760.79, and 8,001.29 ms. The committed file contains 5,000 attempts: 4,867 successful complete lifecycle rows and 133 failures at session_create with no browser available.",
      "quantified": {
        "repository_commit": "847e4ed604764ce8b887265709ddb2d1c3c5f020",
        "commit_authored_at": "2026-01-15T00:20:49-05:00",
        "runner_client_region": "AWS EC2 us-east-1",
        "navigation_target": "https://google.com/",
        "warmups_per_provider_excluded": 10,
        "provider_sdk_version": "anchorbrowser 0.8.3",
        "test_window_utc": "2025-11-05T17:35:22.143Z to 2025-11-06T09:05:04.543Z",
        "attempts": 5000,
        "successful_lifecycle_samples": 4867,
        "success_rate": "97.34%",
        "failed_attempts": 133,
        "mean_session_creation_ms": 3796.55,
        "mean_cdp_connect_ms": 184.29,
        "mean_domcontentloaded_navigation_ms": 1259.66,
        "mean_session_release_ms": 2760.79,
        "mean_total_create_connect_goto_release_ms": 8001.29,
        "total_median_ms": 7919.0,
        "total_p95_ms": 11561.0,
        "total_p99_ms": 13957.14
      },
      "assumptions": [
        "The repository tree, README, runner, package lock, and committed JSONL result at commit 847e4ed604764ce8b887265709ddb2d1c3c5f020 are treated as one pinned benchmark artifact.",
        "The README summary is treated as a reproducible description of the committed sample, not as live provider telemetry.",
        "Stage means and total quantiles retain the runner definitions and provider/test-window boundary; no cross-provider normalization beyond the repository methodology is inferred."
      ],
      "contradictions_or_limits": [
        "This is a benchmark lifecycle sample from the committed BrowserBench corpus, not production telemetry or a production agent trajectory.",
        "Create/start, browser readiness, CDP connection, first-page navigation, task completion, and release are distinct boundaries; the runner measures create, Playwright connectOverCDP, page.goto(..., waitUntil=\"domcontentloaded\"), and release, but does not establish a separate browser-ready time or task-complete time.",
        "The run count is attempts and the success count is successful lifecycle samples; neither is a count of users, user sessions, agent tasks, actions, observations, or tool calls.",
        "The sequential runner, Kernel retry-on-429 behavior, and Browserbase approximately three-second cycle floor do not measure provider concurrency, queue depth, or steady-state throughput. Attempt rate must not be relabeled as any of those quantities.",
        "Mean stage latency is not p50, p95, p99, or a failure distribution. The README total table supplies total median/p95/p99 only for this sample and does not expose stage-tail or retry distributions.",
        "One provider SDK version, test window, AWS us-east-1 client, target page, and repository commit do not establish universal provider performance.",
        "A hosted remote-browser lifecycle is not a desktop-computer trajectory and does not measure action/tool-call distributions, observation bytes, resets, full session duration, escalations, user populations, or retry/failure/tail distributions for production browser or desktop agents."
      ],
      "fak_implications": [
        "Represent a remote-browser execution envelope as typed stage boundaries plus attempt/success accounting; do not collapse lifecycle latency into agent task latency.",
        "Preserve provider, SDK version, repository pin, test window, client region, target page, warm-up policy, throttling/retry behavior, and missing-stage values before using the sample for scheduling or timeout experiments.",
        "Use these bounded benchmark values only as lifecycle priors. Production browser/desktop capacity planning still requires trajectory-level action, tool-call, observation-byte, reset, session-duration, escalation, retry, failure, tail, and user-population measurements."
      ],
      "rumor": {
        "is_rumor": false
      },
      "checked_at": "2026-08-27"
    },
    {
      "id": "browserops-steel-plan-envelope-2026",
      "category": "workload_model",
      "entity": "Steel",
      "topic": "remote-browser operational concurrency, rate, and maximum-session plan envelope",
      "published_at": "2026-06-30",
      "event_at": "2026-06-30",
      "source_title": "Steel Pricing/Limits",
      "source_url": "https://docs.steel.dev/overview/pricinglimits",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Steel documents plan-level concurrent-browser, request-rate, and maximum-session allowances: Launch 10 concurrent sessions, 60 requests/minute, and 15-minute maximum sessions; Scale 100, 600 requests/minute, and 1 hour; Enterprise 1,000+ concurrent sessions, custom requests/minute, and up to 24 hours.",
      "quantified": {
        "launch_concurrent_browser_sessions": 10,
        "scale_concurrent_browser_sessions": 100,
        "enterprise_concurrent_browser_sessions_minimum": 1000,
        "launch_requests_per_minute": 60,
        "scale_requests_per_minute": 600,
        "launch_max_session_minutes": 15,
        "scale_max_session_minutes": 60,
        "enterprise_max_session_hours_upper_bound": 24,
        "accessed_at": "2026-08-27"
      },
      "assumptions": [
        "The pricing page is a current official plan specification; its documented allowances are not measurements of observed concurrency or throughput."
      ],
      "contradictions_or_limits": [
        "Requests per minute is a rate boundary, not a concurrency count.",
        "1,000+ is an open-ended Enterprise allowance, not a measured production population.",
        "Maximum session time is a lifecycle ceiling, not task duration.",
        "The page does not document queue behavior when a concurrency allowance is exhausted."
      ],
      "fak_implications": [
        "Keep remote-browser worker admission plan-aware and represent request-rate and running-session limits as separate resources.",
        "Do not derive browser-task throughput or production population from plan allowances."
      ],
      "rumor": false
    },
    {
      "id": "browserops-browserbase-admission-envelope-2026",
      "category": "workload_model",
      "entity": "Browserbase",
      "topic": "remote-browser concurrency, creation-rate admission, and timeout boundaries",
      "published_at": "2026-08-27",
      "event_at": "2026-08-27",
      "source_title": "Browserbase Concurrency management, Timeouts, and Manage a browser session",
      "source_url": "https://docs.browserbase.com/optimizations/concurrency/overview.md",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Browserbase publishes separate plan boundaries for maximum concurrently active browsers (Free 3, Developer 25, Startup 100, Scale 250+) and session creations per 60-second period (Free 5, Developer 25, Startup 50, Scale 150+). Reaching either limit returns HTTP 429; over-limit session creation is described as effectively dropped rather than queued. Session duration uses a configurable project default and has a 6-hour maximum. Separately, a CDP connection is closed after 10 minutes without CDP commands; that connection-inactivity bound is not the session-duration limit.",
      "quantified": {
        "max_concurrent_browsers_by_plan": {
          "Free": 3,
          "Developer": 25,
          "Startup": 100,
          "Scale": "250+"
        },
        "session_creation_limit_per_minute_by_plan": {
          "Free": 5,
          "Developer": 25,
          "Startup": 50,
          "Scale": "150+"
        },
        "over_limit_http_status": 429,
        "over_limit_creation_disposition": "effectively dropped",
        "maximum_session_duration_hours": 6,
        "cdp_inactivity_timeout_minutes": 10,
        "project_default_session_timeout": "configurable",
        "accessed_at": "2026-08-27",
        "timeout_source": "https://docs.browserbase.com/platform/browser/long-sessions/timeouts.md",
        "manage_session_source": "https://docs.browserbase.com/platform/browser/getting-started/manage-browser-session.md"
      },
      "assumptions": [
        "The two cited official documentation pages describe the hosted Browserbase service as accessed on 2026-08-27."
      ],
      "contradictions_or_limits": [
        "Active-browser concurrency and session creations per minute are distinct limits and must not be conflated.",
        "HTTP 429 is an admission refusal; over-limit session creation is described as dropped, not provider-side queued execution.",
        "The project default session duration is configurable; the cited pages do not support a universal 5-minute default.",
        "The 10-minute CDP inactivity timeout governs a connection with no CDP commands, not total browser-session duration.",
        "Session-duration settings are lifecycle bounds, not task-duration distributions or throughput rates."
      ],
      "fak_implications": [
        "Represent active concurrency and per-minute creation rate as separate Browserbase admission controls.",
        "Treat HTTP 429 as explicit backpressure and honor retry guidance rather than assuming a provider-side queue.",
        "Configure session duration independently from CDP connection heartbeat or inactivity handling."
      ],
      "rumor": false
    },
    {
      "id": "browserops-kernel-idle-and-pool-envelope-2026",
      "category": "workload_model",
      "entity": "Kernel",
      "topic": "remote-browser standby timeout and pool long-poll admission envelope",
      "published_at": "2026-08-27",
      "event_at": "2026-08-27",
      "source_title": "Kernel Termination & Timeouts and Acquire a browser from the pool",
      "source_url": "https://kernel.sh/docs/browsers/termination.md",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Kernel documents a standby-triggered browser deletion timeout of 60 seconds by default, configurable up to 72 hours. Its pool acquire API is a long-poll: it returns immediately when a browser is available or HTTP 204 when the poll times out, after which the client should retry.",
      "quantified": {
        "default_standby_timeout_seconds": 60,
        "maximum_standby_timeout_hours": 72,
        "pool_poll_timeout_http_status": 204,
        "accessed_at": "2026-08-27",
        "companion_pool_source": "https://kernel.sh/docs/api-reference/browser-pools/acquire-a-browser-from-the-pool.md"
      },
      "assumptions": [
        "The termination page and pool API reference are current official Kernel documentation as accessed on 2026-08-27."
      ],
      "contradictions_or_limits": [
        "The 60-second timer starts only after standby; it is not a total task-duration limit.",
        "A long-polling acquisition endpoint is an admission/wait mechanism, not evidence that queued requests are running.",
        "No public numeric organization or project concurrency quantity was found in these pages."
      ],
      "fak_implications": [
        "Model standby reclamation separately from task deadlines.",
        "Count a pool request as waiting until acquisition succeeds; retry after 204 without recording a running browser."
      ],
      "rumor": false
    },
    {
      "id": "browserops-hyperbrowser-configuration-lifecycle-example-2026",
      "category": "workload_model",
      "entity": "Hyperbrowser",
      "topic": "remote-browser timeout configuration/lifecycle example",
      "published_at": "2026-08-27",
      "event_at": "2026-08-27",
      "source_title": "Hyperbrowser Session Timeouts",
      "source_url": "https://hyperbrowser.ai/docs/guides/session-timeouts",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Hyperbrowser documents that each browser session uses the team default session timeout unless the request supplies timeoutMinutes; the official configuration example sets 60 minutes. The reviewed public guide does not state a default value, maximum value, concurrency allowance, queueing rule, or API rate boundary.",
      "quantified": {
        "documented_timeout_example_minutes": 60,
        "accessed_at": "2026-08-27"
      },
      "assumptions": [
        "Treat the 60-minute code sample only as an example accepted by the documented API, not as a default, maximum, plan allowance, or observed task duration."
      ],
      "contradictions_or_limits": [
        "The 60-minute value is only a configuration example, not a default, maximum, plan allowance, marketing concurrency claim, or observed task duration.",
        "No public numeric Hyperbrowser concurrency, queue depth/wait, request-rate boundary, default timeout, or maximum timeout was directly supported by the official guide reviewed.",
        "A configurable session timeout is a lifecycle control, not evidence of task duration or achieved concurrent use."
      ],
      "fak_implications": [
        "Keep Hyperbrowser concurrency, admission/queueing, rate, default-timeout, and maximum-timeout cells unknown until primary evidence supplies them.",
        "If a benchmark sets timeoutMinutes explicitly, record that configured value separately from measured task duration and termination cause."
      ],
      "rumor": false
    },
    {
      "id": "browserops-anchor-timeout-envelope-2026",
      "category": "workload_model",
      "entity": "Anchor Browser",
      "topic": "remote-browser idle and maximum-duration timeout envelope",
      "published_at": "2026-08-27",
      "event_at": "2026-08-27",
      "source_title": "Anchor Browser Session Timeout",
      "source_url": "https://docs.anchorbrowser.io/advanced/session-timeout.md",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Anchor Browser documents two independent session timers: idle_timeout defaults to 5 minutes after the last browser connection disconnects and may be disabled with -1; max_duration defaults to 180 minutes, has no documented upper limit, and terminates the session regardless of activity. The first condition reached ends the session.",
      "quantified": {
        "default_idle_timeout_minutes": 5,
        "idle_timeout_disable_value": -1,
        "default_max_duration_minutes": 180,
        "documented_max_duration_upper_limit": null,
        "accessed_at": "2026-08-27"
      },
      "assumptions": [
        "The official timeout page defines hosted-session lifecycle semantics as accessed on 2026-08-27."
      ],
      "contradictions_or_limits": [
        "Idle timeout starts after disconnection and is not task duration.",
        "No upper limit is a documentation statement, not evidence of infinite practical duration.",
        "No robust public numeric concurrency, request-rate, or queue boundary was found in the reviewed official pages."
      ],
      "fak_implications": [
        "Track Anchor idle reclamation and hard lifetime as separate timers.",
        "Do not use either timeout as a workload-duration distribution."
      ],
      "rumor": false
    },
    {
      "id": "google-gemini-app-user-scale-2026",
      "category": "frontier_lab",
      "entity": "Google",
      "topic": [
        "userbase",
        "developer",
        "enterprise",
        "traffic",
        "demand"
      ],
      "published_at": "2026-07-22",
      "event_at": "2026-07-22",
      "source_title": "Alphabet 2026 Q2 earnings call",
      "source_url": "https://abc.xyz/investor/events/event-details/2026/2026-Q2-Earnings-Call-2026-GgTAq7Is0z/default.aspx",
      "source_kind": "official_transcript",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Alphabet reported 950 million Gemini app monthly active users, more than 2.4 million Antigravity weekly active users, more than 9 million monthly developers across its model APIs and key developer products, approximately 22 billion model-API tokens processed per minute, and two Google Cloud paid-token customer cohorts.",
      "quantified": {
        "gemini_app_monthly_active_users": 950000000,
        "gemini_app_daily_active_users_yoy_multiple": 3,
        "antigravity_weekly_active_users_gt": 2400000,
        "model_api_and_key_developer_product_monthly_developers_gt": 9000000,
        "model_api_tokens_per_minute_approx": 22000000000,
        "google_cloud_customers_tokens_prior_year_gt_1t_approx": 500,
        "google_cloud_enterprises_tokens_prior_12_months_gt_100b": 2000
      },
      "assumptions": [
        "Each quantity has its own product, time-window, and unit boundary: Gemini app MAU, Antigravity WAU, monthly developers across model APIs and key developer products, model-API tokens per minute, and Google Cloud customer/enterprise token-threshold cohorts."
      ],
      "contradictions_or_limits": [
        "The daily-active disclosure is a growth multiple without a daily-active population count.",
        "Gemini app MAU excludes or does not separately identify AI Studio, Gemini API, Workspace, Search AI features, or model-specific traffic; Antigravity WAU and the monthly developer population are separate product scopes.",
        "Model-API tokens per minute are aggregate throughput, not requests, queries, sessions, messages, users, concurrency, geography, per-model allocation, or an interarrival distribution.",
        "The Cloud token cohorts count customers or enterprises crossing thresholds, not total customers, total tokens, tokens per minute, requests, sessions, messages, concurrent users, or a traffic distribution."
      ],
      "fak_implications": [
        "Keep Gemini app MAU, Antigravity WAU, monthly developers, model-API token throughput, and Google Cloud threshold cohorts as separate typed denominators.",
        "Use the token-per-minute disclosure only as an aggregate provider model-API throughput point; do not derive request rate, concurrency, geography, model mix, or traffic shape.",
        "Threshold cohorts can bound the existence of very large enterprise token consumers, but do not identify their exact traffic, concentration, geography, or arrival process."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "meta-ai-user-scale-2025",
      "category": "frontier_lab",
      "entity": "Meta",
      "topic": [
        "userbase",
        "consumer",
        "demand"
      ],
      "published_at": "2025-04-30",
      "event_at": "2025-04-30",
      "source_title": "Meta Reports First Quarter 2025 Results",
      "source_url": "https://investor.atmeta.com/investor-news/press-release-details/2025/Meta-Reports-First-Quarter-2025-Results/default.aspx",
      "source_kind": "official_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Meta reported that Meta AI had almost 1 billion monthly actives.",
      "quantified": {
        "meta_ai_monthly_actives_approx": 1000000000
      },
      "assumptions": [
        "The disclosure establishes an approximate monthly-actives population for Meta AI across the product scope Meta used in the release; it does not define the counted person or account unit."
      ],
      "contradictions_or_limits": [
        "\"Almost 1 billion\" is approximate and does not state the exact count, whether the unit is people or accounts, or the inclusion rules across Meta applications and surfaces.",
        "Meta AI monthly actives are not Meta Family daily active people, registered accounts, requests, sessions, messages, tokens, model-specific Llama traffic, or concurrency."
      ],
      "fak_implications": [
        "Preserve the approximate qualifier and Meta AI product scope.",
        "Do not infer per-user usage, geography, request rate, concurrency, or Llama model share from the monthly-active population."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "microsoft-365-copilot-seat-scale-2026",
      "category": "frontier_lab",
      "entity": "Microsoft",
      "topic": [
        "paid_seats",
        "userbase",
        "developer",
        "enterprise",
        "traffic",
        "demand"
      ],
      "published_at": "2026-07-29",
      "event_at": "2026-07-29",
      "source_title": "Microsoft fiscal year 2026 fourth quarter earnings conference call",
      "source_url": "https://www.microsoft.com/en-us/investor/events/fy-2026/earnings-fy-2026-q4",
      "source_kind": "official_transcript",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Microsoft reported more than 30 million paid Microsoft 365 Copilot seats, 50 million GitHub Copilot users, 100,000 Microsoft Foundry customers, nearly 40 million agents registered in Agent 365 across tens of thousands of companies, and more than 50 billion Copilot interactions audited by Purview to date.",
      "quantified": {
        "microsoft_365_copilot_paid_seats_gt": 30000000,
        "microsoft_365_copilot_net_seat_adds_qoq_multiple_gt": 2,
        "github_copilot_users": 50000000,
        "microsoft_foundry_customers": 100000,
        "agent_365_registered_agents_approx": 40000000,
        "agent_365_company_count": "tens_of_thousands",
        "purview_copilot_interactions_audited_to_date_gt": 50000000000
      },
      "assumptions": [
        "The quantities describe separate products and denominator types: paid Microsoft 365 Copilot entitlements, GitHub Copilot users, Microsoft Foundry customers, registered Agent 365 agents and companies, and cumulative Copilot interactions audited by Purview."
      ],
      "contradictions_or_limits": [
        "Paid seats are not necessarily active people; GitHub Copilot users are not identified as paid seats or active users; Foundry customers are not defined as organizations, developers, or active tenants; registered agents are not active or concurrent agents.",
        "Purview-audited Copilot interactions are a cumulative governance/audit scope, not all Copilot interactions, a time-normalized request rate, sessions, messages, tokens, unique users, or model-specific traffic.",
        "None is provider-wide Copilot usage, and the source does not disclose request interarrival, seat utilization, tenant concentration, geography, or model routing for these populations."
      ],
      "fak_implications": [
        "Model each Microsoft product population and audited-interaction quantity separately, preserving seat, user, customer, registered-agent, company, and cumulative-interaction semantics.",
        "Do not treat population growth, registered agents, or cumulative audited interactions as evidence of current request rate, concurrency, or workload distribution."
      ],
      "rumor": {
        "is_rumor": false
      }
    },
    {
      "id": "mlperf-v51-nvidia-b200-deepseek-r1-server-2025",
      "entity": "MLCommons / NVIDIA",
      "category": "serving_system",
      "topic": "MLPerf Inference v5.1 DeepSeek-R1 server topology and batching",
      "claim": "At the pinned MLCommons v5.1 commit, NVIDIA submitted a valid DeepSeek-R1 result in the Server submission directory on one DGX B200 system with eight B200-SXM-180GB accelerators. The result completed 4.96 samples/s; the LLM summary also reports 18,592.24 completed tokens/s. The paired NVIDIA config explicitly sets FP4 model precision, FP8 KV cache, TP=8, PP=1, MoE EP=8, a configured GPU batch-size ceiling of 512, max input length 3,140, and max sequence length 23,140.",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "published_at": "2025-09-09",
      "event_at": "2025-07-28",
      "source_kind": "official_repository",
      "source_title": "MLPerf Inference v5.1 NVIDIA B200 DeepSeek-R1 Server result, config, and system description, commit 5ea4f62e",
      "source_url": "https://github.com/mlcommons/inference_results_v5.1/blob/5ea4f62ef62536e6bf4d78a9b440fb9035ddfb4a/closed/NVIDIA/results/B200-SXM-180GBx8_TRT/deepseek-r1/Server/performance/run_1/mlperf_log_summary.txt",
      "quantified": {
        "model": "DeepSeek-R1",
        "submission_scenario": "Server",
        "loadgen_scenario": "Server",
        "result_status": "VALID",
        "completed_samples_per_second": 4.96,
        "scheduled_samples_per_second": 5.02,
        "completed_tokens_per_second": 18592.24,
        "result_test_datetime_utc": "2025-07-28T23:33:23Z",
        "system_name": "NVIDIA DGX B200 (8x B200-SXM-180GB, TensorRT)",
        "host_count": 1,
        "accelerator_model": "NVIDIA B200-SXM-180GB",
        "accelerator_count": 8,
        "accelerator_memory": "180 GB each",
        "framework": "TensorRT 10.11, CUDA 12.9",
        "operating_system": "Ubuntu 24.04",
        "model_precision": "FP4",
        "kv_cache_dtype": "FP8",
        "tensor_parallelism": 8,
        "pipeline_parallelism": 1,
        "moe_expert_parallelism": 8,
        "configured_gpu_batch_size_ceiling": 512,
        "configured_max_input_length": 3140,
        "configured_max_sequence_length": 23140
      },
      "assumptions": [
        "Explicit result fields come from closed/NVIDIA/results/B200-SXM-180GBx8_TRT/deepseek-r1/Server/performance/run_1/mlperf_log_summary.txt and mlperf_log_detail.txt at commit 5ea4f62e.",
        "Explicit topology and runtime fields come from closed/NVIDIA/systems/B200-SXM-180GBx8_TRT.json; explicit precision, parallelism, and configured ceilings come from closed/NVIDIA/configs/B200-SXM-180GBx8/Server/deepseek-r1.py at the same commit.",
        "The configured GPU batch size is a ceiling/tuning control, not evidence that 512 requests were simultaneously active. No replica count is stated.",
        "published_at is the MLPerf Inference v5.1 result-release date; the cited pinned repository commit itself is dated 2026-01-16."
      ],
      "contradictions_or_limits": [
        "This is one audited benchmark submission, not a production deployment or a fleet-size census.",
        "Accelerator count is not host or fleet count; TP/PP/EP describe one execution topology and do not establish replica count.",
        "Completed samples/s is the MLPerf result unit. Completed tokens/s is a separate LLM summary metric and must not be substituted for QPS or samples/s.",
        "The paired config states max input and max sequence lengths, but not a separate max-output-length field; no output length is inferred by subtraction.",
        "The pinned commit postdates the test run; published_at records the pinned commit date, while event_at and result_test_datetime_utc record the run."
      ],
      "fak_implications": [
        "Preserve topology, configured batching, offered QPS, latency gate, and achieved token goodput as separate benchmark dimensions.",
        "Use the pinned config as a reproducible stress point, not as a prevalence prior."
      ],
      "rumor": false
    },
    {
      "id": "mlperf-v51-nvidia-b200-llama31-405b-interactive-2025",
      "entity": "MLCommons / NVIDIA",
      "category": "serving_system",
      "topic": "MLPerf Inference v5.1 Llama 3.1 405B interactive topology and batching",
      "claim": "At the pinned MLCommons v5.1 commit, NVIDIA submitted a valid Llama 3.1 405B result under the Interactive submission directory on one DGX B200 system with eight B200-SXM-180GB accelerators. LoadGen records the executed scenario as Server; the Interactive submission applies LLM TTFT/TPOT latency constraints. The result completed 1.15 samples/s and 751.37 completed tokens/s. The paired config explicitly sets FP4 model precision, TP=4, PP=1, a configured GPU batch-size ceiling of 256, max input length 20,000, and max sequence length 22,000.",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "published_at": "2025-09-09",
      "event_at": "2025-07-22",
      "source_kind": "official_repository",
      "source_title": "MLPerf Inference v5.1 NVIDIA B200 Llama 3.1 405B Interactive result, config, and system description, commit 5ea4f62e",
      "source_url": "https://github.com/mlcommons/inference_results_v5.1/blob/5ea4f62ef62536e6bf4d78a9b440fb9035ddfb4a/closed/NVIDIA/results/B200-SXM-180GBx8_TRT/llama3.1-405b/Interactive/performance/run_1/mlperf_log_summary.txt",
      "quantified": {
        "model": "Llama 3.1 405B",
        "submission_scenario": "Interactive",
        "loadgen_scenario": "Server",
        "result_status": "VALID",
        "completed_samples_per_second": 1.15,
        "scheduled_samples_per_second": 1.15,
        "completed_tokens_per_second": 751.37,
        "result_test_datetime_utc": "2025-07-22T11:46:29Z",
        "ttft_99th_percentile_ms": 2510.581561,
        "tpot_99th_percentile_ms": 77.696851,
        "ttft_99th_percentile_constraint_ms": 4500,
        "tpot_99th_percentile_constraint_ms": 80,
        "system_name": "NVIDIA DGX B200 (8x B200-SXM-180GB, TensorRT)",
        "host_count": 1,
        "accelerator_model": "NVIDIA B200-SXM-180GB",
        "accelerator_count": 8,
        "accelerator_memory": "180 GB each",
        "framework": "TensorRT 10.11, CUDA 12.9",
        "operating_system": "Ubuntu 24.04",
        "model_precision": "FP4",
        "tensor_parallelism": 4,
        "pipeline_parallelism": 1,
        "configured_gpu_batch_size_ceiling": 256,
        "configured_max_input_length": 20000,
        "configured_max_sequence_length": 22000
      },
      "assumptions": [
        "Explicit result and latency fields come from closed/NVIDIA/results/B200-SXM-180GBx8_TRT/llama3.1-405b/Interactive/performance/run_1/mlperf_log_summary.txt and mlperf_log_detail.txt at commit 5ea4f62e.",
        "Explicit topology and runtime fields come from closed/NVIDIA/systems/B200-SXM-180GBx8_TRT.json; explicit precision, parallelism, and configured ceilings come from closed/NVIDIA/configs/B200-SXM-180GBx8/Interactive/llama3_1-405b.py at the same commit.",
        "Interactive is the submission-directory taxonomy; the official LoadGen log itself says Scenario: Server and supplies TTFT/TPOT constraints.",
        "The configured GPU batch size is a ceiling/tuning control, not achieved active batch. TP=4 across an eight-accelerator system does not prove two replicas, so replica count is omitted.",
        "published_at is the MLPerf Inference v5.1 result-release date; the cited pinned repository commit itself is dated 2026-01-16."
      ],
      "contradictions_or_limits": [
        "This is one audited benchmark submission, not a production deployment or evidence that Interactive is the common production scenario.",
        "Accelerator count is not host or fleet count; TP/PP do not establish replica count.",
        "Completed samples/s is the MLPerf result unit. Completed tokens/s is separate and must not be substituted for QPS or samples/s.",
        "The config states max input and max sequence lengths, but no separate max-output-length field; no output length is inferred.",
        "The pinned commit postdates the test run; published_at records the pinned commit date, while event_at and result_test_datetime_utc record the run."
      ],
      "fak_implications": [
        "A single accelerator count can coexist with a TP value smaller than device count without proving replicas; retain an explicit unknown replica dimension.",
        "Benchmark interactive token goodput should remain distinct from tail-latency compliance and production concurrency."
      ],
      "rumor": false
    },
    {
      "id": "mlperf-v60-nvidia-b300-deepseek-r1-server-2026",
      "entity": "MLCommons / NVIDIA",
      "category": "serving_system",
      "topic": "MLPerf Inference v6.0 DeepSeek-R1 server topology and batching",
      "claim": "At the pinned MLCommons v6.0 commit, NVIDIA submitted a valid DeepSeek-R1 Server result on one DGX B300 system with eight B300-SXM-270GB accelerators. The result completed 11.39 samples/s; the LLM summary also reports 42,721.39 completed tokens/s. The paired config explicitly sets FP4 model precision, FP8 KV cache, TP=8, PP=1, MoE EP=8, a configured GPU batch-size ceiling of 640, max input length 3,140, and max sequence length 35,908.",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "published_at": "2026-04-02",
      "event_at": "2026-02-05",
      "source_kind": "official_repository",
      "source_title": "MLPerf Inference v6.0 NVIDIA B300 DeepSeek-R1 Server result, config, and system description, commit 4d3916ac",
      "source_url": "https://github.com/mlcommons/inference_results_v6.0/blob/4d3916ac9cf474b679cdfcf492d43a0559418ad1/closed/NVIDIA/results/B300-SXM-270GBx8_TRT/deepseek-r1/Server/performance/run_1/mlperf_log_summary.txt",
      "quantified": {
        "model": "DeepSeek-R1",
        "submission_scenario": "Server",
        "loadgen_scenario": "Server",
        "result_status": "VALID",
        "completed_samples_per_second": 11.39,
        "scheduled_samples_per_second": 11.53,
        "completed_tokens_per_second": 42721.39,
        "result_test_datetime_utc": "2026-02-05T14:35:09Z",
        "system_name": "NVIDIA DGX B300 (8x B300-SXM-270GB, TensorRT)",
        "host_count": 1,
        "accelerator_model": "NVIDIA B300-SXM-270GB",
        "accelerator_count": 8,
        "accelerator_memory": "270 GB each",
        "framework": "TensorRT 10.14, CUDA 13.1, cuDNN 9.17, TensorRT-LLM feat/1.2-mlpinf, NVIDIA Dynamo mlperf-v6.0-dynamo-v0.8.0, vLLM CentML:mlperf-inf-mm-q3vl-v6.0",
        "operating_system": "Ubuntu 24.04",
        "model_precision": "FP4",
        "kv_cache_dtype": "FP8",
        "tensor_parallelism": 8,
        "pipeline_parallelism": 1,
        "moe_expert_parallelism": 8,
        "configured_gpu_batch_size_ceiling": 640,
        "configured_max_input_length": 3140,
        "configured_max_sequence_length": 35908
      },
      "assumptions": [
        "Explicit result fields come from closed/NVIDIA/results/B300-SXM-270GBx8_TRT/deepseek-r1/Server/performance/run_1/mlperf_log_summary.txt and mlperf_log_detail.txt at commit 4d3916ac.",
        "Explicit topology and runtime fields come from closed/NVIDIA/systems/B300-SXM-270GBx8_TRT.json; explicit precision, parallelism, and configured ceilings come from closed/NVIDIA/configs/B300-SXM-270GBx8/Server/deepseek-r1.py at the same commit.",
        "The configured GPU batch size is a ceiling/tuning control, not evidence that 640 requests were simultaneously active. No replica count is stated.",
        "published_at is the MLPerf Inference v6.0 result-release date; the cited pinned repository commit itself is dated 2026-04-03."
      ],
      "contradictions_or_limits": [
        "This is one audited benchmark submission, not a production deployment, fleet-size census, or evidence that the best submitted result is prevalent.",
        "Accelerator count is not host or fleet count; TP/PP/EP describe one execution topology and do not establish replica count.",
        "Completed samples/s is the MLPerf result unit. Completed tokens/s is a separate LLM summary metric and must not be substituted for QPS or samples/s.",
        "The paired config states max input and max sequence lengths, but not a separate max-output-length field; no output length is inferred by subtraction.",
        "The system framework string lists multiple components used by the submission; it does not prove every component was on the DeepSeek-R1 execution path beyond the paired submission metadata.",
        "The pinned commit postdates the test run; published_at records the pinned commit date, while event_at and result_test_datetime_utc record the run."
      ],
      "fak_implications": [
        "Track software release, topology, batch ceiling, concurrency ceiling, offered load, and achieved token rate independently across benchmark generations.",
        "Do not convert a best submission into a default hardware or deployment assumption."
      ],
      "rumor": false
    },
    {
      "id": "mlperf-v60-amd-mi355x-llama31-405b-interactive-2026",
      "entity": "MLCommons / AMD",
      "category": "serving_system",
      "topic": "MLPerf Inference v6.0 Llama 3.1 405B Interactive on MI355X",
      "claim": "At the pinned MLCommons v6.0 commit, AMD submitted a valid Llama 3.1 405B result under the Interactive submission directory on one system with eight AMD Instinct MI355X 288GB accelerators. LoadGen records the executed scenario as Server; the Interactive submission applies LLM TTFT/TPOT latency constraints. The result completed 1.04 samples/s and 793.99 completed tokens/s. The checked-in system description explicitly lists the accelerator, host count, PyTorch/ROCm framework string, and Ubuntu version; no result-linked config file at the pinned commit establishes precision, TP/PP/EP, replicas, batch/concurrency ceilings, or sequence-length controls.",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "published_at": "2026-04-02",
      "event_at": "2026-02-10",
      "source_kind": "official_repository",
      "source_title": "MLPerf Inference v6.0 AMD MI355X Llama 3.1 405B Interactive result and system description, commit 4d3916ac",
      "source_url": "https://github.com/mlcommons/inference_results_v6.0/blob/4d3916ac9cf474b679cdfcf492d43a0559418ad1/open/AMD/results/8xMI355X_2xEPYC_9575F/llama3.1-405b/Interactive/performance/run_1/mlperf_log_summary.txt",
      "quantified": {
        "model": "Llama 3.1 405B",
        "submission_scenario": "Interactive",
        "loadgen_scenario": "Server",
        "result_status": "VALID",
        "completed_samples_per_second": 1.04,
        "scheduled_samples_per_second": 1.04,
        "completed_tokens_per_second": 793.99,
        "result_test_datetime_utc": "2026-02-10T10:17:52Z",
        "ttft_99th_percentile_ms": 2824.777998,
        "tpot_99th_percentile_ms": 73.35065,
        "ttft_99th_percentile_constraint_ms": 4500,
        "tpot_99th_percentile_constraint_ms": 80,
        "system_name": "smci355-ccs-aus-m09-17 (AS -4126GS-NMR-LCC)",
        "host_count": 1,
        "accelerator_model": "AMD Instinct MI355X 288GB HBM3e",
        "accelerator_count": 8,
        "accelerator_memory": "288 GB each",
        "framework": "PyTorch 2.9.1+git8907517, ROCm 7.0.0, PyTorch 2.9.0a0+git1c57644, ROCm 7.1.0",
        "operating_system": "Ubuntu 22.04.5 LTS"
      },
      "assumptions": [
        "Explicit result and latency fields come from open/AMD/results/8xMI355X_2xEPYC_9575F/llama3.1-405b/Interactive/performance/run_1/mlperf_log_summary.txt and mlperf_log_detail.txt at commit 4d3916ac.",
        "Explicit system fields come from open/AMD/systems/8xMI355X_2xEPYC_9575F.json at the same commit.",
        "Interactive is the submission-directory taxonomy; the official LoadGen log itself says Scenario: Server and supplies TTFT/TPOT constraints.",
        "No checked-in AMD measurement/config file tied to this result was found at the pinned commit, so precision, TP/PP/EP, replicas, batch/concurrency controls, and sequence-length controls are intentionally omitted.",
        "published_at is the MLPerf Inference v6.0 result-release date; the cited pinned repository commit itself is dated 2026-04-03."
      ],
      "contradictions_or_limits": [
        "This is one audited benchmark submission, not a production deployment, fleet-size census, or prevalence claim.",
        "Accelerator count is not host or fleet count. No topology or replica quantity is inferred from the eight-accelerator system name.",
        "Completed samples/s is the MLPerf result unit. Completed tokens/s is separate and must not be substituted for QPS or samples/s.",
        "The system framework string contains two PyTorch/ROCm pairs; the source does not disambiguate which pair served this exact result.",
        "The pinned commit postdates the test run; published_at records the pinned commit date, while event_at and result_test_datetime_utc record the run."
      ],
      "fak_implications": [
        "Represent high-quality benchmark rows even when topology fields are explicitly unknown, so cross-vendor comparisons do not silently fill gaps.",
        "Keep accelerator count, offered QPS, tokens/s, and latency compliance as separate fields."
      ],
      "rumor": false
    },
    {
      "id": "mlperf-v60-redhat-b200-qwen3vl-235b-server-2026",
      "entity": "MLCommons / Red Hat",
      "category": "serving_system",
      "topic": "MLPerf Inference v6.0 Qwen3-VL 235B-A22B Server on vLLM",
      "claim": "At the pinned MLCommons v6.0 commit, Red Hat submitted a valid Qwen3-VL-235B-A22B Server result on one Dell system with eight NVIDIA B200-SXM-180GB accelerators. The result completed 67.86 samples/s; unlike the LLM text-generation summaries, this result summary does not report completed tokens/s. The checked-in system description explicitly lists the accelerator, host count, vLLM/NVIDIA Dynamo/MLPerf Qwen3-VL harness framework string, and RHEL 10.1; no result-linked config file at the pinned commit establishes precision, TP/PP/EP, replicas, batch/concurrency ceilings, or sequence-length controls.",
      "evidence_class": "benchmark_measurement",
      "confidence": "high",
      "published_at": "2026-04-02",
      "event_at": "2026-02-12",
      "source_kind": "official_repository",
      "source_title": "MLPerf Inference v6.0 Red Hat B200 Qwen3-VL-235B-A22B Server result and system description, commit 4d3916ac",
      "source_url": "https://github.com/mlcommons/inference_results_v6.0/blob/4d3916ac9cf474b679cdfcf492d43a0559418ad1/closed/RedHat/results/8xb200_NV_vllm_qwen/qwen3-vl-235b-a22b/server/performance/run_1/mlperf_log_summary.txt",
      "quantified": {
        "model": "Qwen3-VL-235B-A22B",
        "submission_scenario": "Server",
        "loadgen_scenario": "Server",
        "result_status": "VALID",
        "completed_samples_per_second": 67.86,
        "scheduled_samples_per_second": 68.32,
        "result_test_datetime_utc": "2026-02-12T16:10:06Z",
        "latency_99th_percentile_ms": 11687.990824,
        "system_name": "Dell B200,8xB200-SXM-180GB,RHEL 10.1,vLLM CentML:mlperf-inf-mm-q3vl-v6.0",
        "host_count": 1,
        "accelerator_model": "NVIDIA B200-SXM-180GB",
        "accelerator_count": 8,
        "accelerator_memory": "180 GB each",
        "framework": "vLLM + NVIDIA-dynamo + MLPerf Inference Qwen3-VL harness, RHEL 10.1",
        "operating_system": "Red Hat Enterprise Linux 10.1 (Coughlan)"
      },
      "assumptions": [
        "Explicit result fields come from closed/RedHat/results/8xb200_NV_vllm_qwen/qwen3-vl-235b-a22b/server/performance/run_1/mlperf_log_summary.txt and mlperf_log_detail.txt at commit 4d3916ac.",
        "Explicit system fields come from closed/RedHat/systems/8xb200_NV_vllm_qwen.json at the same commit.",
        "No checked-in Red Hat measurement/config file tied to this result was found at the pinned commit, so precision, TP/PP/EP, replicas, batch/concurrency controls, and sequence-length controls are intentionally omitted.",
        "published_at is the MLPerf Inference v6.0 result-release date; the cited pinned repository commit itself is dated 2026-04-03."
      ],
      "contradictions_or_limits": [
        "This is one audited benchmark submission, not a production deployment, fleet-size census, or prevalence claim.",
        "Accelerator count is not host or fleet count. No topology or replica quantity is inferred from the eight-accelerator system name.",
        "Completed samples/s is the MLPerf result unit; the source does not report completed tokens/s, so no token-rate conversion is made.",
        "The 99th-percentile end-to-end latency is a reported observation, not a separately stated latency constraint in the summary.",
        "The pinned commit postdates the test run; published_at records the pinned commit date, while event_at and result_test_datetime_utc record the run."
      ],
      "fak_implications": [
        "Preserve workload result units exactly, particularly for multimodal serving.",
        "Unknown runtime topology and batching must remain explicit rather than normalized from aggregate hardware."
      ],
      "rumor": false
    },
    {
      "id": "nvidia-hugging-face-acquisition-report-2026",
      "category": "market_signal",
      "entity": "NVIDIA / Hugging Face",
      "topic": [
        "acquisition",
        "model_distribution_platform",
        "rumor"
      ],
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Nvidia Agrees to Buy Open Source AI Platform Hugging Face For $12.9 Billion",
      "source_url": "https://www.theinformation.com/articles/nvidia-agrees-buy-open-source-model-repository-hugging-face-12-9-billion",
      "source_kind": "credible_reporting",
      "evidence_class": "rumor",
      "confidence": "low",
      "claim": "Business Insider reported that NVIDIA and Hugging Face had discussed an acquisition valuing Hugging Face above $13B and had not reached a deal; The Information later reported that NVIDIA agreed to buy Hugging Face for $12.9B. As of August 27, 2026, neither party had publicly announced or confirmed a signed agreement, and no completed acquisition was established.",
      "quantified": {
        "reported_talks_valuation_usd_gt": 13000000000,
        "reported_agreement_value_usd": 12900000000
      },
      "assumptions": [
        "Control of a major model and dataset distribution platform could affect ecosystem neutrality and infrastructure routing, but no ownership or operational change is assumed before a completed transaction is evidenced."
      ],
      "contradictions_or_limits": [
        "Business Insider's 2026-08-27T00:34:46Z report said the companies had not reached a deal and talks could fail; The Information's later August 26 PDT report said an agreement existed. The reports describe different states and neither is a party announcement.",
        "No NVIDIA or Hugging Face announcement confirming signing was found by the 2026-08-27 cutoff; a reported agreement is not evidence of execution, public announcement, or current ownership.",
        "Transaction structure and terms, including consideration form, adjustments, governance, approvals, conditions, termination rights, and operating commitments, remain undisclosed and are not inferred.",
        "Regulatory conditions, filings, review jurisdictions, remedies, closing timetable, and close remain unresolved."
      ],
      "fak_implications": [
        "Track model-distribution ownership and platform-neutrality risk as a watch item, but do not change provider, hardware, or continuity assumptions from this report alone.",
        "Preserve talks, reported agreement, party announcement or signing, regulatory clearance, and close as separate lifecycle states."
      ],
      "rumor": {
        "is_rumor": true,
        "origin": "Independent Business Insider and The Information reporting on successive but conflicting transaction states",
        "corroboration": "Business Insider reported detailed talks above $13B with no deal; The Information later reported a $12.9B agreement. TechCrunch documented both accounts and said neither company had responded. This corroborates acquisition activity as a report lifecycle, not a signed or completed transaction.",
        "status": "reported_agreement_open",
        "last_checked_at": "2026-08-27",
        "expires_at": "2026-11-30",
        "resolution": "Open. Exact unresolved fields are signing or party announcement, transaction structure and terms, regulatory conditions, and close. Resolve only with primary party or regulatory evidence, or mark expired_unresolved after the review date."
      }
    },
    {
      "id": "google-cluster-director-network-topology-2026",
      "category": "ai_cloud",
      "entity": "Google Cloud",
      "topic": "Cluster Director managed GPU cluster network hierarchy and topology envelope",
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Supported networking services in Cluster Director",
      "source_url": "https://docs.cloud.google.com/cluster-director/docs/networking",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Google Cloud's current Cluster Director networking documentation defines a managed, rail-aligned hierarchy from physical host to single-rack sub-block to non-blocking block to multi-block cluster. It specifies one-hop GPU reachability within a sub-block, at most two network hops within a block, and support for clusters that scale to thousands of GPUs. Dedicated NVIDIA NICs carry GPU-to-GPU RoCE traffic, while Titanium NICs carry host, storage, and Google Cloud service traffic on a separate path.",
      "quantified": {
        "document_last_updated": "2026-08-26",
        "hierarchy": [
          "physical host",
          "single-rack sub-block",
          "block",
          "cluster"
        ],
        "subblock_rack_count": 1,
        "intra_subblock_max_gpu_network_hops": 1,
        "intra_block_max_gpu_network_hops": 2,
        "documented_cluster_scale": "thousands of GPUs",
        "block_fabric": "non-blocking",
        "gpu_data_plane": "dedicated NVIDIA NICs with technologies such as RDMA over Converged Ethernet (RoCE)",
        "host_storage_data_plane": "Titanium NICs",
        "multi_vpc_examples": [
          "A4",
          "A3 Ultra"
        ],
        "lifecycle_type": "supported architecture envelope",
        "verification_date": "2026-08-27"
      },
      "assumptions": [
        "published_at and event_at use the page's exposed Last updated date; this is living product documentation rather than an original launch publication.",
        "The phrase thousands of GPUs is a supported cluster-scale envelope, not a disclosed configured maximum or a measurement of a customer deployment.",
        "Node or host is a physical server; a Compute Engine instance is provisioned on top of a host and must not be counted as a separate physical hierarchy level without machine-family evidence."
      ],
      "contradictions_or_limits": [
        "The page does not state a numeric maximum number of hosts, sub-blocks, blocks, clusters, or GPUs, and it does not say that thousands of GPUs is the prevalent topology.",
        "The page does not disclose Cluster Director region coverage or label this specific networking envelope as preview or GA; product-stage and location claims require separate sources.",
        "The page reports supported topology and traffic separation, not observed active workloads, schedulable fleet size, utilization, queue wait, achieved goodput, failure or retry distributions, power, or cost.",
        "One-hop and two-hop values are reachability bounds inside the documented hierarchy, not end-to-end latency measurements."
      ],
      "fak_implications": [
        "Represent GPU placement as host, rack/sub-block, block, and cluster rather than a flat accelerator pool, and retain separate GPU and host/storage network paths.",
        "Label thousands of GPUs as a supported scale class; never instantiate that count as available or active capacity without a deployment receipt."
      ],
      "rumor": false
    },
    {
      "id": "google-gke-tpu-multislice-envelope-2026",
      "category": "ai_cloud",
      "entity": "Google Cloud",
      "topic": "current GKE TPU Multislice support, topology, and orchestration envelope",
      "published_at": "2026-08-20",
      "event_at": "2026-08-20",
      "source_title": "Deploy TPU Multislices in GKE",
      "source_url": "https://docs.cloud.google.com/kubernetes-engine/docs/how-to/tpu-multislice",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Google Cloud's current GKE documentation defines a Multislice workload as two or more multi-host TPU slices: ICI connects chips within a slice and DCN connects slices. All slices in one workload must share TPU type, size, and topology; only synchronous multicontroller training is supported. GKE Standard supports Multislice from 1.27.4-gke.900, Autopilot from 1.29.2-gke.1521000, and the page documents atomic multi-host node-pool scaling from zero to the topology-derived maximum node count.",
      "quantified": {
        "document_last_updated": "2026-08-20",
        "minimum_slices_per_multislice_workload": 2,
        "maximum_slices_per_multislice_workload": null,
        "gke_standard_minimum_version": "1.27.4-gke.900",
        "gke_autopilot_minimum_version": "1.29.2-gke.1521000",
        "minimum_supported_jax_version": "2.1",
        "supported_frameworks": [
          "JAX",
          "PyTorch"
        ],
        "supported_training_mode": "synchronous multicontroller",
        "supported_slice_host_scope": "multi-host only",
        "same_type_size_topology_required": true,
        "tpu_v3_supported": false,
        "chips_per_tpu_slice_vm_documented_values": [
          1,
          4,
          8
        ],
        "intra_slice_network": "inter-chip interconnect (ICI)",
        "inter_slice_network": "data center network (DCN)",
        "multi_host_node_pool_scaling": "atomic from zero to the topology-derived maximum size",
        "example_single_host_shapes_excluded_from_multislice": [
          "TPU v4 2x2x1",
          "TPU v5e 2x2"
        ],
        "documented_scale_claim": "near-linear scaling up to tens of thousands of TPU chips",
        "recommended_workload_api": "JobSet",
        "lifecycle_type": "current supported GKE configuration envelope",
        "verification_date": "2026-08-27"
      },
      "assumptions": [
        "published_at and event_at use the page's exposed Last updated date; the page is living documentation.",
        "The topology product determines chips per slice, and chips per VM depend on TPU machine type; no universal hosts-per-slice value is inferred.",
        "The page's tens-of-thousands statement is a vendor scalability claim, not a numeric supported maximum or evidence of an achieved run."
      ],
      "contradictions_or_limits": [
        "The current GKE page does not disclose a maximum number of slices per JobSet or a maximum number of TPU chips per Multislice workload.",
        "Minimum GKE versions establish software support, not regional TPU capacity, quota, or immediate obtainability; the page does not provide one exhaustive regional availability list or an explicit preview/GA label for every listed combination.",
        "Atomic node-pool scaling is a configured lifecycle rule, not proof that the full topology can be provisioned at request time.",
        "The page does not disclose queue wait, utilization, achieved active scale, failure or retry distributions, power, cost, or production prevalence."
      ],
      "fak_implications": [
        "Preserve job to slices to VMs/hosts to chips as separate hierarchy levels; use ICI within a slice and DCN between slices in topology and communication-cost models.",
        "Admission must treat a multi-host slice as atomic and reject heterogeneous type, size, or topology inside one Multislice workload."
      ],
      "rumor": false
    },
    {
      "id": "google-cloud-tpu-api-multislice-queued-resource-envelope-2026",
      "category": "accelerator_platform",
      "entity": "Google Cloud",
      "topic": "legacy Cloud TPU API Multislice queued-resource ceiling and failure semantics",
      "published_at": "2026-08-11",
      "event_at": "2026-08-11",
      "source_title": "Cloud TPU Multislice Overview",
      "source_url": "https://docs.cloud.google.com/tpu/docs/multislice-introduction",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "The legacy Cloud TPU API Multislice documentation supports v4 and later, defines a Multislice node as one TPU slice, requires homogeneous slice shape, and caps a queued-resource request at 256 slices. It defines gang scheduling as provisioning all slices together or none, uses ICI inside a slice and host-mediated DCN between slices, and states that TPU v4 Multislice can run on more than 4,096 chips. On disruption, the impacted slice is replaced and all slices are reset; if replacement capacity is unavailable, training stops.",
      "quantified": {
        "document_last_updated": "2026-08-11",
        "api_status": "no longer under active development; bug fixes and security updates only",
        "recommended_successor": "Google Kubernetes Engine for Multislice",
        "minimum_supported_tpu_generation": "TPU v4",
        "minimum_slices_per_multislice": 2,
        "maximum_slices_per_queued_resource": 256,
        "multislice_node_definition": "one TPU slice",
        "heterogeneous_slice_shapes_supported": false,
        "v4_single_run_chip_scale": ">4096 TPU v4 chips",
        "v4_chips_per_host": 4,
        "v4_host_max_network_bandwidth_gbps": 50,
        "gang_scheduling": "all slices provisioned together or none",
        "intra_slice_network": "inter-chip interconnect (ICI)",
        "inter_slice_network": "host-mediated data center network (DCN)",
        "capacity_modes": [
          "reservation",
          "Spot",
          "on-demand queued"
        ],
        "on_demand_availability_guaranteed": false,
        "lifecycle_type": "supported configured maximum on a maintenance-only API",
        "verification_date": "2026-08-27"
      },
      "assumptions": [
        "published_at and event_at use the page's exposed Last updated date; the page is living documentation for the maintenance-only Cloud TPU API.",
        "The 256-slice value is a request ceiling for this API, not a current GKE maximum, a prevalent topology, or an observed allocation.",
        "The more-than-4,096-chip statement is a capability statement for TPU v4 Multislice, not a disclosed pod size or achieved workload count."
      ],
      "contradictions_or_limits": [
        "This page applies to an API that Google says is no longer under active development, so its 256-slice ceiling must not be silently transferred to current GKE or future TPU control planes.",
        "The page does not enumerate a universal supported-shape table; it only requires all slices in one request to use the same accelerator type/shape.",
        "Reservation, Spot, and on-demand are admission/capacity modes; none states observed queue wait, approval probability, utilization, or achieved active scale.",
        "The page does not disclose pod count, host count for a maximum request, production prevalence, failure rate, retry count, power, or cost."
      ],
      "fak_implications": [
        "Keep the 256-slice field versioned to the legacy Cloud TPU queued-resource API and require a control-plane identifier on every topology ceiling.",
        "Model all-or-none admission and whole-environment restart separately from ordinary per-host replacement."
      ],
      "rumor": false
    },
    {
      "id": "google-dws-provisioning-envelope-2026",
      "category": "ai_cloud",
      "entity": "Google Cloud",
      "topic": "Dynamic Workload Scheduler Flex-start and reservation-bound admission envelope",
      "published_at": "2026-08-26",
      "event_at": "2026-08-26",
      "source_title": "Compute Engine instances provisioning models",
      "source_url": "https://docs.cloud.google.com/compute/docs/instances/provisioning-models",
      "source_kind": "official_documentation",
      "evidence_class": "vendor_specification",
      "confidence": "high",
      "claim": "Google Cloud's current Compute Engine provisioning-model documentation separates Dynamic Workload Scheduler-backed Flex-start from reservation-bound capacity. Flex-start is best-effort: standalone requests may wait up to two hours, MIG requests can persist until resources are available or canceled, resources run for 10 minutes to seven days, and dense placement is best-effort. Calendar-mode future reservations support workloads up to 90 days; once Google Cloud approves a reservation request, capacity assurance is described as very high and the customer has exclusive access during the reservation period.",
      "quantified": {
        "document_last_updated": "2026-08-26",
        "flex_start_standalone_max_wait_hours": 2,
        "flex_start_mig_wait_end_condition": "resources become available or requester cancels",
        "flex_start_minimum_run_minutes": 10,
        "flex_start_maximum_run_days": 7,
        "flex_start_capacity_assurance": "best-effort",
        "flex_start_placement": "dense on a best-effort basis",
        "flex_start_creation_modes": [
          "standalone instance",
          "MIG individual creation as capacity becomes available",
          "MIG all-at-once resize request"
        ],
        "flex_start_supported_accelerator_families": [
          "A4",
          "A3",
          "A2",
          "G4",
          "G2",
          "N1 with attached GPUs",
          "TPU7x (allowlist restricted)",
          "TPU v6e",
          "TPU v5p",
          "H4D"
        ],
        "calendar_mode_workload_maximum_days": 90,
        "calendar_mode_capacity_assurance_after_approval": "very high",
        "calendar_mode_access": "exclusive for the reservation period",
        "calendar_mode_tpu_families": [
          "TPU7x (allowlist restricted)",
          "TPU v6e",
          "TPU v5p"
        ],
        "calendar_mode_gpu_families": [
          "A4",
          "A3 Ultra",
          "A3 Mega 8-GPU",
          "A3 High 8-GPU",
          "A3 Edge",
          "H4D"
        ],
        "lifecycle_type": "documented admission and provisioning semantics",
        "verification_date": "2026-08-27",
        "calendar_mode_gpu_tpu_reservation_request_quota_required": false
      },
      "assumptions": [
        "published_at and event_at use the page's exposed Last updated date; this is living documentation.",
        "The two-hour value is a configurable standalone Flex-start waiting ceiling, not an observed queue-wait distribution or service-level guarantee.",
        "Very high assurance applies after approval to reserved capacity and is not equivalent to guaranteed job success or guaranteed immediate VM creation if other prerequisites are missing."
      ],
      "contradictions_or_limits": [
        "Flex-start best-effort admission, reservation approval, delivered capacity, created VMs, and a running job are separate lifecycle states.",
        "Dense best-effort placement is not a topology guarantee; a machine-family list is not a region-by-region availability receipt.",
        "The page does not label every current mode or machine combination as preview or GA and does not provide one exhaustive regional availability list.",
        "No actual queue-wait distribution, admission probability, utilization, achieved active accelerator count, failure or retry distribution, power, or total workload cost is disclosed."
      ],
      "fak_implications": [
        "Benchmark DWS with separate best-effort Flex-start and approved-reservation states, and record requested, pending, admitted, provisioned, running, and expired capacity independently.",
        "Treat wait ceilings and run-duration ceilings as configured controls, not empirical workload distributions."
      ],
      "rumor": false
    },
    {
      "id": "google-xpk-gke-tpu-v5e-50944-job-2023",
      "category": "ai_cloud",
      "entity": "Google Cloud",
      "topic": "XPK and GKE orchestration of an achieved 50,944-chip TPU v5e training job",
      "published_at": "2023-11-08",
      "event_at": "2023-11",
      "source_title": "Google Cloud demonstrates the world's largest distributed training job for large language models across 50000+ TPU v5e chips",
      "source_url": "https://cloud.google.com/blog/products/compute/the-worlds-largest-distributed-llm-training-job-on-tpu-v5e",
      "source_kind": "official_engineering_release",
      "evidence_class": "official_statement",
      "confidence": "high",
      "claim": "Google Cloud reports an achieved November 2023 Multislice training job on 50,944 TPU v5e chips spanning 199 pods. Each v5e pod contained 256 chips connected by ICI; pods communicated over Jupiter DCN. GKE managed TPU capacity; XPK decoupled capacity provisioning from job execution through separate APIs, created and resized clusters, submitted jobs to GKE Kueue as JobSets, managed those JobSets, and exposed cluster state. Google announced Cloud TPU Multislice Training as generally available in the same release.",
      "quantified": {
        "publication_date": "2023-11-08",
        "event_month": "2023-11",
        "accelerator_generation": "Cloud TPU v5e",
        "achieved_active_training_chips": 50944,
        "observed_pod_count": 199,
        "chips_per_pod": 256,
        "reported_total_peak_exaflops_16_bit": 10,
        "reported_total_peak_exaops_8_bit": 20,
        "intra_pod_network": "inter-chip interconnect (ICI)",
        "inter_pod_network": "Jupiter data center network (DCN)",
        "model_parameter_counts_billions": [
          16,
          32,
          64,
          128
        ],
        "inter_pod_parallelism": "data parallelism",
        "intra_pod_parallelism": "FSDP for 16B, 32B, and 64B; FSDP plus tensor parallelism for 128B",
        "xpk_orchestration_actions": [
          "create clusters",
          "resize clusters",
          "submit Kueue JobSets",
          "manage JobSets",
          "expose cluster state"
        ],
        "multislice_training_release_status": "general availability announced",
        "lifecycle_type": "achieved active training workload disclosed by the provider",
        "verification_date": "2026-08-27",
        "xpk_separates_capacity_and_job_apis": true
      },
      "assumptions": [
        "The source exposes an exact publication date but only an event month; event_at therefore preserves month precision rather than inventing a run day.",
        "The 50,944-chip quantity is treated as an achieved active training workload because the source says Google used Multislice Training to run it; it is not treated as announced fleet capacity or a schedulable customer maximum.",
        "The source reports 199 pods but does not explicitly equate pod count to TPU slice count, so no slice count is inferred."
      ],
      "contradictions_or_limits": [
        "This is a Google-authored disclosure, not an independent audit, a production-fleet census, or evidence that 50,944 chips were continuously useful throughout the run.",
        "The reported 10 exa-FLOPs and 20 exa-OPs are total peak capability, not achieved goodput or model FLOPs utilization.",
        "The page does not disclose host count, exact slice count, cluster queue wait, utilization, failure/retry frequency, checkpoint loss, power, energy, or total cost.",
        "One achieved large job does not establish a generally schedulable 50,944-chip customer fleet, a supported maximum, or a prevalent topology."
      ],
      "fak_implications": [
        "Keep achieved active workload evidence separate from supported maxima and from announced capacity; record pod and chip counts without manufacturing slice or host counts.",
        "Represent XPK/GKE/Kueue/JobSet as distinct control-plane layers whose orchestration overhead and admission state must be included in end-to-end receipts."
      ],
      "rumor": false
    }
  ]
}
