{
 "meta": {
  "title": "Where IOTA's numbers travelled",
  "updated": "2026-09-28",
  "edited": "2026-09-29",
  "subject": "IOTA · Bittensor subnet 9 · Macrocosmos",
  "corrections": "connect@mikyo.one",
  "corpus": "research/iota/ in the mikyo.one brain (not public); the Orion Register at /iota-runs/ and the field map at /iota-scape/ are public"
 },
 "lanes": [
  "Macrocosmos",
  "Carried by",
  "Read back"
 ],
 "labels": {
  "verified": "checked against the source on the date given",
  "stated": "the page says it; not independently checked",
  "unknown": "looked for, not found or not reachable"
 },
 "claims": [
  {
   "id": "speed",
   "short": "About 65% of datacenter training speed",
   "claim": "Orion-100B trained at about 65% of the speed of an equivalent datacenter setup.",
   "note": "Peak sustained 82%, in the same post.",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "Steffen Cruz",
     "kind": "origin",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"Orion-100B achieved roughly 65% of the training speed of equivalent datacenter setups.\" Also \"~65% (peak sustained 82%) as fast as using co-located devices.\"",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "@MacrocosmosAI launch thread",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2061493162582118695",
     "q": "\"upward of 65% of data-center training efficiency on hardware costing a fraction of the price\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-03",
     "who": "@WSquires, reply",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/WSquires/status/2062256898368516410",
     "q": "\"70% of centralised MFU is frankly wild. We achieved 30% on our third run.\"",
     "label": "stated",
     "from": "sch",
     "edge": "reply"
    },
    {
     "lane": 0,
     "date": "2026-06-17",
     "who": "The Case for Liquid Training",
     "sub": "Substack",
     "kind": "dot",
     "url": "https://macrocosmosai.substack.com/p/the-case-for-liquid-training",
     "q": "\"100B-parameter models trained within 65% of centralised paradigms, at 30% of the cost\"",
     "label": "verified",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-24",
     "who": "@IOTA_SN9, restated",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/IOTA_SN9/status/2103175140699934764",
     "q": "\"roughly 65% of the speed the same job reaches on co-located high-bandwidth hardware\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "tao.media",
     "sub": "cites the post",
     "kind": "dot",
     "url": "https://www.tao.media/macrocosmos-unveils-orion-100b-a-100b-parameter-distributed-ai-training-run/",
     "q": "\"65% of the training speed of comparable datacenter deployments.\" Cites the Substack post.",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "SimplyTao",
     "sub": "no source",
     "kind": "dot",
     "url": "https://simplytao.ai/blog/orion-100b-macrocosmos",
     "q": "\"reached roughly 65% of the training speed of an equivalent datacenter setup.\" No source given.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "CoinMarketCap",
     "sub": "paraphrase",
     "kind": "para",
     "url": "https://coinmarketcap.com/top-stories/6a1e4486e1de341d4442a696/",
     "q": "Paraphrase: \"a decentralized network can approach data-center-like efficiency.\" Cites two X posts.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "Proof of Talk recap",
     "sub": "page gone",
     "kind": "gone",
     "url": "",
     "q": "\"up to 65% of datacenter training efficiency.\" SimplyTao's summit recap; the page is now 404, quoted from the sweep row.",
     "label": "unknown",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-05",
     "who": "Manifold Labs",
     "sub": "paraphrase",
     "kind": "para",
     "url": "https://manifoldlabs.substack.com/p/capital-finds-the-machine",
     "q": "Paraphrase: \"the first credible demonstration that frontier-scale decentralized pretraining can compete with datacenters.\" No source given.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-21",
     "who": "The Tech Pulse",
     "sub": "video",
     "kind": "dot",
     "url": "https://www.youtube.com/watch?v=6RHDAaGiMlY",
     "q": "Title carries the figure: \"100B across five datacenters at 65% speed.\" Japanese-language explainer.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-07-02",
     "who": "@_Qubic_",
     "sub": "a rival network",
     "kind": "dot",
     "url": "https://x.com/_Qubic_/status/2072698452174082341",
     "q": "\"It reached roughly 65% of comparable datacenter training speed.\" \"So read Orion-100B as outside confirmation.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 2,
     "date": "2026-06-03",
     "who": "@GSchvey, six-part thread",
     "sub": "X",
     "kind": "read",
     "url": "https://x.com/GSchvey/status/2062233327742894539",
     "q": "\"30% MFU per chip. In a datacenter, well-optimized training typically reaches 40-60%.\" \"16 hand-selected A100 nodes in a controlled run.\"",
     "label": "stated",
     "id": "sch",
     "from": "o"
    }
   ],
   "figure": "65%",
   "unit": "of datacenter training speed"
  },
  {
   "id": "cost",
   "short": "$1.25 a GPU-hour, $20 a replica-hour, 2.5x or 3x cheaper",
   "claim": "An Orion replica of 16 non-colocated A100s at $1.25 an hour costs about $20 an hour, against about $50 for a high-spec node.",
   "note": "Three phrasings, all Macrocosmos's: 2.5x lower entry cost per replica (the post, and the 25 September thread), 3x on the stage (28 September), and 30% of the cost (17 June).",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "Steffen Cruz",
     "kind": "origin",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"An Orion replica (16 non-colocated A100s at $1.25 per hour) can be provisioned for as low as $20 per hour\", against a \"Covenant-class peer (8xB200) ... approximately $50 per hour\".",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-25",
     "who": "@MacrocosmosAI",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2103537520558661976",
     "q": "\"on single A100s at around $1.25 an hour each\"; \"a full replica at roughly $20 an hour, and the entry cost about 2.5x below approaches that need high-spec nodes throughout.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-17",
     "who": "The Case for Liquid Training",
     "sub": "Substack",
     "kind": "dot",
     "url": "https://macrocosmosai.substack.com/p/the-case-for-liquid-training",
     "q": "\"at 30% of the cost by using heterogeneous, non-frontier, available hardware\"",
     "label": "verified",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "SimplyTao",
     "sub": "no source",
     "kind": "dot",
     "url": "https://simplytao.ai/blog/orion-100b-macrocosmos",
     "q": "48 A100 GPUs, \"2.5× cheaper\". No source given.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-05",
     "who": "Manifold Labs",
     "sub": "no source",
     "kind": "dot",
     "url": "https://manifoldlabs.substack.com/p/capital-finds-the-machine",
     "q": "\"roughly $20 per hour per training group against ~$50 for previous approaches\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-08",
     "who": "Riflo animations",
     "sub": "video",
     "kind": "dot",
     "url": "https://www.youtube.com/watch?v=JBLjfF-fo6o",
     "q": "Title: \"The AI Model That Was Trained for $1.25 Per Hour\".",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-26",
     "who": "The TAO Daily",
     "sub": "attributes to Macrocosmos",
     "kind": "dot",
     "url": "https://taodaily.io/iotas-orion-100b-still-proves-frontier-training-doesnt-need-one-giant-data-centre/",
     "q": "\"16 non-colocated A100s came in at around $20 per hour\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "@taodaily_io, from the stage",
     "sub": "reports Cruz",
     "kind": "dot",
     "url": "https://x.com/taodaily_io/status/2104601940294340972",
     "q": "Cruz on stage: \"We were able to train it 3× cheaper than if we had simply reserved a node in a data center.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "@tylerdurdeth, attendee",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/tylerdurdeth/status/2104614585030308119",
     "q": "\"train a model with up to 10T parameters (i.e. frontier) at up to 3x cheaper cost\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "tao.media",
     "sub": "cites an attendee",
     "kind": "dot",
     "url": "https://www.tao.media/macrocosmos-announces-iota-sdk-and-liquid-compute-at-exploit-summit/",
     "q": "\"Train models with up to 10 trillion parameters at up to three times lower cost than centralized labs.\" Cites an attendee's post.",
     "label": "stated",
     "from": "o"
    }
   ],
   "figure": "$20",
   "unit": "a replica-hour, 16 A100s at $1.25"
  },
  {
   "id": "run",
   "short": "100B across 16 stages x 3 replicas on 48 A100s",
   "claim": "Orion-100B ran as 16 pipeline stages times 3 replicas, 48 A100-80GB GPUs across five US datacenters, about 1.1B tokens in about two days.",
   "note": "",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "Steffen Cruz",
     "kind": "origin",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"16-stage pipeline-parallel run with 3 replicas, equalling a total of 48 devices\", each \"a single A100 80GB GPU\", \"distributed across 5 datacenters within the United States\"; \"approximately 1.1 billion tokens ... over the period of approximately 2 days\".",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "@MacrocosmosAI launch thread",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2061493162582118695",
     "q": "\"a 100 billion parameter model across 16 pipeline-parallel stages and 3 replicas\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-07-27",
     "who": "@MacrocosmosAI, Orion-16B launch",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2081791889674772618",
     "q": "\"Orion-100B, trained across 48 GPUs, was the largest decentralised training run ever conducted by model size.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-24",
     "who": "@IOTA_SN9, restated",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/IOTA_SN9/status/2103175140699934764",
     "q": "\"We trained a 100B parameter model across five data centres, on single A100s talking to each other over ordinary internet links.\" \"It went 1.1B tokens over two days and stopped there on cost grounds.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-25",
     "who": "@MacrocosmosAI",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2103537520558661976",
     "q": "\"It ran as a viability proof rather than to a finished model, and it held.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "tao.media",
     "sub": "cites the post",
     "kind": "dot",
     "url": "https://www.tao.media/macrocosmos-unveils-orion-100b-a-100b-parameter-distributed-ai-training-run/",
     "q": "\"100 billion parameter model across 16 pipeline-parallel stages and three replicas using geographically distributed Nvidia A100 GPUs.\" Cites the Substack post.",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "SimplyTao",
     "sub": "no source",
     "kind": "dot",
     "url": "https://simplytao.ai/blog/orion-100b-macrocosmos",
     "q": "\"48 A100 GPUs\"; \"1.1 billion tokens of the fineweb-edu-score-2 dataset\"; \"Average system throughput sat around 9000 tokens per second\".",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-05",
     "who": "Manifold Labs",
     "sub": "paraphrase",
     "kind": "para",
     "url": "https://manifoldlabs.substack.com/p/capital-finds-the-machine",
     "q": "\"The proof-of-concept stopped after two days due to budget constraints.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-21",
     "who": "The Tech Pulse",
     "sub": "video",
     "kind": "dot",
     "url": "https://www.youtube.com/watch?v=6RHDAaGiMlY",
     "q": "Title: \"100B across five datacenters\".",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-08-03",
     "who": "cryptonews.net",
     "sub": "paraphrase",
     "kind": "para",
     "url": "https://cryptonews.net/news/altcoins/33236569/",
     "q": "Contrasts Orion-16B with \"Orion-100B's fixed hardware\". No source given.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-26",
     "who": "The TAO Daily",
     "sub": "cites the X post",
     "kind": "dot",
     "url": "https://taodaily.io/iotas-orion-100b-still-proves-frontier-training-doesnt-need-one-giant-data-centre/",
     "q": "\"We trained a 100B parameter model across five data centres, on single A100s talking to each other over ordinary internet links.\" Cites the 24 September post.",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 2,
     "date": "2026-06-03",
     "who": "@GSchvey, six-part thread",
     "sub": "X",
     "kind": "read",
     "url": "https://x.com/GSchvey/status/2062233327742894539",
     "q": "\"These were A100 professional GPUs, not consumer cards that IOTA is best known for using.\" \"These were 16 hand-selected A100 nodes in a controlled run.\"",
     "label": "stated",
     "from": "o"
    }
   ],
   "figure": "48",
   "unit": "A100s, 16 stages x 3 replicas, five datacenters"
  },
  {
   "id": "mfu",
   "short": "30.8% average MFU, 38% peak",
   "claim": "Orion-100B held 30.8% average model FLOP utilisation on A100-80GB chips, with a 38% peak sustained over six hours.",
   "note": "",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "Steffen Cruz",
     "kind": "origin",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"average training MFU of 30.8% and a peak sustained training MFU of 38% over a 6 hour period\"",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "@MacrocosmosAI launch thread",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2061493162582118695",
     "q": "\"achieving upwards of 30% Model FLOP Utilisation (MFU) on Nvidia A100-80GB chips\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-03",
     "who": "@WSquires, reply",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/WSquires/status/2062256898368516410",
     "q": "\"We achieved 30% on our third run.\" \"Colossus 2 on JAX was running at c. 20% before Elon's team rewrote the NVidia logic in C.\"",
     "label": "stated",
     "from": "sch",
     "edge": "reply"
    },
    {
     "lane": 0,
     "date": "2026-07-27",
     "who": "@MacrocosmosAI, Orion-16B launch",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2081791889674772618",
     "q": "\"supporting sustained MFU values over 30%\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-24",
     "who": "@IOTA_SN9, restated",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/IOTA_SN9/status/2103175140699934764",
     "q": "\"It held 30.8% average MFU, with a sustained 38% peak\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-26",
     "who": "The TAO Daily",
     "sub": "cites the X post",
     "kind": "dot",
     "url": "https://taodaily.io/iotas-orion-100b-still-proves-frontier-training-doesnt-need-one-giant-data-centre/",
     "q": "\"30.8% average MFU, with a sustained 38% peak\". Cites the 24 September post.",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 2,
     "date": "2026-06-03",
     "who": "@GSchvey, six-part thread",
     "sub": "X",
     "kind": "read",
     "url": "https://x.com/GSchvey/status/2062233327742894539",
     "q": "\"They achieved 30% MFU per chip. In a datacenter, well-optimized training typically reaches 40-60%.\"",
     "label": "stated",
     "id": "sch",
     "from": "o"
    }
   ],
   "figure": "30.8%",
   "unit": "average MFU, 38% peak"
  },
  {
   "id": "compression",
   "short": "128x activation compression",
   "claim": "IOTA's bottleneck blocks compress activations up to 128x with no significant loss in convergence, shown on a 1.5B model over up to 400M tokens.",
   "note": "",
   "nodes": [
    {
     "lane": 0,
     "date": "2025-07-16",
     "who": "IOTA technical primer",
     "sub": "arXiv 2507.17766",
     "kind": "origin",
     "url": "https://arxiv.org/abs/2507.17766",
     "q": "\"128x symmetrical compression rate for both activations and their gradients, with no significant loss in convergence when training a modified 1.5B parameter three-bottleneck Llama3 model on up to 400 million tokens\"",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-04-13",
     "who": "ResBM paper",
     "sub": "arXiv 2604.11947",
     "kind": "dot",
     "url": "https://arxiv.org/abs/2604.11947",
     "q": "\"ResBMs achieve state-of-the-art 128x activation compression without significant loss in convergence rates.\" The paper does not name IOTA.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-05",
     "who": "Alan Aboudib interview",
     "sub": "Substack",
     "kind": "dot",
     "url": "https://macrocosmosai.substack.com/p/the-middle-point-between-researcher-924",
     "q": "\"these tensors are simply too slow to transmit over standard internet speeds\"; \"We had to redesign the LLM architecture from the ground up in a decentralised-native approach.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2025-06-30",
     "who": "Ventura Labs, Ep 50",
     "sub": "video description",
     "kind": "dot",
     "url": "https://www.youtube.com/watch?v=zjRAyYRpImA",
     "q": "Description: activation compression 128x; \"world-first, fully decentralized pipeline-parallel training architecture\".",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-02-11",
     "who": "SimplyTao guide",
     "sub": "no attribution",
     "kind": "dot",
     "url": "https://simplytao.ai/blog/your-simple-guide-to-iota-sn9",
     "q": "\"This block achieved up to 128x activation compression on a 1.5B parameter model.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-02-15",
     "who": "Grokipedia",
     "sub": "cites the primer",
     "kind": "dot",
     "url": "https://grokipedia.com/page/IOTA_Bittensor_subnet",
     "q": "\"up to 128× reduction in communication bandwidth while preserving model convergence on 1.5 billion parameter models\". Page undated; fact-checked about February 2026.",
     "label": "stated",
     "from": "o",
     "cites": true,
     "approx": true
    },
    {
     "lane": 1,
     "date": "2026-05-22",
     "who": "Let's Data Science",
     "sub": "cites arXiv",
     "kind": "dot",
     "url": "https://letsdatascience.com/news/sn9-deploys-iota-for-distributed-large-scale-model-training-a8604137",
     "q": "\"activation compression, reported in the paper to reduce communication bandwidths by up to 128x via model-bottleneck techniques\"",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 2,
     "date": "2026-08-06",
     "who": "Pith review",
     "sub": "machine referee",
     "kind": "read",
     "url": "https://pith.science/paper/2507.17766",
     "q": "\"A promising but under-evidenced synthesis\"; the 128x claim \"is supported by a single 1.5B-parameter run on 400M tokens, with no error bars, no seeds, no code\".",
     "label": "verified",
     "from": "o"
    }
   ],
   "figure": "128x",
   "unit": "activation compression"
  },
  {
   "id": "fourteen",
   "short": "700M to 14B models by August 2024",
   "claim": "Before IOTA, subnet 9's competition had miners pretraining models from 700 million to 14 billion parameters by August 2024.",
   "note": "The one claim with independent evidence older than its origin.",
   "nodes": [
    {
     "lane": 0,
     "date": "2025-07-16",
     "who": "IOTA technical primer",
     "sub": "arXiv 2507.17766",
     "kind": "origin",
     "url": "https://arxiv.org/abs/2507.17766",
     "q": "\"In August 2024, Bittensor's Subnet 9 (SN9) demonstrated that a distributed network of incentivized, permissionless actors could each pretrain large language models (LLMs) ranging from 700 million to 14 billion parameters\"",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-01-01",
     "who": "docs: Subnet 9 IOTA",
     "sub": "docs.macrocosmos.ai, undated",
     "kind": "dot",
     "url": "https://docs.macrocosmos.ai/subnets/subnet-9-iota",
     "q": "\"700 million to 14 billion parameters\" (August 2024). Page undated.",
     "label": "stated",
     "from": "o",
     "undated": true
    },
    {
     "lane": 1,
     "date": "2026-02-15",
     "who": "Grokipedia",
     "sub": "cites the primer",
     "kind": "dot",
     "url": "https://grokipedia.com/page/IOTA_Bittensor_subnet",
     "q": "\"cooperative pretraining of large language models ranging from 700 million to 14 billion parameters across heterogeneous, unreliable devices\"",
     "label": "stated",
     "from": "o",
     "cites": true,
     "approx": true
    },
    {
     "lane": 1,
     "date": "2026-05-21",
     "who": "Crypto Briefing",
     "sub": "no source",
     "kind": "dot",
     "url": "https://cryptobriefing.com/sn9-iota-ai-model-training/",
     "q": "\"By August 2024, that setup had successfully pretrained large language models with up to 14 billion parameters\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-05-22",
     "who": "Let's Data Science",
     "sub": "cites arXiv",
     "kind": "dot",
     "url": "https://letsdatascience.com/news/sn9-deploys-iota-for-distributed-large-scale-model-training-a8604137",
     "q": "\"SN9 pretrained models up to 14 billion parameters in August 2024\"",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 1,
     "date": "2026-05-25",
     "who": "CommsTrader",
     "sub": "no source",
     "kind": "dot",
     "url": "https://commstrader.com/business/crypto/sn9s-iota-architecture-powers-large-scale-ai-model-training/",
     "q": "\"pretrain large language models with up to 14 billion parameters by August 2024\". Page dated \"4 months ago\" when read.",
     "label": "stated",
     "from": "o",
     "approx": true
    },
    {
     "lane": 2,
     "date": "2024-12-09",
     "who": "Crucible Labs",
     "sub": "before the origin",
     "kind": "read",
     "url": "https://cruciblelabs.com/articles/diving-into-decentralized-training-on-bittensor",
     "q": "\"SN 9, 29, and others are leading competitions with up to 14B+ parameter models\" (Crucible's X summary of its report; the PDF is 404).",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 2,
     "date": "2024-06-30",
     "who": "Hugging Face checkpoints",
     "sub": "before the origin",
     "kind": "read",
     "url": "https://huggingface.co/tensorplex-labs/Sumo-T9-7B-v0.1",
     "q": "About 50 miner uploads from January to June 2024, up to 7B (tensorplex-labs/pretraining-sn9-7B and others); the 7B cap dates from 19 April 2024.",
     "label": "stated",
     "from": "o",
     "approx": true
    }
   ],
   "figure": "14B",
   "unit": "largest pre-IOTA model, by August 2024"
  },
  {
   "id": "experiments",
   "short": "700 experiments, almost 15T tokens",
   "claim": "Between August 2025 and April 2026 Macrocosmos ran over 700 controlled experiments on its 1.5B, three-layer testbed, training almost 15 trillion tokens in total.",
   "note": "The post gives 700 on the 1.5B testbed, and 750 across 1.5B to 21B.",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "Steffen Cruz",
     "kind": "origin",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"Between August 2025 and April 2026 ... we ran over 700 controlled experiments on our 1.5B/3L testbed. In total, we trained for almost 15 trillion tokens during this period.\"",
     "label": "verified",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-06-01",
     "who": "Orion-100B, Substack",
     "sub": "the same post",
     "kind": "dot",
     "url": "https://macrocosmosai.substack.com/p/orion-100b-distributed-pretraining",
     "q": "\"By the time we launched Orion-100B, the system had been validated through over 750 experiments from 1.5B through to 21B parameter model scale.\"",
     "label": "verified",
     "id": "o750",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-04-20",
     "who": "@MacrocosmosAI, dashboard release",
     "sub": "X, seven weeks earlier",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2046259134446846259",
     "q": "\"since December 2025, IOTA has become 6x faster with better convergence\", \"driven by over 600 experiments\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "tao.media",
     "sub": "cites Macrocosmos",
     "kind": "dot",
     "url": "https://www.tao.media/macrocosmos-unveils-orion-100b-a-100b-parameter-distributed-ai-training-run/",
     "q": "\"over 700 experiments and ~15 trillion trained tokens\"",
     "label": "stated",
     "from": "o",
     "cites": true
    },
    {
     "lane": 1,
     "date": "2026-06-02",
     "who": "SimplyTao",
     "sub": "no source",
     "kind": "dot",
     "url": "https://simplytao.ai/blog/orion-100b-macrocosmos",
     "q": "\"The team ran over 700 controlled experiments on that testbed and trained for nearly 15 trillion tokens\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-06-05",
     "who": "Manifold Labs",
     "sub": "carries the figure",
     "kind": "dot",
     "url": "https://manifoldlabs.substack.com/p/capital-finds-the-machine",
     "q": "'Built on over 750 protocol-funded experiments.' The post gives 750 across 1.5B to 21B, and 700 on the testbed alone.",
     "label": "stated",
     "from": "o750"
    }
   ],
   "figure": "700",
   "unit": "experiments, almost 15T tokens on the testbed"
  },
  {
   "id": "orion16",
   "short": "Orion-16B: 256 GPUs, 100B tokens, about 20% MFU",
   "claim": "Orion-16B trained on 256 heterogeneous GPUs across three continents, passed 100B tokens by 14 August, and finished at about 80,000 tokens a second and roughly 20% MFU.",
   "note": "Four fleet sizes on four dates, all Macrocosmos's own: 256, c.180+, 225, and a mixed roster at the finish.",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-07-27",
     "who": "@MacrocosmosAI, launch thread",
     "sub": "X",
     "kind": "origin",
     "url": "https://x.com/MacrocosmosAI/status/2081791889674772618",
     "q": "\"16B parameters - 256 heterogeneous GPUs - 10 Pipeline stages\"",
     "label": "stated",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-07-30",
     "who": "@macrocrux",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/macrocrux/status/2082926870215917591",
     "q": "\"Orion-16B is training at an average MFU of around 23% making it the most efficient training run of its kind (and cheapest).\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-08-05",
     "who": "Deep Dive 2",
     "sub": "Substack",
     "kind": "dot",
     "url": "https://macrocosmosai.substack.com/p/deep-dive-2-the-technical-requirements",
     "q": "\"Orion-16B is currently training across c.180+ heterogeneous GPUs, a mix of RTX 4090s and 5090s, coordinated as a single run.\"",
     "label": "verified",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-08-14",
     "who": "@macrocrux",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/macrocrux/status/2088298605912174989",
     "q": "\"Orion-16B just passed 100B training tokens and its still improving\"; \"the largest LLM ever pretrained using DPP: Currently at 10^22 FLOPS\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-08-20",
     "who": "@macrocrux",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/macrocrux/status/2090544176274223467",
     "q": "\"Orion-16B is now training across 3 continents using 225 GPUs.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-23",
     "who": "@MacrocosmosAI, the finish",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/MacrocosmosAI/status/2102808976442458457",
     "q": "\"around 80k tokens per second across 4090s, 5090s, A100s, L40s, and A6000s, at roughly 20% MFU\"; \"about 18 B200s equivalent at FP16\"; \"Imperfect compute, one finished model.\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-08-03",
     "who": "cryptonews.net",
     "sub": "no source",
     "kind": "dot",
     "url": "https://cryptonews.net/news/altcoins/33236569/",
     "q": "\"Orion-16B training, a 16-billion-parameter model running on $IOTA across three continents\"; \"scaling to 256 GPUs at any given time\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-08-20",
     "who": "Subnet Alpha",
     "sub": "cites the X post",
     "kind": "dot",
     "url": "https://subnetalpha.ai/subnet/iota/",
     "q": "\"Orion-16B just passed 100B training tokens ... the largest LLM ever pretrained using DPP\". Quotes the 14 August post.",
     "label": "stated",
     "from": "o",
     "cites": true,
     "approx": true
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "@SubnetSummerTAO, from the stage",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/SubnetSummerTAO/status/2104591996459360637",
     "q": "\"100B tokens trained • 16B model at 24% MFU\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 2,
     "date": "2026-09-03",
     "who": "@Chibuzor_pat, dashboard snapshot",
     "sub": "X",
     "kind": "read",
     "url": "https://x.com/Chibuzor_pat/status/2095602058908901770",
     "q": "\"a 16.2B-parameter pretrain across ten pipeline stages and twenty-one replicas\"; 193.28B tokens and loss 2.977 at 04:09 UTC; \"a mid-run backend interruption and a 5.93B-token rollback\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 2,
     "date": "2026-09-28",
     "who": "The dashboard, run 28265",
     "sub": "read in a browser",
     "kind": "record",
     "url": "https://iota.macrocosmos.ai/dashboard/overview?run=28265",
     "q": "Tiles: tokens 219.05B, active nodes 116, parameters 16.2B; header \"10 stages · 11 replicas\"; \"Research · archive\".",
     "label": "verified",
     "from": "o"
    }
   ],
   "figure": "256",
   "unit": "GPUs at launch; 100B tokens; about 20% MFU"
  },
  {
   "id": "companies",
   "short": "164 companies in conversation, pilots from October",
   "claim": "Macrocosmos has 164 companies in conversation and expects pilots to go live from October 2026.",
   "note": "Said on stage on 28 September.",
   "nodes": [
    {
     "lane": 0,
     "date": "2026-09-28",
     "who": "Said on stage, not published",
     "sub": "no Macrocosmos text",
     "kind": "none",
     "url": "",
     "q": "Spoken at the Exploit Summit on 28 September 2026. Not in the announcement post, the SDK page, or any Substack post as of that evening.",
     "label": "unknown",
     "id": "o"
    },
    {
     "lane": 0,
     "date": "2026-09-28",
     "who": "@WSquires",
     "sub": "X, the same day",
     "kind": "dot",
     "url": "https://x.com/WSquires/status/2104681373533679638",
     "q": "\"We're ramping up slowly so we can serve customers super well, but we want to hear from you!\" No number given.",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "@SubnetSummerTAO, from the stage",
     "sub": "X",
     "kind": "dot",
     "url": "https://x.com/SubnetSummerTAO/status/2104591996459360637",
     "q": "\"164 companies already in conversation • Pilots expected to go live from October\"",
     "label": "stated",
     "from": "o"
    },
    {
     "lane": 1,
     "date": "2026-09-28",
     "who": "@tylerdurdeth, attendee",
     "sub": "X",
     "kind": "para",
     "url": "https://x.com/tylerdurdeth/status/2104614585030308119",
     "q": "Paraphrase of the market claim: \"the target market is in the billions of dollars with no serious competition\". No count.",
     "label": "stated",
     "from": "o"
    }
   ],
   "figure": "164",
   "unit": "companies in conversation, said on stage"
  }
 ],
 "timeline": [
  {
   "date": "2024-05-13",
   "tag": "Competition",
   "who": "Subnet 9: Scaling up parameters",
   "sub": "Substack",
   "url": "https://macrocosmosai.substack.com/p/subnet-9-scaling-up-parameters-to",
   "q": "Miners train whole models and the best one takes the reward; the parameter cap rises to 7B, \"a 10x increase.\"",
   "label": "stated"
  },
  {
   "date": "2024-09-21",
   "tag": "Distributed",
   "who": "Distributed training on Bittensor, announced live",
   "sub": "Opentensor Foundation stream",
   "url": "https://www.youtube.com/watch?v=UM5UhZadrRQ",
   "q": "\"Macrocosmos Major Update ++ Distributed training on Bittensor! [Announced LIVE]\"",
   "label": "stated"
  },
  {
   "date": "2024-12-22",
   "tag": "Federated",
   "who": "Distributed learning for decentralized LLMs: preliminary results",
   "sub": "Substack",
   "url": "https://macrocosmosai.substack.com/p/distributed-learning-for-decentralized",
   "q": "\"we had eight miners, all contributing to one global model\"; \"The federated learners consistently beat the single miner\"",
   "label": "stated"
  },
  {
   "date": "2025-06-17",
   "tag": "Swarm",
   "who": "IOTA: Bittensor's biggest pretraining breakthrough is here",
   "sub": "Substack; mainnet 2 June",
   "url": "https://macrocosmosai.substack.com/p/iota-bittensors-biggest-pretraining",
   "q": "\"Subnet 9 is now the home to the world's first ever model parallel and data parallel incentivized trustless pretraining protocol\"; the 2 June thread: \"Today, IOTA begins training a 15bn model across five layers.\"",
   "label": "verified"
  },
  {
   "date": "2026-06-17",
   "tag": "Liquid training",
   "who": "The Case for Liquid Training",
   "sub": "Substack and X",
   "url": "https://macrocosmosai.substack.com/p/the-case-for-liquid-training",
   "q": "\"iota turns idle, stranded compute into liquid training capacity for the global AI economy.\" \"Compute is the limiting reagent of modern intelligence.\"",
   "label": "verified"
  }
 ]
}
