{
  "_help": "Insight cards shown at the top of the site (#insights), in this order. Fields: id (→ #insight-<id>), headline, big (the big number), bigLabel (what the number is), text (one plain line), href + evidence (link to the chart/section), stats / n / ci / suite (advanced view), featured (largest card). live: 'calibration' | 'agreement' | 'sizing' → big/text/stats are recomputed from bench.json when the export has that block; the static values here are the fallback. Markdown subset in text/stats: **bold**, *italic*, `code`.",
  "cards": [
    {
      "id": "calibration",
      "featured": true,
      "live": "calibration",
      "headline": "The models know when they're unsure",
      "big": "78% vs 54%",
      "bigLabel": "best move when sure vs. when torn",
      "text": "When Luna is more than 80% sure it picks the best move 78% of the time; when it's torn (40–60%) only 54% — and those decisions cost about 3× more (2.9% vs 8.6% of the pot).",
      "href": "#confidence",
      "evidence": "See the confidence chart",
      "stats": "Luna, top probability > 80%: best 78%, loss 2.9% pot; 60–80%: 69%, 4.8%; 40–60%: 54%, 8.6%; < 40%: 32%, 39.9% (n = 28).",
      "n": "8,865 answers",
      "ci": "",
      "suite": "All stored answers (every suite, split and variant)"
    },
    {
      "id": "confident-info",
      "headline": "Better information makes them more sure — and still honest about it",
      "big": "39% → 59%",
      "bigLabel": "decisions where the model is ‘very sure’",
      "text": "With the better description, models were very sure on 59% of decisions instead of 39%, and every confidence level got cheaper (very sure: 2.3% → 1.7% of the pot). Torn decisions still cost ~60% more — the warning light keeps working.",
      "href": "#confidence",
      "evidence": "See the confidence chart",
      "stats": "All five models, five full-hand suites, v2 (n=4,941) vs nuts (n=4,336): >80% bucket 39% → 59% of decisions, loss 2.33 → 1.72% pot; 60–80% 3.13 → 2.13%; 40–60% 4.15 → 2.80%; best-pick rate 78/69/64% → 78/73/71%.",
      "n": "9,277 answers",
      "ci": "",
      "suite": "Cash dev/test, MTT 40/20bb, Spin full hands"
    },
    {
      "id": "description",
      "headline": "How you describe the hand matters as much as which model you use",
      "big": "−30–50%",
      "bigLabel": "postflop losses",
      "text": "Telling the model “what beats me” and its win chance against the likely range cut losses by 30–50% — for all five models, in cash games, tournaments and Spin & Go, confirmed on hands they had never seen.",
      "href": "#luna",
      "evidence": "See the experiments",
      "stats": "`nuts` vs `v2`: cash test EV loss 2.19% → 1.52% pot (paired Δ −0.67 ± 0.65%); bb/100 lost MTT40 89 → 55, MTT20 56 → 40, Spin 46 → 28 (all paired-significant). All five models on MTT 20bb (300 decisions): Clef 3.35% → 1.84%, Mercury 3.71% → 2.09%, d1 2.40% → 1.79%, Luna 2.49% → 1.86%, Jev 2.71% → 2.42% pot. Mercury cash (free endpoint): 4.3% → 1.8–2.3% pot on both dev and test halves.",
      "n": "324 test decisions + 3 independent suites",
      "ci": "cash test Δ −0.67 ± 0.65% pot",
      "suite": "Cash dev/test split, MTT 40bb, MTT 20bb, Spin full hands"
    },
    {
      "id": "cheap-model",
      "headline": "A cheap model can match the best one",
      "big": "89 = 89",
      "bigLabel": "points, at 1/5 of the price",
      "text": "With the better description the five models converge: four land within one point of each other (88–89/100 on 300 tournament decisions). Mercury Decide matched GPT-6 Luna at about one fifth of the price — and is also available free.",
      "href": "#luna",
      "evidence": "See the experiments",
      "stats": "MTT 20bb, nuts variant, 300 decisions: Mercury 88.9, Luna 88.7, d1 88.4, Clef 88.2, Jev 85.6 (v2 spread 80.8–86.5). Cost: Mercury $0.008 vs Luna $0.040 per 300 decisions.",
      "n": "300 decisions",
      "ci": "",
      "suite": "MTT 20bb full hands"
    },
    {
      "id": "agreement",
      "live": "agreement",
      "headline": "When all five agree, they're usually right",
      "big": "0.55% vs 4.43%",
      "bigLabel": "of the pot lost: unanimous vs. split 3–2",
      "text": "Unanimous picks lost 0.55% of the pot vs 4.4% when split 3–2, and a majority vote beat the best single model (1.99% vs 2.57%). Once the hand is described well, the models agree far more often (65% unanimous) and voting no longer adds much.",
      "href": "#agreement",
      "evidence": "See the agreement chart",
      "stats": "200-spot suite (v2): 5/5 35% of spots, 0.55% pot; 3/5 28%, 4.43%; vote 1.99% vs Luna 2.57%. MTT 20bb (nuts, 300): 5/5 65%, 1.08%; 4/5 21%, 3.33%; vote 1.90% vs best single model (d1) 1.79%.",
      "n": "200 spots × 5 models",
      "ci": "",
      "suite": "200-spot suite, v2"
    },
    {
      "id": "money",
      "headline": "Closer to perfect ≠ more money — it depends on the opponent",
      "big": "+275",
      "bigLabel": "bb/100 vs a maniac",
      "text": "Over ~15,000 full hands, the better description made Mercury crush an over-aggressive bot far harder (+275 bb/100, significant) but changed nothing against a solid regular (≈0). No model reliably beats the solid regular; every model beats a calling station.",
      "href": "#arena",
      "evidence": "See the arena",
      "stats": "Paired on identical deals, ~plus − standard: Mercury vs maniac +274.8 ± 165.0 (500 deals); vs TAG @100bb pooled ≈ +1 ± 11 (3,250 deals: +9.5 ± 17.7, −0.0 ± 16.3, −26.5 ± 45.3); vs TAG @20bb −13.4 ± 12.3 (1,000 deals). Luna vs TAG: +5.7 ± 23.9 @100bb (500), +12.7 ± 15.6 @20bb (400). Standard vs TAG: Mercury +22 ± 19 (2,000 deals), Luna −21 ± 22 (500).",
      "n": "100–250 duplicate deals per match",
      "ci": "±33 to ±62 bb/100",
      "suite": "Arena, 100bb heads-up"
    },
    {
      "id": "escalation",
      "headline": "A second opinion only when unsure helps a little",
      "big": "−6%",
      "bigLabel": "mistakes, escalating only ~5–7% of moves",
      "text": "Handing only Luna's least-confident moves (about 5–7%) to the other models' vote trimmed its mistakes by about 6%; handing over more made things worse. Useful, but a small gain once the description is already good.",
      "href": "#confidence",
      "evidence": "See the escalation simulation",
      "stats": "MTT 20bb, nuts, 300 decisions: Luna alone 1.86% pot → 1.76% escalating to the 5-model vote when top probability < 0.6 (7% escalated); < 0.7 (20%) → 1.81%. Earlier 150-decision subset: 1.56% → 1.37% (−12%).",
      "n": "150 decisions all models answered",
      "ci": "",
      "suite": "MTT 20bb full hands, nuts"
    },
    {
      "id": "shortstack",
      "headline": "One sentence of advice halved short-stack preflop mistakes",
      "big": "18.6 → 8.9",
      "bigLabel": "bb/100 lost, Spin & Go preflop",
      "text": "“Limp, fold or go all-in; small raises are rarely right” more than halved Luna's preflop losses in Spin & Go. Before, it never went all-in preflop.",
      "href": "#luna",
      "evidence": "See the experiments",
      "stats": "`shortstack` vs `v2`: Spin preflop score 75.7 → 87.2 and 71.0 → 84.6 on an independent suite; MTT20 71.4 → 84.0 (all significant).",
      "n": "228 + 268 Spin preflop decisions",
      "ci": "",
      "suite": "Spin preflop suites, MTT 20bb preflop"
    },
    {
      "id": "facing-bet",
      "headline": "Facing a bet is where better information helps most",
      "big": "4.0% → 2.0%",
      "bigLabel": "of the pot lost when facing a bet",
      "text": "With the improved description, Luna's mistakes when facing a bet halved.",
      "href": "#luna",
      "evidence": "See the experiments",
      "stats": "Luna `nuts` vs `v2`, postflop, all four formats: facing a bet 4.01% → 1.99% pot (−50%).",
      "n": "1,231 postflop decisions",
      "ci": "",
      "suite": "Cash, MTT 40bb, MTT 20bb, Spin full hands"
    },
    {
      "id": "sizing",
      "live": "sizing",
      "headline": "Bet sizing is a hidden cost",
      "big": "34%",
      "bigLabel": "of Luna's loss: right move, wrong size",
      "text": "34% of Luna's loss comes from picking the right action at the wrong size (Clef: 5%). We tried explaining what each bet size is for — it didn't help, and was slightly worse than the simpler fix.",
      "href": "#agreement",
      "evidence": "See the sizing chart",
      "stats": "Share of EV loss where the chosen and best actions are both bet/raise/all-in of different sizes: Luna 34%, Mercury 27%, Jev 20%, d1 11%, Clef 5%. `sizing` variant vs `nuts`: Luna cash dev 2.24 → 2.28%, test 1.52 → 1.74% pot; Mercury (free) slightly worse on all 5 suites.",
      "n": "200 spots per model",
      "ci": "",
      "suite": "200-spot suite, v2"
    },
    {
      "id": "monsters",
      "headline": "Monster hands are their biggest leak",
      "big": "5.2%",
      "bigLabel": "of the pot lost with > 85% equity",
      "text": "With very strong hands the models lose about 5% of the pot per decision, versus ~1% with medium hands: they raise when they should trap and check the nuts on the river.",
      "href": "#luna",
      "evidence": "See the experiments",
      "stats": "Luna `nuts`, postflop: monsters (> 85% equity) 5.2% pot (from 7.7% with `v2`) — still the largest leak.",
      "n": "1,231 postflop decisions",
      "ci": "",
      "suite": "Cash, MTT 40bb, MTT 20bb, Spin full hands"
    },
    {
      "id": "deep-vs-short",
      "headline": "Deep cash games are the hardest; very short stacks barely benefit",
      "big": "131 vs 28",
      "bigLabel": "bb/100 lost postflop: cash vs Spin",
      "text": "After the flop, the improved models still give away the most in deep cash games. At 13 big blinds heads-up the “what beats me” fact didn't help.",
      "href": "#formats",
      "evidence": "See the formats",
      "stats": "Postflop bb/100 lost (improved): cash 131 vs Spin 28. Spin 15bb heads-up 2.46% pot (no gain) vs Spin 25bb 1.09% (from 3.06%).",
      "n": "",
      "ci": "",
      "suite": "Full-hand suites per format"
    },
    {
      "id": "personality",
      "headline": "Each model has a personality",
      "big": "45% vs 21%",
      "bigLabel": "aggressive actions vs. perfect play",
      "text": "Mercury, d1, Jev and Clef bet and raise too often, Luna is passive (10%), Clef calls too much. Their opposite biases are why averaging and voting help.",
      "href": "#arena",
      "evidence": "See the arena stats",
      "stats": "Share of aggressive actions: Mercury, d1, Jev, Clef ≈ 45%; GTO 21%; Luna 10%.",
      "n": "",
      "ci": "",
      "suite": "200-spot suite and arena"
    },
    {
      "id": "mixing",
      "headline": "Rolling the dice on their own odds doesn't pay",
      "big": "−31",
      "bigLabel": "bb/100 vs top pick",
      "text": "Decision models return a probability for every move. Playing a move at random according to those odds — instead of just taking the favourite — cost Mercury about 31 big blinds per 100 hands against the solid bot. Their probabilities are informative, but too spread out to play as-is.",
      "href": "#arena",
      "evidence": "See the arena",
      "stats": "Mercury ~mix − argmax vs TAG @100bb: −31.2 ± 40.2 bb/100 (1,000 deals, paired; not significant). Benchmark: playing the reported mix loses more EV than the argmax pick for every model; sharpening (τ=4) recovers most of it.",
      "n": "2,000 hands",
      "ci": "−31.2 ± 40.2",
      "suite": "Arena, 100bb duplicate"
    },
    {
      "id": "hud",
      "headline": "Giving them opponent stats backfired",
      "big": "−79",
      "bigLabel": "bb/100 vs a calling station",
      "text": "Real players exploit opponents using stats (how often they fold, bet, call). When we fed the model a live summary of its opponent's tendencies, it didn't exploit better — against a calling station it won 79 big blinds per 100 hands less. It seems to over-adjust.",
      "href": "#arena",
      "evidence": "See the arena",
      "stats": "Mercury ~plus, opponent-stats summary on vs off, paired on identical deals: vs calling station −78.9 ± 44.4 bb/100 (500 deals, significant); vs TAG +9.1 ± 12.1 (1,000 deals); vs maniac −93.3 ± 194.2 (500 deals).",
      "n": "4,000 hands",
      "ci": "−78.9 ± 44.4",
      "suite": "Arena, 100bb duplicate"
    }
  ]
}