{
  "dataset": "Foreverse Fiction Bench — novel-continuation model leaderboard",
  "version": "2026-07-24",
  "testedAt": "2026-07-16",
  "publishedAt": "2026-07-24",
  "priceSnapshotAt": "2026-07-24",
  "protocol": {
    "roundsPerChain": 20,
    "genres": [
      {
        "id": "xuanhuan",
        "book": "《踏天境》（传统玄幻，8.9M 字）",
        "directive": "legacy (pre-D2); see directiveNote"
      },
      {
        "id": "nvpin",
        "book": "《后宫·甄嬛传》（女频古言，210 章）",
        "directive": "D2 (fixed wording)"
      }
    ],
    "anchor": "55% 深度章内段落边界（两书同公式）",
    "temperature": 0.7,
    "blindReview": "双盲子 agent × 两套随机映射（玄幻 p303/p404，女频 p505/p606），评审互不知映射与对方存在",
    "directiveNote": "玄幻九链跑在旧 directive 条件下（三新模型与旧六链条件一致故可合并横评）；女频九链全部用 D2 终版 directive。gemini-3.1-pro 的玄幻第 9 名主要由「从头复读」造成，该复读已被消融实验证实是我们指令措辞的 bug（修复后 0/20）。",
    "costFormula": "usdPer10kChars = 25 segments × (8000 input tok × in$/M + 550 output tok × out$/M) / 1e6, list price, no cache discount"
  },
  "rankSemantics": "Ranks are double-blind review positions (two independent shuffled packs per genre), NOT absolute scores. Range ranks like '2-4' reflect low-confidence middle positions we refuse to split artificially.",
  "models": [
    {
      "slug": "deepseek-v4-flash",
      "id": "deepseek-v4-flash",
      "name": "DeepSeek V4 Flash",
      "vendor": "DeepSeek",
      "channel": "DeepSeek official API",
      "modelReleasedAt": "2026-04-24",
      "xuanhuan": {
        "rank": "1",
        "packs": "p303 第1 · p404 第1",
        "consensus": "Won all three windows; judges called it “the only system that gets better as it writes” — opens new arcs in the late window instead of flagging. High-confidence #1."
      },
      "courtRomance": {
        "rank": "6-7",
        "packs": "p505 第7 · p606 第6",
        "consensus": "The xuanhuan champion slid to lower-mid on court romance — its plain, fast-talking strengths don't transfer to formal period prose."
      },
      "longRunFailureMode": {
        "id": "none",
        "label": "None observed",
        "detail": "None of the three long-run failure modes observed across the 20-round chain; zero marker echoes."
      },
      "structuralNotes": "Sentence-length CV 0.56→0.64 (gold 0.837); dialogue-driven, standalone onomatopoeia, fast pacing; zero marker echoes in 20 rounds.",
      "crossGenreProfile": "King of xuanhuan: plain speech, fast pacing and barked forms of address are its native register; formal literary prose is not.",
      "pricing": {
        "inUsdPerM": 0.14,
        "outUsdPerM": 0.28,
        "contextWindow": 1000000,
        "usdPer10kChars": 0.0318
      }
    },
    {
      "slug": "deepseek-v4-pro",
      "id": "deepseek-v4-pro",
      "name": "DeepSeek V4 Pro",
      "vendor": "DeepSeek",
      "channel": "DeepSeek official API",
      "modelReleasedAt": "2026-04-24",
      "xuanhuan": {
        "rank": "2-4",
        "packs": "p303 第3 · p404 第4",
        "consensus": "Steady with no weak spot; the late-window surrender-negotiation scene was closest to the source — “like the same-genre author with a finer pen.”"
      },
      "courtRomance": {
        "rank": "1-2",
        "packs": "p505 第1 · p606 第2",
        "consensus": "“The only system that stably reproduces the source's two-layer structure — loaded dialogue plus narrated decoding”; its late Empress-Dowager trial scene was the peak of the whole pack."
      },
      "longRunFailureMode": {
        "id": "none",
        "label": "None observed",
        "detail": "No long-run failure observed on either genre chain."
      },
      "structuralNotes": "Systematically denser narration than the source (its xuanhuan demerit); sharpest colloquial dialogue in the field (“Leaving?” “I fold.”).",
      "crossGenreProfile": "Queen of court romance: the trait that cost it points on xuanhuan (a finer, denser pen) is exactly what scores on period prose — the same quality flips sign across genres.",
      "pricing": {
        "inUsdPerM": 0.435,
        "outUsdPerM": 0.87,
        "contextWindow": 1000000,
        "usdPer10kChars": 0.099
      }
    },
    {
      "slug": "gpt-5-6-terra",
      "id": "gpt-5.6-terra",
      "name": "GPT-5.6 Terra",
      "vendor": "OpenAI",
      "channel": "Eval gateway (OpenAI-compatible pass-through)",
      "modelReleasedAt": "2026-07-09",
      "xuanhuan": {
        "rank": "2-4",
        "packs": "p303 第4 · p404 第2",
        "consensus": "The prose-mimicry ceiling of the field, plus a late-chain plot rewind — its rank depends on how hard you punish the crash."
      },
      "courtRomance": {
        "rank": "1-3",
        "packs": "p505 第3 · p606 第1",
        "consensus": "“Zero accidents end to end; closest evidence-chain writing” — “no hard faults, steadily climbing into the late window.”"
      },
      "longRunFailureMode": {
        "id": "ending-rewind",
        "label": "Ending rewind",
        "detail": "Ending rewind (xuanhuan r18-20): the plot rewound to the continuation point and replayed chapter one, reusing its own early lines — found independently by both blind packs. The court-romance chain held: zero accidents."
      },
      "structuralNotes": "The only one of nine to clone micro-typography: tilde onomatopoeia (“rumble~~~”), the source's 的/地 particle habit, and pitch-perfect villain cadence.",
      "crossGenreProfile": "The texture-mimicry ceiling; crashed the xuanhuan 20-round marathon but held steady on romance — texture mimicry and long-run plot coherence are independent abilities.",
      "pricing": {
        "inUsdPerM": 2.5,
        "outUsdPerM": 15,
        "contextWindow": 1050000,
        "usdPer10kChars": 0.7063
      }
    },
    {
      "slug": "glm-5-2",
      "id": "glm-5.2",
      "name": "GLM-5.2",
      "vendor": "Zhipu AI",
      "channel": "Eval gateway (OpenAI-compatible pass-through)",
      "modelReleasedAt": "2026-06-13",
      "xuanhuan": {
        "rank": "2-4",
        "packs": "p303 第2 · p404 第3",
        "consensus": "Flat and drift-free end to end; half-width quotation marks were its only recurring demerit."
      },
      "courtRomance": {
        "rank": "2-3",
        "packs": "p505 第2 · p606 第3",
        "consensus": "“Purest limited-POV observational texture”; 90 half-width quote pairs recurred across books — an ingrained generation habit."
      },
      "longRunFailureMode": {
        "id": "none",
        "label": "None observed",
        "detail": "No long-run failure on either chain; its issue is a constant deviation (quote glyphs), not drift."
      },
      "structuralNotes": "The only model in the top three on both genres; half-width quotes throughout (108 pairs xuanhuan / 90 romance) — full-width-only detectors read 0% dialogue when the real rate was 13.5%, a documented metric artifact.",
      "crossGenreProfile": "The only stable top-three across both genres; half-width quotes are its cross-book stubborn habit.",
      "pricing": {
        "inUsdPerM": 1.4,
        "outUsdPerM": 4.4,
        "contextWindow": 1000000,
        "usdPer10kChars": 0.3405
      }
    },
    {
      "slug": "qwen3-7-max",
      "id": "qwen3.7-max",
      "name": "Qwen3.7-Max",
      "vendor": "Alibaba Qwen",
      "channel": "Eval gateway (OpenAI-compatible pass-through)",
      "modelReleasedAt": "2026-05-21",
      "xuanhuan": {
        "rank": "5-6",
        "packs": "p303 第5 · p404 第6",
        "consensus": "A constant “scent fingerprint” — smell/touch description in every window of a source book that contains zero scent writing; consistently passable, consistently unlike."
      },
      "courtRomance": {
        "rank": "8",
        "packs": "p505 第8 · p606 第8",
        "consensus": "A unanimous #8 in both packs; the “polished sensory stream” reads even more out of place in period prose."
      },
      "longRunFailureMode": {
        "id": "none",
        "label": "None observed",
        "detail": "No long-run failure; its issue is a constant stylistic overlay, not degradation over rounds."
      },
      "structuralNotes": "Best structural metrics in the field (sentence length 32.1 vs gold 32, 18.1% dialogue, lowest stock-phrase density 1.74‰) — yet blind-ranked 5-6: statistics can't measure “writing what the source never writes.”",
      "crossGenreProfile": "The “polished sensory stream” clashes hardest with period prose; the field's exemplar of statistically-closest, temperamentally-furthest.",
      "pricing": {
        "inUsdPerM": 2.5,
        "outUsdPerM": 7.5,
        "contextWindow": 1000000,
        "usdPer10kChars": 0.6031
      }
    },
    {
      "slug": "claude-opus-4-8",
      "id": "claude-opus-4.8",
      "name": "Claude Opus 4.8",
      "vendor": "Anthropic",
      "channel": "yunwu aggregator gateway",
      "modelReleasedAt": "2026-05-28",
      "xuanhuan": {
        "rank": "5-6",
        "packs": "p303 第6 · p404 第5",
        "consensus": "Literati cadence, calling a humanoid demon lord “it,” and late half-width-punctuation drift; the widest swing of the earlier six-model round (worst early → mid highlight → late decay)."
      },
      "courtRomance": {
        "rank": "4-5",
        "packs": "p505 第4 · p606 第5",
        "consensus": "“Sharpest verbal fencing in the pack” — but with craft-level decay: increasing half-width punctuation and drifting name-rank forms."
      },
      "longRunFailureMode": {
        "id": "none",
        "label": "None observed",
        "detail": "No failure in the taxonomy sense; its late-window punctuation drift is craft decay, not plot failure."
      },
      "structuralNotes": "A deep-V curve: heaviest purple prose early (“a faint arc, like a slumbering cosmos”) → a mid-window highlight (its execution scene could rank #3) → systematic late half-width-punctuation decay.",
      "crossGenreProfile": "Stable mid-field; the poster case for “writes well” and “writes like the source” being different things.",
      "pricing": {
        "inUsdPerM": 5,
        "outUsdPerM": 25,
        "contextWindow": 1000000,
        "usdPer10kChars": 1.3438
      }
    },
    {
      "slug": "gemini-3-1-pro",
      "id": "gemini-3.1-pro",
      "name": "Gemini 3.1 Pro",
      "vendor": "Google",
      "channel": "yunwu aggregator gateway",
      "modelReleasedAt": "2026-02-19",
      "xuanhuan": {
        "rank": "9",
        "packs": "p303 第9 · p404 第9",
        "consensus": "Under the old directive it rewound to the continuation point 20/20 rounds (rewriting the opening every time, zero progress) — ablation proved this was a bug in our instruction wording; 0/20 after the fix."
      },
      "courtRomance": {
        "rank": "4-6",
        "packs": "p505 第6 · p606 第4",
        "consensus": "With the fixed directive it climbed from disqualified to mid-field; its simile-stock density of 2.50‰ stayed the field's highest — the purple prose is its nature, the looping was our bug."
      },
      "longRunFailureMode": {
        "id": "restart-loop",
        "label": "Restart loop",
        "detail": "Restart loop (triggered by our old directive): every round returned to the source's last line and rewrote the opening — 20 rounds, zero progress. Three-way ablation: old wording 20/20 rewinds, no explanation ~6/8, fixed wording 0/20. Its default reading of an unexplained marker was “annotated text isn't canon”; the explanation sentence is load-bearing."
      },
      "structuralNotes": "Highest simile-stock density in the field (2.47‰ xuanhuan post-fix / 2.50‰ romance vs the 1.0‰ source baseline); constant purple-prose lean.",
      "crossGenreProfile": "The directive-wording fix was its watershed: from disqualified to normal competition.",
      "pricing": {
        "inUsdPerM": 2,
        "outUsdPerM": 12,
        "contextWindow": 1048576,
        "usdPer10kChars": 0.565
      }
    },
    {
      "slug": "kimi-k2-6",
      "id": "kimi-k2.6",
      "name": "Kimi K2.6",
      "vendor": "Moonshot AI",
      "channel": "Eval gateway (OpenAI-compatible pass-through)",
      "modelReleasedAt": "2026-04-21",
      "xuanhuan": {
        "rank": "7",
        "packs": "p303 第7 · p404 第7",
        "consensus": "Highest simile density in the field (nearly one per paragraph), whole-passage self-copying from early to mid, and setting slippage (an ancient-tree valley sprouting inside a void blood-array)."
      },
      "courtRomance": {
        "rank": "5-7",
        "packs": "p505 第5 · p606 第7",
        "consensus": "One pack flagged it as “the only clear reverse-improver” — webnovel rage-cadence early, settling down by the late window."
      },
      "longRunFailureMode": {
        "id": "mid-collapse",
        "label": "Mid-chain collapse",
        "detail": "Mid-chain collapse (whole-passage self-copying early→mid): archived chain r9 hit 97.6% cross-round 12-gram repetition. Successor K3, retested under the same protocol, showed no verbatim failure (tt peak 6.8%) — see the incremental duels section."
      },
      "structuralNotes": "Field-highest simile density (nearly one per paragraph); archived xuanhuan chain r9 measured 97.6% cross-round 12-gram repetition — machine evidence of whole-passage self-copying.",
      "crossGenreProfile": "Lower-mid on both; the simile-density ailment crosses genres.",
      "pricing": {
        "inUsdPerM": 0.95,
        "outUsdPerM": 4,
        "contextWindow": 262144,
        "usdPer10kChars": 0.245
      }
    },
    {
      "slug": "grok-4-5",
      "id": "grok-4.5",
      "name": "Grok 4.5",
      "vendor": "xAI",
      "channel": "Eval gateway (OpenAI-compatible pass-through)",
      "modelReleasedAt": "2026-07-08",
      "xuanhuan": {
        "rank": "8",
        "packs": "p303 第8 · p404 第8",
        "consensus": "Sensory-barrage run-on sentences, plus a mid window looping the same two passages verbatim three times — found independently by both packs."
      },
      "courtRomance": {
        "rank": "9",
        "packs": "p505 第9 · p606 第9",
        "consensus": "A unanimous last place; looping and plot-rewind recurred across genres (verbatim late repeats, 20 rounds stalled on the day of the incident) plus first-person density at 5‰ — half the source's — a collapse of POV discipline."
      },
      "longRunFailureMode": {
        "id": "mid-collapse",
        "label": "Mid-chain collapse",
        "detail": "Mid-chain collapse: the xuanhuan mid window looped the same two passages verbatim three times (late recovered with new text); the romance chain repeated verbatim late and rewound its plot, stalling all 20 rounds on the day of the incident — the ailment recurs across genres."
      },
      "structuralNotes": "Sensory-barrage run-on sentences; on romance its first-person density of 5‰ was half the source's — limited-POV discipline collapsed.",
      "crossGenreProfile": "Bottom of both genres; the looping/rewind ailment recurs across genres.",
      "pricing": {
        "inUsdPerM": 2,
        "outUsdPerM": 6,
        "contextWindow": 500000,
        "usdPer10kChars": 0.4825
      }
    }
  ],
  "incrementalDuels": [
    {
      "id": "kimi-k3-vs-k2-6",
      "challenger": "Kimi K3",
      "incumbent": "Kimi K2.6 (board: #7 / 5-7)",
      "challengerReleasedAt": "2026-07-16",
      "testedAt": "2026-07-18",
      "votes": {
        "xuanhuan": "10 : 0（2 票无效）",
        "courtRomance": "11 : 1"
      },
      "verdict": "K3 sweeps its predecessor on both genres: stock-phrase density converged sharply (0.63‰ on romance, below the source baseline), and K2.6's whole-passage self-copying vanished (cross-round repetition peak 6.8% vs 97.6%). The single dissenting romance vote flipped with the mapping — position bias. The half-width-quote habit remains (~85% of dialogue).",
      "note": "Paired double-blind duels only; not merged into the nine-model full ranking."
    },
    {
      "id": "qwen3-8-vs-qwen3-7",
      "challenger": "Qwen3.8-Max-Preview",
      "incumbent": "Qwen3.7-Max (board: 5-6 / #8)",
      "challengerReleasedAt": "2026-07-19",
      "testedAt": "2026-07-21",
      "votes": {
        "xuanhuan": "7 : 5",
        "courtRomance": "11 : 1"
      },
      "verdict": "A crushing upgrade on romance (11:1, all three windows) — 3.7's polished-sensory-stream ailment visibly converged; on xuanhuan only a slim 7:5, and fixing word-level repetition bought plot-level looping (judges: “still stuck in an escape-and-seal loop after 20 rounds”). Against the same month's Kimi K3 it lost 1:10 / 2:10. Previews are moving targets; conclusions bind to the 2026-07-21 hosted build.",
      "note": "Paired double-blind duels only; not merged into the nine-model full ranking."
    },
    {
      "id": "gemini-3-6-flash-vs-3-5",
      "challenger": "Gemini 3.6 Flash",
      "incumbent": "Gemini 3.5 Flash (not on the main board)",
      "challengerReleasedAt": "2026-07-21",
      "testedAt": "2026-07-23",
      "votes": {
        "xuanhuan": "9 : 3",
        "courtRomance": "1 : 10"
      },
      "verdict": "Opposite directions on the two books — 3.6 wins xuanhuan 9:3 but loses romance 1:10 (five judges stable for 3.5 across both mappings). The essence isn't genre preference but “who crashes uglier”: under this 20-round protocol both generations fell into plot loops; on xuanhuan 3.5 collapsed to the verbatim level, on romance it was 3.6 that did. A generational update is not a monotonic upgrade — it's a redistribution of failure modes.",
      "note": "Paired double-blind duels only; not merged into the nine-model full ranking."
    }
  ],
  "sources": [
    "Nine-model dual-genre evaluation: Foreverse Research, M69 style-directive study, 2026-07-16 (20-round chains, double-blind, structural metrics)",
    "Incremental duels: Foreverse Research eval archives (Kimi K3 2026-07-18, Qwen3.8 2026-07-21, Gemini 3.6 Flash 2026-07-23)",
    "Prices: models.dev list-price snapshot 2026-07-24"
  ],
  "howToCite": "Foreverse Research, “Fiction Bench: novel-continuation model leaderboard,” 2026-07. https://foreverse.app/research/fiction-bench",
  "homepage": "https://foreverse.app/research/fiction-bench"
}