← Files Avoid AI WritingARCHIVED FILE

skills/preservation-verifier/scripts/patterns.js

127 KB · Oct 2, 2026 · 00:32 UTC

↓ Download file

/**
 * Avoid AI Writing — detection engine (canonical source of truth)
 * Implements regex, structural, and stylometric pattern detection. This repo's SKILL.md
 * catalogs the human-editable pattern rules; this engine is the executable
 * expression of the regex-detectable subset and extends it with stylometric and
 * AI-tool-fingerprint detectors that don't make sense as skill prose
 * (cross-paragraph burstiness, smart-punct signatures, function-word
 * trigram entropy, low type-token ratio, AI-tool URL parameters,
 * chatbot citation markup leaks, unfilled placeholders).
 *
 * Scoring model:
 *   Each category has a weight in the ISSUE_WEIGHTS table. Detection runs
 *   produce raw (possibly duplicate) issues which are then deduplicated by
 *   (type, text) pair. rawScore is the sum of category weights across the
 *   deduped list — so the number reflects the same distinct signals the
 *   user sees in the issue list.
 *
 *   Weights are deliberately non-flat across severity tags. Cutoff
 *   disclaimers (10) and chatbot artifacts (8) weigh more than vague
 *   attributions (5), even though all three are tagged `critical`, because
 *   the skill treats them as stronger or weaker AI-origin signals.
 *
 *   rawScore is then normalized to 0-100 via `log2(wordCount/50)` so longer
 *   texts don't accumulate unboundedly on the same density of patterns.
 */

const AIDetector = (() => {
  // ═══ Tier 1 pre-pass: normalize bypass tricks ══════════════════════
  //
  // Humanizer tools and prompt-injection bypass techniques insert
  // invisible / lookalike chars to defeat exact-string detectors. Strip
  // them BEFORE pattern matching so "delve" with a Cyrillic 'е' still
  // hits the Tier 1 list. Unicode ranges sourced from
  // It-s-AI/llm-detection/detection/attacks/.
  //
  // Tracks what was stripped so the trinary classifier can use
  // "normalization triggered" as a corroborating AI signal — humans don't
  // paste ZWSPs into their own writing.
  const CYRILLIC_LOOKALIKES = {
    'а': 'a', 'е': 'e', 'о': 'o', 'р': 'p', 'с': 'c', 'х': 'x',
    'у': 'y', 'к': 'k', 'м': 'm', 'н': 'h', 'в': 'b', 'т': 't',
    'А': 'A', 'Е': 'E', 'О': 'O', 'Р': 'P', 'С': 'C', 'Х': 'X',
    'У': 'Y', 'К': 'K', 'М': 'M', 'Н': 'H', 'В': 'B', 'Т': 'T',
  };
  const GREEK_LOOKALIKES = { 'ο': 'o', 'Ο': 'O', 'α': 'a', 'Α': 'A', 'ρ': 'p', 'Ρ': 'P' };

  function normalizeText(text) {
    const flags = { zeroWidth: 0, homoglyph: 0, roleplay: 0 };
    let out = text;

    // 1. Strip zero-width chars (ZWSP U+200B, ZWNJ U+200C, ZWJ U+200D,
    //    BOM U+FEFF, word joiner U+2060).
    out = out.replace(/[​-‍⁠]/g, () => { flags.zeroWidth++; return ''; });

    // 2. Swap Cyrillic / Greek Latin-lookalike chars back to Latin so
    //    pattern matching catches obfuscated tokens.
    out = out.replace(/[Ѐ-ӿͰ-Ͽ]/g, (m) => {
      const swap = CYRILLIC_LOOKALIKES[m] ?? GREEK_LOOKALIKES[m];
      if (swap) { flags.homoglyph++; return swap; }
      return m;
    });

    // 3. Strip *roleplay-action* markers — paired *...* containing an
    //    action verb (nods, sighs, laughs, smiles, etc.) anchored to
    //    the start of the inner phrase. This is the actual chat-model
    //    artifact shape. Markdown `**bold**` is rejected by the
    //    lookbehind/lookahead; legitimate multi-word `*italic*` is
    //    preserved because the verb whitelist is narrow.
    const ROLEPLAY_VERBS = /^(?:nods|sighs|laughs|smiles|frowns|shrugs|grins|winks|chuckles|gasps|pauses|thinks|wonders|whispers|shouts|gestures|raises|leans|turns|looks|glances|smirks|blinks|nodding|sighing|laughing|smiling|thinking|gesturing)\b/i;
    out = out.replace(/(?<!\*)\*([^*\n]{1,80}?)\*(?!\*)/gu, (m, inner) => {
      if (ROLEPLAY_VERBS.test(inner)) { flags.roleplay++; return ''; }
      return m;
    });

    return { text: out, flags };
  }

  // ─── Tier 1: Always flag ───────────────────────────────────────────
  const TIER1 = {
    'delve': 'explore, dig into, look at',
    'tapestry': 'describe the actual complexity',
    'paradigm': 'model, approach, framework',
    'beacon': 'rewrite entirely',
    'robust': 'strong, reliable, solid',
    'comprehensive': 'thorough, complete, full',
    'cutting-edge': 'latest, newest, advanced',
    'pivotal': 'important, key, critical',
    'meticulous': 'careful, detailed, precise',
    'meticulously': 'carefully, precisely',
    'seamless': 'smooth, easy, without friction',
    'seamlessly': 'smoothly, easily',
    'game-changer': 'describe what changed',
    'game-changing': 'describe what changed',
    'nestled': 'is located, sits',
    'vibrant': 'describe what makes it active',
    'thriving': 'growing, active',
    'bustling': 'busy, active',
    'intricate': 'complex, detailed',
    'intricacies': 'complexities, details',
    'ever-evolving': 'changing, growing',
    'enduring': 'lasting, long-running',
    'daunting': 'hard, difficult',
    'holistic': 'complete, full, whole',
    'holistically': 'completely, fully',
    'actionable': 'practical, useful, concrete',
    'impactful': 'effective, significant',
    'learnings': 'lessons, findings, takeaways',
    'synergy': 'describe the combined effect',
    'synergies': 'describe the combined effect',
    'interplay': 'relationship, connection',
    'symphony': 'describe the coordination',
    'embrace': 'adopt, accept, use',
  };

  // Multi-word tier 1 phrases
  const TIER1_PHRASES = [
    { pattern: /\bdelve\s+into\b/gi, replace: 'explore, dig into' },
    { pattern: /\blandscape\b/gi, replace: 'field, space, industry', filter: true },
    { pattern: /\brealm\b/gi, replace: 'area, field, domain' },
    { pattern: /\btestament\s+to\b/gi, replace: 'shows, proves' },
    { pattern: /\bleverag(?:e|es|ing|ed)\b/gi, replace: 'use' },
    { pattern: /\bwatershed\s+moment\b/gi, replace: 'turning point, shift' },
    { pattern: /\bmarking\s+a\s+pivotal\s+moment\b/gi, replace: 'state what happened' },
    { pattern: /\bthe\s+future\s+looks\s+bright\b/gi, replace: 'cut or say something specific' },
    { pattern: /\bonly\s+time\s+will\s+tell\b/gi, replace: 'cut or say something specific' },
    { pattern: /\bdespite\s+challenges[^.]*continues?\s+to\s+thrive\b/gi, replace: 'name the challenge and response' },
    { pattern: /\bdeep\s+dive\b/gi, replace: 'look at, examine' },
    { pattern: /\bdive\s+into\b/gi, replace: 'look at, examine' },
    { pattern: /\bunpack(?:ing)?\b/gi, replace: 'explain, break down' },
    { pattern: /\bcomplexities\b/gi, replace: 'name the actual problems' },
    { pattern: /\bthought\s+leader(?:ship)?\b/gi, replace: 'expert, authority' },
    { pattern: /\bbest\s+practices\b/gi, replace: 'what works, proven methods' },
    { pattern: /\bat\s+its\s+core\b/gi, replace: 'cut, just state it' },
    { pattern: /\bin\s+order\s+to\b/gi, replace: 'to', clarity: true },
    { pattern: /\bdue\s+to\s+the\s+fact\s+that\b/gi, replace: 'because', clarity: true },
    { pattern: /\bserves\s+as\b/gi, replace: 'is', clarity: true },
    { pattern: /\bfeatures\b/gi, replace: 'has, includes', filter: true, clarity: true },
    { pattern: /\bboasts\b/gi, replace: 'has', clarity: true },
    { pattern: /\butiliz(?:e|es|ing|ed)\b/gi, replace: 'use', clarity: true },
    { pattern: /\bshowcas(?:e|es|ing|ed)\b/gi, replace: 'show, demonstrate' },
    { pattern: /\bembark(?:s|ing|ed)?\b/gi, replace: 'start, begin' },
    { pattern: /\bcommenc(?:e|es|ing|ed)\b/gi, replace: 'start, begin', clarity: true },
    { pattern: /\bascertain(?:s|ing|ed)?\b/gi, replace: 'find out, determine', clarity: true },
    { pattern: /\bendeavou?r(?:s|ing|ed)?\b/gi, replace: 'effort, attempt, try', clarity: true },
    { pattern: /\bunderscor(?:es|ing|ed)\b/gi, replace: 'highlights, shows' },
    // Hyphen required. The unhyphenated "load bearing" is ordinary English —
    // "the load bearing down on the bridge" — where `bearing` is a participle,
    // not part of a compound modifier. The tell is always hyphenated.
    //
    // Construction carve-out: exempt attributive `load-bearing` before a literal
    // structural noun, with one optional material/position adjective in between
    // ("load-bearing structural wall"). The noun list is limited to commonly
    // physical nouns; the abstract-capable ones most likely to carry the
    // metaphor (structure, element, frame, foundation) are omitted so "the
    // load-bearing structure of the argument" still fires. Some listed nouns
    // (member, column, partition) can still be used metaphorically and are
    // silently exempt — a recall loss in the safe direction, tracked in #56.
    // Predicative use ("the wall is load-bearing") is NOT exempt: the tell
    // lives in the subject, which a lookahead cannot reach. Also #56.
    { pattern: /\bload-bearing\b(?!\s+(?:(?:structural|exterior|interior|internal|external|concrete|steel|timber|wooden|brick|masonry|perimeter|basement|main|primary|existing|original)\s+)?(?:walls?|beams?|columns?|joists?|truss(?:es)?|members?|footings?|slabs?|studs?|partitions?|masonry|lintels?|piers?|rafters?|girders?|capacity|capacities)\b)/gi, replace: 'essential, critical, or say what breaks if you remove it' },
  ];

  // ─── Tier 2: Flag in clusters (2+ per paragraph) ──────────────────
  const TIER2 = {
    'harness': 'use, take advantage of',
    'navigate': 'work through, handle',
    'navigating': 'working through, handling',
    'foster': 'encourage, support, build',
    'elevate': 'improve, raise, strengthen',
    'unleash': 'release, enable, unlock',
    'streamline': 'simplify, speed up',
    'empower': 'enable, let, allow',
    'bolster': 'support, strengthen',
    'spearhead': 'lead, drive, run',
    'resonate': 'connect with, appeal to',
    'resonates': 'connects with, appeals to',
    'revolutionize': 'change, transform',
    'facilitate': 'enable, help, allow',
    'facilitates': 'enables, helps, allows',
    'underpin': 'support, form the basis of',
    'nuanced': 'specific, subtle, detailed',
    'crucial': 'important, key, necessary',
    'multifaceted': 'describe the actual facets',
    'ecosystem': 'system, community, network',
    'myriad': 'many, numerous',
    'plethora': 'many, a lot of',
    'encompass': 'include, cover, span',
    'catalyze': 'start, trigger, accelerate',
    'reimagine': 'rethink, redesign, rebuild',
    'galvanize': 'motivate, rally, push',
    'augment': 'add to, expand, supplement',
    'cultivate': 'build, develop, grow',
    'illuminate': 'clarify, explain, show',
    'elucidate': 'explain, clarify',
    'juxtapose': 'compare, contrast',
    'transformative': 'describe what changed',
    'transformation': 'describe what changed',
    'cornerstone': 'foundation, basis, key part',
    'paramount': 'most important, top priority',
    'poised': 'ready, set, about to',
    'burgeoning': 'growing, emerging',
    'nascent': 'new, early-stage',
    'quintessential': 'typical, classic, defining',
    'overarching': 'main, central, broad',
    'quietly': 'cut, or name the concrete contrast',
    'underpinning': 'basis, foundation',
    'underpinnings': 'basis, foundations',
    'paradigm-shifting': 'describe what shifted',
  };

  // Conditional Tier 2 entries: everyday words whose AI tell is a specific
  // significance collocation, not the word itself. They join a paragraph's
  // Tier 2 cluster only when the collocation matches — bare uses ("deeply
  // nested JSON", "cares deeply") never count, because the base rate of
  // these words in innocent prose is far higher than the rest of the table
  // and an unconditional entry measurably flags clean human writing.
  const TIER2_CONDITIONAL = [
    {
      word: 'deeply',
      pattern: /\bdeeply\s+(?:integrated|committed|rooted|personal|human|flawed|resonant|transformative|interconnected|ingrained|embedded|meaningful)\b/i,
      suggestion: 'cut, or name what specifically runs deep',
    },
  ];

  // ─── Tier 3: Flag by density ───────────────────────────────────────
  const TIER3 = [
    'significant', 'significantly', 'innovative', 'innovation',
    'effective', 'effectively', 'dynamic', 'dynamics',
    'scalable', 'scalability', 'compelling', 'unprecedented',
    'exceptional', 'exceptionally', 'remarkable', 'remarkably',
    'sophisticated', 'instrumental',
    'world-class', 'state-of-the-art', 'best-in-class',
    // `verbatim` is usually redundant with the verb it modifies ("copies X
    // verbatim" = "copies X"). It has a genuine term-of-art sense in legal,
    // research, and QA registers ("verbatim transcript"), so it lives at Tier 3:
    // density-gated, it only fires on overuse, not on a single legitimate use.
    'verbatim',
  ];

  // Multi-word Tier 3 phrases. Density-gated like single Tier 3 words because
  // these legitimately show up in human crypto/web3/dev writing. Threshold is
  // intentionally lower than single-word Tier 3 (≥2 occurrences of the same
  // phrase) — repeating the same multi-word boilerplate is a stronger AI tell
  // than re-using "significant."
  const TIER3_PHRASES = [
    /\bemerging\s+(?:sector|space|category|industry)\b/gi,
    /\bthe\s+integration\s+of\b/gi,
    /\bthe\s+intersection\s+of\b/gi,
    /\bcommunity-?driven\b/gi,
    /\blong-?term\s+sustainability\b/gi,
    /\buser\s+engagement\b/gi,
    /\bdecentralized\s+compute\b/gi,
    /\b(?:sustainable\s+)?reward\s+emissions?\b/gi,
    /\btokenized\s+incentive\s+structures?\b/gi,
    /\bdesigned\s+for\s+long-?term\b/gi,
  ];

  // O(1) lookup from any token form (hyphenated or dashless) to its canonical
  // Tier 3 word. Counting originally did nested-loop word matching which was
  // O(tokens × TIER3) — slow on long pastes.
  const TIER3_LOOKUP = new Map();
  for (const word of TIER3) {
    TIER3_LOOKUP.set(word, word);
    const dashless = word.replace(/-/g, '');
    if (dashless !== word) TIER3_LOOKUP.set(dashless, word);
  }

  // Per-category score weights. Applied to distinct (deduplicated) issues so
  // the score reflects the same signals the user sees in the issue list.
  // Non-uniform on purpose: critical rules like cutoff disclaimers (×10) and
  // chatbot artifacts (×8) weigh more than vague attributions (×5), even
  // though all three are tagged `critical`.
  const ISSUE_WEIGHTS = {
    tier1: 5,
    // Wordiness, not an AI-frequency marker. Weighted like tier2 so a
    // clarity fix cannot push a document toward an AI classification.
    'tier1-clarity': 3,
    tier2: 3,
    tier3: 2,
    transition: 2,
    chatbot: 8,
    sycophantic: 8,
    filler: 2,
    'generic-conclusion': 3,
    'lets-construction': 2,
    'reasoning-artifact': 6,
    'acknowledgment-loop': 3,
    'significance-inflation': 4,
    'vague-attribution': 5,
    'hollow-intensifier': 2,
    'emotional-flatline': 2,
    'lingering-attention': 3,
    'novelty-inflation': 3,
    'cutoff-disclaimer': 10,
    'template-phrase': 3,
    'false-concession': 2,
    'rhetorical-question': 2,
    'confidence-calibration': 2,
    'em-dash': 4,
    uniformity: 5,
    formatting: 3,
    'tier3-phrase': 3,
    // Structural / cluster signals are deliberately weighted high. Unlike
    // single-vocabulary hits they're near-dispositive on social-length
    // posts (a 15-hashtag block, a 6-item bullet-NP list, three distinct
    // crypto-shill phrases stacked) and would otherwise be suppressed by
    // the log2(words/50) length divisor on short pastes.
    'tier3-phrase-cluster': 12,
    'hashtag-stuff': 12,
    'bullet-np-list': 10,
    'hedge-stack': 6,
    'future-narrative': 12,
    'real-actual-inflation': 5,
    // Social endorsement / CTA closer. Weighted like formulaic-opener: a
    // strong single-hit social tell that the length divisor would
    // otherwise wash out on a short LinkedIn-length post.
    'social-cta-closer': 8,
    // Performed-insight tics: single hits are common in human essays, so
    // weighted like tier2 vocabulary — density does the classifying.
    'performed-insight': 3,
    // Negation chains are a strong single-hit structural tell.
    'negation-chain': 5,
    'dev-blog-boilerplate': 3,
    'formulaic-opener': 8,
    // Speculative scenario opener ("Imagine a world where…"). Weighted like
    // formulaic-opener: a single strong opener tell the length divisor would
    // otherwise wash out on a short post.
    'speculative-opener': 8,
    // Launch-copy introduction ("Enter X.", "Meet X, your new...").
    // Weighted like the other single-hit opener tells: strong on the
    // short launch posts where it actually appears.
    'launch-intro': 8,
    // Dramatized crowd contrast. Gated hard on the dismissive verb, so
    // a hit is meaningful, but the surface shares words with ordinary
    // narrative — weighted below the opener tells on purpose.
    'crowd-contrast': 6,
    // Fake-casual props (stage directions, wink asides). Near-costume
    // when present; same class as the opener tells on short posts.
    'fake-casual-prop': 8,
    'title-case-header': 4,
    'parenthetical-hedge': 3,
    'smart-punct-signature': 6,
    'punct-distribution': 6,
    'fnword-trigram-entropy': 5,
    'cross-para-burstiness': 5,
    'normalization-flag': 9,
    // Vocabulary-diversity signal (type-token ratio). Weighted modestly
    // because the threshold (>=200 tokens AND TTR<0.4) is conservative;
    // it stacks with structural signals to push borderline scores up.
    'low-ttr': 3,
    // AI-tool fingerprints. Weighted higher than statistical patterns
    // because each is a near-definitive single-hit signal — the AI tool
    // literally left its mark in the text. citation-markup ranks highest
    // (smoking gun: the literal internal markup of ChatGPT/Grok/etc.),
    // UTM tracking second (auto-appended by the tool to URLs it writes),
    // placeholders third (strong but humans use bracketed slots in
    // templates legitimately and forget them — still a publishing bug
    // but slightly less definitive AI evidence).
    'ai-placeholder': 10,
    'ai-citation-markup': 15,
    'ai-utm-source': 12,
    // Curated P2 copyedits, not authorship evidence. Keep them visible in the
    // issue list without moving the AI score, label, probabilities, or
    // trinary classification on short documents.
    'unnecessary-hyphenation': 0,
  };

  // ─── Transition phrases ────────────────────────────────────────────
  const TRANSITIONS = [
    /\bmoreover\b/gi,
    /\bfurthermore\b/gi,
    /\badditionally\b/gi,
    /\bin\s+today'?s\b/gi,
    /\bin\s+an\s+era\s+where\b/gi,
    /\bit'?s\s+worth\s+noting\s+that\b/gi,
    /\bnotably\b/gi,
    /\bin\s+conclusion\b/gi,
    /\bin\s+summary\b/gi,
    /\bto\s+summarize\b/gi,
    /\bwhen\s+it\s+comes\s+to\b/gi,
    /\bat\s+the\s+end\s+of\s+the\s+day\b/gi,
    /\bthat\s+(?:being\s+)?said\b/gi,
  ];

  // ─── Chatbot artifacts ─────────────────────────────────────────────
  const CHATBOT_ARTIFACTS = [
    /\bi\s+hope\s+this\s+helps\b/gi,
    /\bcertainly!\b/gi,
    /\babsolutely!\b/gi,
    /\bgreat\s+question!\b/gi,
    /\bexcellent\s+point!\b/gi,
    /\bfeel\s+free\s+to\s+reach\s+out\b/gi,
    /\blet\s+me\s+know\s+if\s+you\s+need\s+anything\b/gi,
    /\bin\s+this\s+article,?\s+we\s+will\s+explore\b/gi,
    /\blet'?s\s+dive\s+in!?\b/gi,
  ];

  // ─── Sycophantic tone ──────────────────────────────────────────────
  const SYCOPHANTIC = [
    /\byou'?re\s+absolutely\s+right\b/gi,
    /\bthat'?s\s+a\s+really\s+insightful\b/gi,
    /\bthat'?s\s+a\s+great\s+question\b/gi,
    /\bexcellent\s+question\b/gi,
  ];

  // ─── Filler phrases ────────────────────────────────────────────────
  const FILLERS = [
    /\bit\s+is\s+important\s+to\s+note\s+that\b/gi,
    /\bin\s+terms\s+of\b/gi,
    /\bthe\s+reality\s+is\s+that\b/gi,
    /\bit'?s\s+important\s+to\s+note\s+that\b/gi,
  ];

  // ─── Generic conclusions ───────────────────────────────────────────
  const GENERIC_CONCLUSIONS = [
    /\bthe\s+future\s+looks\s+bright\b/gi,
    /\bonly\s+time\s+will\s+tell\b/gi,
    /\bone\s+thing\s+is\s+certain\b/gi,
    /\bas\s+we\s+move\s+forward\b/gi,
  ];

  // ─── "Let's" constructions ─────────────────────────────────────────
  const LETS_PATTERNS = [
    /\blet'?s\s+explore\b/gi,
    /\blet'?s\s+take\s+a\s+look\b/gi,
    /\blet'?s\s+break\s+this\s+down\b/gi,
    /\blet'?s\s+examine\b/gi,
    /\blet'?s\s+(?:consider|discuss|delve|unpack|walk\s+through)\b/gi,
  ];

  // ─── Reasoning chain artifacts ─────────────────────────────────────
  const REASONING_ARTIFACTS = [
    /\blet\s+me\s+think\s+step\s+by\s+step\b/gi,
    /\bbreaking\s+this\s+down\b/gi,
    /\bto\s+approach\s+this\s+systematically\b/gi,
    /\bhere'?s\s+my\s+thought\s+process\b/gi,
    /\bfirst,?\s+let'?s\s+consider\b/gi,
    /\bworking\s+through\s+this\s+logically\b/gi,
  ];

  // ─── Acknowledgment loops ──────────────────────────────────────────
  const ACKNOWLEDGMENT_LOOPS = [
    /\byou'?re\s+asking\s+about\b/gi,
    /\bthe\s+question\s+of\s+whether\b/gi,
    /\bto\s+answer\s+your\s+question\b/gi,
  ];

  // ─── Significance inflation ────────────────────────────────────────
  const SIGNIFICANCE_INFLATION = [
    /\bmarking\s+a\s+(?:pivotal|significant|important)\s+moment\b/gi,
    /\ba\s+watershed\s+moment\s+for\b/gi,
    /\bin\s+the\s+evolution\s+of\b/gi,
    /\ba\s+(?:pivotal|defining)\s+moment\s+in\b/gi,
  ];

  // ─── Vague attributions ────────────────────────────────────────────
  const VAGUE_ATTRIBUTIONS = [
    /\bexperts\s+(?:believe|say|suggest|agree)\b/gi,
    /\bstudies\s+(?:show|suggest|indicate)\b/gi,
    /\bresearch\s+(?:shows|suggests|indicates)\b/gi,
    /\bindustry\s+leaders\s+(?:agree|believe|say)\b/gi,
  ];

  // ─── Hollow intensifiers ──────────────────────────────────────────
  const HOLLOW_INTENSIFIERS = [
    /\bgenuine(?:ly)?\b/gi,
    /\btruly\b/gi,
    /\bquite\s+frankly\b/gi,
    /\bto\s+be\s+honest\b/gi,
    /\blet'?s\s+be\s+clear\b/gi,
  ];

  // ─── Emotional flatline ────────────────────────────────────────────
  // The "interesting (part|thing|aspect|piece)" family is matched in two
  // shapes: (1) "the most interesting X" inline (the canonical AI list
  // intro), and (2) bare "Interesting X:" used as a section-header opener,
  // which is the section-break variant that slipped past v3.3.x.
  const EMOTIONAL_FLATLINE = [
    /\bwhat\s+surprised\s+me\s+most\b/gi,
    /\bi\s+was\s+fascinated\s+to\b/gi,
    /\bwhat\s+struck\s+me\s+was\b/gi,
    /\bi\s+was\s+excited\s+to\s+learn\b/gi,
    /\bthe\s+most\s+interesting\s+(?:part|thing|aspect|piece)\b/gi,
    // Multiline flag (/m) so `^` matches at every line start, including
    // position 0 of a pasted text that has no leading newline. The earlier
    // `(?:^|\n)` form silently missed bare openers at the very start of
    // input — caught by silent-failure audit 2026-05-16.
    /^\s*interesting\s+(?:part|thing|aspect|piece)(?:\s+of\s+(?:the\s+)?\w+)?\s*:/gim,
  ];

  // ─── Lingering-attention claims ────────────────────────────────────
  // The share-post opener that claims duration of attention instead of
  // saying anything about the thing ("the line I keep coming back to").
  //
  // Precision note: the bare verb phrase "I keep coming back to X" is NOT
  // matched on its own, because it is legitimate whenever a reason follows
  // ("I keep coming back to Hirschman because it predicts who quits"), and
  // the reason clause is not reliably detectable by regex. Only the
  // noun-anchored frame ("the line/quote/bit ... I keep coming back to")
  // fires, which is the shape that introduces a subject rather than
  // asserting something about it. The bare form stays a skill-prose
  // judgment call. See SKILL.md §Lingering-attention claims carve-out.
  const LINGERING_ATTENTION = [
    // "the line I keep coming back to", "the one quote that I keep coming back to"
    // Noun-anchored frame only. "can't stop thinking about" is deliberately
    // left to its own pattern below rather than folded into this alternation,
    // so "the line I can't stop thinking about" scores once, not twice.
    /\b(?:the|that|this)\s+(?:one\s+)?(?:line|quote|bit|part|idea|point|framing|comment|thing)\s+(?:that\s+)?i\s+keep\s+(?:coming\s+back\s+to|thinking\s+about)\b/gi,
    /\bi\s+can'?t\s+stop\s+thinking\s+about\b/gi,
    /\bstill\s+thinking\s+about\s+(?:this|that)\s+one\b/gi,
    /\b(?:been|be)\s+rattling\s+around\s+(?:in\s+)?my\s+(?:head|brain)\b/gi,
    /\bi'?ve\s+been\s+chewing\s+on\s+(?:this|that)\b/gi,
  ];

  // ─── Novelty inflation ─────────────────────────────────────────────
  const NOVELTY_INFLATION = [
    /\bthe\s+failure\s+mode\s+nobody'?s?\s+naming\b/gi,
    /\ba\s+problem\s+nobody\s+talks\s+about\b/gi,
    /\bthe\s+insight\s+everyone'?s?\s+missing\b/gi,
    /\bwhat\s+nobody\s+tells\s+you\b/gi,
  ];

  // ─── Cutoff disclaimers ────────────────────────────────────────────
  // Includes the canonical LLM self-identification phrases — these are
  // near-dispositive on their own (real humans don't write "as an AI
  // language model" in first person). Patterns cover the major model
  // families' default disclaimer language.
  const CUTOFF_DISCLAIMERS = [
    /\bas\s+of\s+my\s+last\s+update\b/gi,
    /\bas\s+of\s+my\s+(?:knowledge\s+)?(?:cut-?off|last\s+training)\b/gi,
    /\bi\s+don'?t\s+have\s+access\s+to\s+real-?time\s+(?:data|information)\b/gi,
    /\bbased\s+on\s+available\s+information\b/gi,
    /\bas\s+an?\s+(?:ai|artificial\s+intelligence|large\s+language|ai\s+language)\s+(?:language\s+)?model\b/gi,
    /\bi\s+(?:am|'m)\s+an?\s+(?:ai|artificial\s+intelligence|large\s+language)\s+(?:assistant|model)?\b/gi,
    /\bi\s+cannot\s+(?:provide|give|offer)\s+(?:legal|medical|financial|professional)\s+advice\b/gi,
    /\bmy\s+training\s+data\s+(?:only\s+)?(?:goes\s+up\s+to|extends\s+to|ends\s+(?:in|at))\b/gi,
  ];

  // ─── AI-tool fingerprints ──────────────────────────────────────────
  // Three near-definitive AI-origin signals adapted from
  // Aboudjem/humanizer-skill P33-P35 (see docs/competitive/audits/
  // 2026-05-17-aboudjem-humanizer-skill.md). Unlike the statistical
  // patterns above, single hit on any of these is strong evidence —
  // the AI tool literally left its fingerprint in the text.

  // Unfilled slot-fill placeholders. Catches the canonical "[Your Name]"
  // family plus dated stubs and HTML/MD comments with placeholder verbs.
  const AI_PLACEHOLDERS = [
    // Directive stubs ("[Your Name]", "[INSERT SOURCE URL]",
    // "[Describe the specific section]") — verb-led, the user was
    // told to fill in.
    /\[(?:Your|Insert|Add|Enter|Describe|Specify|Choose|Pick)[^\]\n]{1,80}\]/gi,
    // Noun-only template variables common in AI-generated email or
    // letter boilerplate. Match the bare noun OR noun + qualifier.
    // Conservative list: only nouns that almost never appear as
    // bracketed real content (citation refs, code identifiers, etc.
    // are excluded because they typically contain dots, slashes,
    // hyphens, or version numbers).
    /\[(?:Recipient|Sender|Topic|Subject|Salutation|Closing|Position|Department|Project Name|Company Name|Date)(?:\s+[^\]\n]{0,60})?\]/gi,
    // All-caps directive forms ("[INSERT X]", "[FILL IN]") — the
    // uppercase tells you it's a slot, not real content.
    /\[(?:INSERT|FILL\s+IN|ADD|TODO|TBD|PLACEHOLDER)[^\]\n]{0,80}\]/g,
    // Date stubs.
    /\b(?:19|20)\d{2}-XX-XX\b/g,
    /\bXX\/XX\/(?:19|20)\d{2}\b/g,
    // HTML/Markdown comment placeholders with placeholder verbs.
    /<!--\s*(?:add|fill\s+in|insert|todo|placeholder)[^>]{0,120}-->/gi,
  ];

  // Chatbot citation/markup tokens that leak through copy-paste.
  // `citeturn0search0` / `citeturn0news5` from ChatGPT, contentReference
  // tokens, oai_citation, attached_file references, grok_card markers.
  // Each is a near-definitive signature of a specific tool.
  const AI_CITATION_MARKUP = [
    /\bcite(?:turn|news|search|navigation)\d+(?:search|turn|news|navigation)\d+/gi,
    /contentReference\s*\[oaicite:[^\]]+\]\s*\{[^}]*\}/gi,
    /\boai_citation\b/gi,
    /\[attached_file:\d+\]/gi,
    /\bgrok_card\b/gi,
  ];

  // UTM/tracking parameters auto-appended by AI tools to URLs they
  // generate. Survives copy-paste even when nothing else does.
  const AI_UTM_SOURCE = [
    /[?&]utm_source=(?:chatgpt|openai|copilot|claude|grok|gemini|perplexity)(?:\.com|\.ai)?\b/gi,
    /[?&]referrer=(?:chatgpt|copilot|grok|claude|gemini|perplexity)\.(?:com|ai)\b/gi,
  ];

  // ─── Template phrases ──────────────────────────────────────────────
  const TEMPLATE_PHRASES = [
    /\ba\s+\w+\s+step\s+(?:towards?|forward\s+for)\b/gi,
    /\bwhether\s+you'?re\s+\w+\s+or\s+\w+/gi,
    /\bi\s+recently\s+had\s+the\s+pleasure\s+of\b/gi,
  ];

  // ─── False concession ──────────────────────────────────────────────
  const FALSE_CONCESSION = [
    /\bwhile\s+\w+\s+is\s+impressive\b/gi,
    /\balthough\s+\w+\s+has\s+made\s+strides\b/gi,
    /\bdespite\s+\w+\s+challenges?\b/gi,
  ];

  // ─── Rhetorical question openers ───────────────────────────────────
  const RHETORICAL_QUESTIONS = [
    /\bbut\s+what\s+does\s+this\s+mean\s+for\b/gi,
    /\bso\s+why\s+should\s+you\s+care\b/gi,
    /\bwhat'?s\s+next\?\s*/gi,
  ];

  // ─── Hedge-stacked predictions ─────────────────────────────────────
  // Stacks a modal with a hedge adverb: "could potentially create new
  // opportunities", "may eventually unlock value." Either word alone is
  // fine; the stack is the tell.
  const HEDGE_STACK = [
    // At most one intervening word, and never a negator. The old {0,2} gap
    // matched ordinary English: "could not possibly" (plain emphatic
    // negation) and inverted questions like "could a savage possibly" both
    // fired. Measured on the human-control corpus, 3 of 4 hedge-stack flags
    // were this over-match. See issue #69.
    /\b(?:could|may|might)\s+(?:(?!not\b|never\b|hardly\b|scarcely\b|barely\b)\w+\s+)?(?:potentially|eventually|ultimately|possibly|conceivably)\b/gi,
    /\b(?:potentially|eventually|ultimately)\s+(?:could|may|might)\b/gi,
  ];

  // ─── Generic future-narrative closers ──────────────────────────────
  // The "may become one of the most important narratives" template — vague
  // future significance with no falsifiable claim. Covers narratives /
  // stories / trends / themes / chapters / movements.
  const FUTURE_NARRATIVE = [
    /\b(?:may|could|will|is\s+(?:poised|set)\s+to)\s+become\s+(?:one\s+of\s+)?(?:the\s+)?(?:most\s+)?\w+\s+(?:narratives?|stories|developments?|trends?|movements?|chapters?|themes?|forces?)\b/gi,
    /\bone\s+of\s+the\s+most\s+important\s+(?:narratives?|stories|trends?|themes?)\s+of\s+the\s+(?:next|coming)\s+\w+\b/gi,
  ];

  // ─── "Real/actual" adjective inflation ─────────────────────────────
  // "Real on-chain tokenomics", "actual reward sustainability" — using
  // real/actual/genuine/true as an empty intensifier on an abstract noun
  // to imply the rest of the field is fake/superficial.
  const REAL_ACTUAL_INFLATION = [
    /\b(?:real|actual|genuine|true)\s+(?:on-?chain\s+)?(?:tokenomics|economics|utility|adoption|sustainability|impact|revenue|fundamentals|demand|value|innovation|traction)\b/gi,
  ];

  // ─── Formulaic openers ─────────────────────────────────────────────
  // The "In the rapidly evolving world of X, Y has emerged as..." family
  // — LLM-default essay openers.
  const FORMULAIC_OPENERS = [
    /\bin\s+the\s+(?:rapidly\s+|ever-?\s*)?(?:evolving|changing|expanding|growing|shifting)\s+(?:world|landscape|realm|space|field|domain|era)\s+of\b/gi,
    /\bin\s+(?:an?|the)\s+(?:digital\s+)?age\s+(?:where|of)\b/gi,
    /\bas\s+(?:we|the\s+world|society|industries?)\s+(?:continue|move|navigate|enter)\s+(?:to\s+)?(?:evolve|forward|into|through)\b/gi,
    // "has emerged as a leader/force/category" — gated to the inflated
    // nouns that signal pseudo-significance, since bare "has emerged as
    // a" matches normal English ("Rust has emerged as a serious systems
    // language"). Same gating for "has become increasingly".
    /\bhas\s+emerged\s+as\s+(?:a|the|one\s+of)\s+(?:leading|key|major|critical|essential|fundamental|pivotal|prominent|dominant|important)\s+\w+/gi,
    /\bhas\s+become\s+increasingly\s+(?:important|critical|popular|relevant|prominent|essential)\b/gi,
  ];

  // ─── Speculative scenario openers ──────────────────────────────────
  // "Imagine a world where...", "Picture a future in which...", "Envision
  // a world where..." — the LLM habit of opening an argument with a
  // hypothetical that lists desirable outcomes instead of making a claim.
  // Gated to the world/future/reality object plus where/in-which so it
  // stays off instructional "imagine you have an array" (a teaching device
  // pointing at a concrete example, not a speculative world) and bare
  // "imagine that" asides. "consider a scenario where…" is deliberately
  // excluded: that is analytical framing common in technical reasoning,
  // not the marketing-opener tell. An optional short comma interrupter
  // catches the equally common "Imagine, for a moment, a world where…"
  // cadence. Known accepted false positive: fiction openings and staged
  // thought experiments match too — the engine has no fiction context
  // mode, so that call is left to the skill's carve-out (highlight-only;
  // a lone hit cannot flip a document's classification).
  const SPECULATIVE_OPENERS = [
    /\b(?:imagine|picture|envision)(?:\s*,[^,\n]{1,30},)?\s+a\s+(?:world|future|reality)\s+(?:where|in\s+which)\b/gi,
  ];

  // ─── Launch-copy dramatic introductions ────────────────────────────
  // "Meet Flowdesk, your new favorite treasury dashboard" / "Think
  // Notion meets Figma" — the LLM-default product-introduction move
  // in launch and announcement copy. Both surfaces are gated to the
  // sentence-initial imperative followed by ONE capitalized token of 2
  // to 30 characters, which is a recall limit: a two-token product name
  // ("Meet North Star", "Think Google Docs meets Microsoft Word") is a
  // deliberate miss. The Meet surface additionally requires one of four
  // launch-copy heads — "your new favorite", "your new go-to", and
  // "the new home/way/standard", the last three only when followed by
  // "of" / "to" / "in|for" or by end-of-clause punctuation. Without
  // that tail the head noun swallows a compound noun and ordinary prose
  // fires: "Meet Rosa, the new home secretary" and "Meet Emma, the new
  // way station manager" both matched before the tail was required.
  // Bare "Meet Sarah, your new account manager" is how humans introduce
  // colleagues, pets, and babies, so that form stays with the skill's
  // judgment side. Two surfaces from
  // the same family are deliberately NOT detected. "Say hello to X",
  // because "Say hello to Grandma." is ordinary human prose. And bare
  // "Enter X.", because the sentence-initial capitalized-noun form is
  // how UI and doc instructions are written: "Enter Password.", "Enter
  // Amount.", "Enter Username — your work email." Dropping the dash
  // terminator does not reach the period-terminated class, and neither
  // does a field-name denylist, so that surface stays with the skill's
  // judgment side and the UI forms are pinned as must-not-fire
  // fixtures. The anchors are lookbehinds so adjacent intros each
  // count and the reported span starts at the tell itself.
  const LAUNCH_INTROS = [
    /(?<=^|[.!?]\s|\n)Meet\s+[A-Z][\w'-]{1,29}\s*,\s*(?:your\s+new\s+(?:favorite|go-to)\b|the\s+new\s+(?:home\s+of\b|way\s+to\b|standard\s+(?:in|for)\b|(?:standard|way|home)(?=\s*(?:[.!?,;:\u2013\u2014]|$))))/g,
    /(?<=^|[.!?]\s|\n)[Tt]hink\s+[A-Z][\w'-]{1,29}\s+meets\s+[A-Z][\w'-]{1,29}\b/g,
  ];

  // ─── Dramatized contrast against the crowd ─────────────────────────
  // "shipped it in 2022, while everyone else was still debating
  // timelines" — a claim propped on an implied lagging crowd. The gate
  // is a dismissive verb PLUS the "was still" dramatization marker,
  // because bare "while everyone else" is ordinary simultaneity ("she
  // read while everyone else watched the movie") and even the
  // dismissive verbs are ordinary English in literal use ("others
  // debated the amendment" in wire copy). That gate is on this FIRST
  // branch only. Its stems are restricted to the -ing form, so the
  // adjective ("was still deliberate about the tradeoff"), the passive
  // ("was still debated by pundits") and the bare present ("was still
  // debates timelines") all stay clean — allowing e/es/ed let all
  // three through. The other two branches carry no "was still"
  // requirement: they match their own stereotyped wording ("writing
  // think-pieces", "playing catch-up"), and the skill entry scopes the
  // claim the same way. Verb stems carry explicit inflection tails so
  // agent nouns ("the market speculators") and adverbs ("deliberately
  // ignored") never match. Measured residue, all accepted: branch one
  // fires on ANY literal progressive use of its verbs ("while the
  // market was still speculating about the price"), not only on "was
  // still debating"; branches two and three fire on literal contrasts
  // of their own ("while everyone else wrote think-pieces from
  // Washington", "while everyone else played catch-up in the spring").
  // Recall is deliberately sacrificed: "was busy debating" without
  // "still" stays a miss, per precision-over-recall.
  const CROWD_CONTRAST = [
    /\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was|were|is|are)\s+still\s+(?:busy\s+)?(?:(?:debat|deliberat|hesitat|theoriz|philosophiz|pontificat|speculat|argu)ing|(?:dither|bicker)ing)\b/gi,
    /\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was\s+|were\s+)?(?:busy\s+)?(?:writing|wrote)\s+think-?\s?pieces\b/gi,
    /\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was\s+|were\s+|is\s+|are\s+)?(?:still\s+)?play(?:ed|ing|s)?\s+catch[-\s]?up\b/gi,
  ];

  // ─── Fake-casual props (stage directions and wink asides) ──────────
  // The regexable props from the fake-casual register: theatrical
  // asterisk stage directions ("*checks notes*", "*chef's kiss*",
  // "*mic drop*") and wink asides. Both lists are closed and short, and
  // that is a recall limit: exactly six stage directions ("checks
  // notes", "chef's kiss", "mic drop", "takes a deep breath", "sips
  // coffee|tea", "nervous laughter") and exactly four parentheticals,
  // the full (yes|no) x (really|seriously) grid. Neighbours in the same
  // register are deliberate misses: "*checks calendar*" and "(yes,
  // honestly)" do not fire. The kiss pattern requires the apostrophe —
  // making it optional matched the ordinary sentence "At midnight,
  // *chefs kiss* their spouses goodbye".
  // The rest of the register (one-word verdict closers, label-prefix
  // openers, the self-QA volley) needs register judgment and stays
  // skill-only — "wild." is a word, not a regex target. "because of
  // course …" joins them: a tense gate does not separate the wink from
  // the ordinary human grumble, because "The build failed because of
  // course it did." is that grumble in the same present-tense-plus-did
  // form the wink uses. Under precision-over-recall the surface is
  // judgment-only, and the grumble is pinned as a fixture. The kiss
  // pattern accepts the curly apostrophe (U+2019) — the form smart
  // punctuation and LLMs actually emit. Known accepted false positive:
  // a human writer using a wink aside on purpose; the props are
  // weighted as a strong single hit, not a classification by
  // themselves.
  const FAKE_CASUAL_PROPS = [
    /\*\s?(?:checks\s+notes|chef['\u2019]s\s+kiss|mic\s+drop|takes\s+a\s+deep\s+breath|sips\s+(?:coffee|tea)|nervous\s+laughter)\s?\*/gi,
    /\(\s?(?:yes|no)\s?,\s?(?:really|seriously)\s?\)/gi,
  ];

  // ─── Performed-insight phrases ─────────────────────────────────────
  // Essayist tics that announce profundity instead of delivering it.
  // Curated noun/complement lists keep precision high: "the whole family"
  // and "sit with him" are ordinary English and must not fire. Adapted
  // from Simon Willison's LLM cliché highlighter
  // (tools.simonwillison.net/llm-cliche-highlighter).
  const PERFORMED_INSIGHT = [
    /\bsit(?:s|ting)?\s+with\s+(?:that|this)(?=\s*(?:[.!?,;:)\u2013\u2014\u2019"']|for\s+a\s+(?:moment|minute|second|beat)\b|$))(?:\s+for\s+a\s+(?:moment|minute|second|beat))?/gi,
    /\bsit(?:s|ting)?\s+with\s+(?:the|your)\s+(?:discomfort|tension|uncertainty|ambiguity|grief|unease)\b/gi,
    /\b(?:that|this|it|which)(?:['\u2019]s|\s+(?:is|was))\s+not\s+nothing\b/gi,
    /\byou\s+already\s+know\s+the\s+answer\b/gi,
    /\b(?:do\s+not|don['\u2019]t)\s+(?:have\s+to\s+)?take\s+my\s+word\s+for\s+it\b/gi,
    /(?<=^|[.!?]\s|\n)Turns\s+out\b/g,
    /(?:['\u2019]s|\b(?:is|was|are|were))\s+the\s+(?:whole|entire)\s+(?:point|game|ballgame|trick|pitch|idea|play|business\s+model|value\s+proposition)\b/gi,
    /\b(?:that|this)(?:['\u2019]s|\s+(?:is|was))\s+the\s+part\s+(?:that|I|you|we|nobody|no\s+one|most\s+people)\b/gi,
    /\bthe\s+only\s+[\w'\u2019-]+\s+that\s+(?:matters|counts)\b/gi,
    /\bis\s+dead\s*[.;,:\u2013\u2014]\s*long\s+live\b/gi,
    /\b(?:that|this)(?:['\u2019]s|\s+(?:is|was))\s+why\s+[^.!?\n]{0,60}\s+mattered\b/gi,
  ];

  // ─── Negation chains ───────────────────────────────────────────────
  // "No fluff, no filler, no jargon" / "It didn't ask, didn't wait" /
  // "Don't call it X. Call it Y." Precision guards, in order: the
  // "no …" chain must open its sentence (mid-sentence inventories like
  // "takes no arguments, no headers, and no body" are factual, not
  // rhetorical); the "did not" chain must be comma-joined with the
  // subject elided ("I did not sleep. I did not eat" is ordinary
  // narration and stays clean); the stop-list keeps idiomatic pairs
  // ("no more, no less", "no matter what") from firing. Adapted from
  // Simon Willison's LLM cliché highlighter.
  const NO_ITEM_STOP = "(?!matter\\b|one\\b|doubt\\b|longer\\b|way\\b|less\\b|more\\b|such\\b|other\\b|means\\b)";
  const NO_ITEM_SECOND_STOP = "(?!(?:in|on|at|of|to|for|with|from|by|is|are|was|were|be|been|being|will|would|can|could|should|shall|may|might|must|have|has|had|do|does|did)\\b)";
  const NEGATION_CHAIN = [
    new RegExp(
      "(?<=^|[.!?]\\s|\\n|[:\\u2013\\u2014]\\s)No\\s+" + NO_ITEM_STOP + "[a-z'\u2019-]+(?:\\s+" + NO_ITEM_SECOND_STOP + "[a-z'\u2019-]+)?" +
      "(?:\\s*,\\s*(?:and\\s+|or\\s+|just\\s+)?no\\s+" + NO_ITEM_STOP + "[a-z'\u2019-]+(?:\\s+" + NO_ITEM_SECOND_STOP + "[a-z'\u2019-]+)?){2,}",
      'gm'
    ),
    /\b(?:did\s+not|didn['\u2019]t)\s+[a-z]+[^,.;!?\n]{0,20},\s*(?:did\s+not|didn['\u2019]t)\s+[a-z]+/gi,
    /\b(?:do\s+not|don['\u2019]t)\s+(?:just\s+)?(\w+)\s+it\b[^.!?\n]{0,60}[.!?;:,][\s'"\u201d\u2019]*(?:just\s+)?\1\s+it\b/gi,
  ];

  // ─── Dev-blog boilerplate ──────────────────────────────────────────
  // Stock simplicity slogans from developer marketing. Adapted from
  // Simon Willison's LLM cliché highlighter.
  const DEV_BLOG_BOILERPLATE = [
    /\bit\s+just\s+works\b(?!\s+out\b(?![-\s]+of[-\s]+the[-\s]+box\b))/gi,
    /\bzero[-\s]config(?:uration)?\b/gi,
    /\bsane\s+defaults\b/gi,
    /\b(?:hold|fit|fits|holds)\s+in\s+your\s+head\b/gi,
  ];

  // Function words whose presence MID-title marks the AI section-header shape.
  // Word-anchored: without \b the "A" alternative matches inside any word and
  // the guard silently degrades to "four tokens".
  const FUNCTION_WORD = /\b(?:And|Or|Of|The|In|For|To|A|An)\b/;

  // Must accept exactly what TITLE_CASE_HEADER accepts, or the prefix survives
  // into the token count and reintroduces the ##-as-token bug.
  const MD_HEADING_PREFIX = /^#{1,6}[ \t]+/;

  /** Byte ranges covered by fenced code blocks, computed once per scan.
   *
   * A document that documents Markdown is the normal case for this rule -- a
   * fenced `## Heading` example is illustration, not the author's own section
   * header, and flagging it makes every docs page flag itself.
   *
   * This tracks the opening delimiter instead of counting them, because a
   * parity count is wrong on the very case the rule exists for. CommonMark
   * closes a fence only on the same character at the same length or longer, so
   * a four-backtick fence wrapping a three-backtick example -- exactly how you
   * document fences -- nests in practice, and counting delimiters inverts on
   * it. Up to three spaces of indent are legal. An unclosed fence runs to end
   * of document, matching how renderers treat it.
   *
   * Computed once per scan rather than rescanned per hit: the previous version
   * sliced the whole document for every candidate, which is quadratic on a
   * heading-dense file. */
  function fenceRanges(text) {
    const re = /^[ \t]{0,3}(`{3,}|~{3,})([^\n]*)$/gm;
    const ranges = [];
    let open = null;
    let m;
    while ((m = re.exec(text)) !== null) {
      const marker = m[1];
      if (!open) {
        open = { char: marker[0], len: marker.length, start: m.index };
      } else if (
        marker[0] === open.char &&
        marker.length >= open.len &&
        /^[ \t]*\r?$/.test(m[2])
      ) {
        ranges.push([open.start, m.index + m[0].length]);
        open = null;
      }
    }
    if (open) ranges.push([open.start, text.length]);

    return ranges;
  }

  function inFenceRange(ranges, index) {
    return typeof index === 'number' && ranges.some(([a, b]) => index >= a && index < b);
  }

  function blankRange(chars, start, end) {
    for (let i = start; i < end && i < chars.length; i += 1) {
      if (chars[i] !== '\n') chars[i] = ' ';
    }
  }

  // Copy of the text with fenced blocks and inline code spans blanked out.
  // Index-preserving: each masked character becomes a space and newlines are
  // kept, so offsets into the result still address the same position in the
  // original. For rules where a character inside code is something the author
  // is quoting rather than using — a `#fff` in a CSS sample is not a tag.
  // Fences are blanked first so their backticks cannot pair with a later
  // inline span and swallow the prose between them.
  function maskCode(text) {
    const chars = text.split('');
    for (const [a, b] of fenceRanges(text)) blankRange(chars, a, b);
    // Indented code blocks are deliberately NOT masked. Four spaces is a code
    // block only at top level; under a list marker it is a paragraph
    // continuation, so blanking it silences real tag blocks. #90 reports
    // fences and inline spans, and those are what this masks.
    const withoutFences = chars.join('');
    const inlineRe = /(`+)(?:(?!\1)[^\n])+\1/g;
    let m;
    while ((m = inlineRe.exec(withoutFences)) !== null) blankRange(chars, m.index, m.index + m[0].length);
    return chars.join('');
  }

  function initialFrontmatterRange(text) {
    const lines = [];
    const lineRe = /[^\r\n]*(?:\r\n|\n|\r|$)/g;
    let match;
    while ((match = lineRe.exec(text)) !== null && match[0]) {
      const body = match[0].replace(/(?:\r\n|\n|\r)$/, '');
      lines.push({ body, start: match.index, end: match.index + body.length });
    }

    if (lines.length < 3 || !/^---[ \t]*$/.test(lines[0].body.replace(/^\uFEFF/, ''))) return null;

    let closingLine = -1;
    for (let i = 1; i < lines.length; i += 1) {
      if (/^---[ \t]*$/.test(lines[i].body)) {
        closingLine = i;
        break;
      }
    }
    if (closingLine === -1) return null;

    // A pair of thematic breaks can also surround ordinary Markdown prose.
    // Require the first substantive line to begin like a YAML mapping entry
    // before treating the delimited block as frontmatter. Leading blank lines
    // and YAML comments are valid, and the line parser accepts LF, CRLF, or CR.
    const yamlKey = /^[ \t]*(?:[A-Za-z0-9_.-]+|"[^"\r\n]+"|'[^'\r\n]+')[ \t]*:/;
    const firstContent = lines
      .slice(1, closingLine)
      .find((line) => line.body.trim() && !/^[ \t]*#/.test(line.body));
    if (!firstContent || !yamlKey.test(firstContent.body)) return null;

    return { start: 0, end: lines[closingLine].end };
  }

  // Mask source-only Markdown spans while preserving source offsets. The
  // detector can then score what a reader sees without making later issue
  // indexes or sentence highlights point at the wrong source location.
  function maskRenderedMarkdown(text) {
    const chars = text.split('');

    let maskedFrontmatter = 0;
    const frontmatter = initialFrontmatterRange(text);
    if (frontmatter) {
      blankRange(chars, frontmatter.start, frontmatter.end);
      maskedFrontmatter = 1;
    }

    const maskCommentCode = () => {
      const codeChars = maskCode(chars.join('')).split('');
      maskTopLevelIndentedCode(codeChars, { listAware: true });
      return codeChars.join('');
    };
    let maskedHtmlComments = 0;
    let searchIndex = 0;
    while (searchIndex < text.length) {
      const openingIndex = maskCommentCode().indexOf('<!--', searchIndex);
      if (openingIndex === -1) break;
      const closingIndex = text.indexOf('-->', openingIndex + 2);
      const end = closingIndex === -1 ? text.length : closingIndex + 3;
      blankRange(chars, openingIndex, end);
      maskedHtmlComments += 1;
      searchIndex = end;
    }

    return { text: chars.join(''), maskedFrontmatter, maskedHtmlComments };
  }

  function maskMultilineBlockquotes(text) {
    const chars = text.split('');
    const lines = [];
    const lineRe = /[^\r\n]*(?:\r\n|\n|\r|$)/g;
    let match;
    while ((match = lineRe.exec(text)) !== null && match[0]) {
      const body = match[0].replace(/[\r\n]+$/, '');
      lines.push({ text: body, start: match.index, end: match.index + body.length });
    }

    let quotedLines = 0;
    const isQuote = lines.map((line) => /^\s*>\s/.test(line.text));
    for (let i = 0; i < lines.length; i += 1) {
      if (isQuote[i] && ((isQuote[i - 1] && i > 0) || isQuote[i + 1])) {
        blankRange(chars, lines[i].start, lines[i].end);
        quotedLines += 1;
      }
    }

    return { text: chars.join(''), quotedLines };
  }

  // Keep the historical deletion behavior for default plain mode. Paragraph-
  // scoped rules depend on the surrounding lines being rejoined exactly this
  // way, so changing this prepass would change scores for existing callers.
  function stripMultilineBlockquotes(text) {
    const rawLines = text.split(/\r?\n/);
    const isQuote = rawLines.map((line) => /^\s*>\s/.test(line));
    const stripIndexes = new Set();
    for (let i = 0; i < rawLines.length; i += 1) {
      if (isQuote[i] && ((isQuote[i - 1] && i > 0) || isQuote[i + 1])) stripIndexes.add(i);
    }
    return {
      text: rawLines.filter((_, i) => !stripIndexes.has(i)).join('\n'),
      quotedLines: stripIndexes.size,
    };
  }

  function maskTopLevelIndentedCode(chars, { listAware = false } = {}) {
    const lines = chars.join('').split('\n');
    let offset = 0;
    let inBlock = false;
    let previousBlank = true;
    let listContext = false;
    for (let i = 0; i < lines.length; i += 1) {
      const line = lines[i];
      const indented = /^(?: {4}|\t)\S/.test(line);
      const blank = line.trim() === '';
      const isCode = indented && (inBlock || (previousBlank && (!listAware || !listContext)));
      if (isCode) {
        blankRange(chars, offset, offset + line.length);
        inBlock = true;
      } else if (!blank) {
        inBlock = false;
      }
      if (listAware && !blank && !isCode) {
        if (/^ {0,3}(?:[-*+]|\d{1,9}[.)])(?:\s|$)/.test(line)) listContext = true;
        else if (/^\S/.test(line)) listContext = false;
      }
      previousBlank = blank;
      offset += line.length + 1;
    }
  }

  function maskYamlFrontmatter(chars) {
    const lines = chars.join('').split('\n');
    const bare = (line) => line.replace(/\r$/, '');
    const first = bare(lines[0]).replace(/^\uFEFF/, '');
    if (first !== '---' || lines.length < 2 || /^\s*$/.test(bare(lines[1]))) return;

    let closingLine = -1;
    for (let i = 1; i < lines.length; i += 1) {
      if (bare(lines[i]) === '---') {
        closingLine = i;
        break;
      }
    }
    if (closingLine === -1) return;

    let end = 0;
    for (let i = 0; i <= closingLine; i += 1) {
      end += lines[i].length;
      if (i < lines.length - 1) end += 1;
    }
    blankRange(chars, 0, end);
  }

  function maskYamlMetadata(chars) {
    const lines = chars.join('').split('\n');
    let offset = 0;
    let nestedAfterIndent = null;
    for (const line of lines) {
      const bare = line.replace(/\r$/, '');
      // Lowercase keys are the common unfenced-YAML shape. Keeping this
      // case-sensitive prevents prose labels such as "Note: ..." from being
      // mistaken for metadata and silencing a real copyedit on the line.
      const key = bare.match(/^([ \t]*)(?:-[ \t]+)?[a-z_][a-z0-9_.-]*[ \t]*:(.*)$/);
      const indentation = (bare.match(/^[ \t]*/) || [''])[0].length;
      let shouldMask = false;

      if (key) {
        shouldMask = true;
        nestedAfterIndent = key[2].trim() === '' ? key[1].length : null;
      } else if (nestedAfterIndent !== null && bare.trim() !== '' && indentation > nestedAfterIndent) {
        shouldMask = true;
      } else if (bare.trim() === '') {
        nestedAfterIndent = null;
      } else {
        nestedAfterIndent = null;
      }

      if (shouldMask) blankRange(chars, offset, offset + line.length);
      offset += line.length + 1;
    }
  }

  function maskMarkdownTables(chars) {
    const lines = chars.join('').split('\n');
    const delimiter = /^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*\r?$/;
    const rows = new Set();
    for (let i = 0; i < lines.length; i += 1) {
      if (!delimiter.test(lines[i])) continue;
      if (i > 0 && lines[i - 1].includes('|')) rows.add(i - 1);
      rows.add(i);
      for (let j = i + 1; j < lines.length && lines[j].includes('|'); j += 1) rows.add(j);
    }

    let offset = 0;
    for (let i = 0; i < lines.length; i += 1) {
      if (rows.has(i)) blankRange(chars, offset, offset + lines[i].length);
      offset += lines[i].length + 1;
    }
  }

  function maskDelimitedQuotes(chars, open, close, apostropheAware = false) {
    const isWord = (char) => char !== undefined && /[a-z0-9]/i.test(char);
    const isEscaped = (index) => {
      let slashes = 0;
      for (let i = index - 1; i >= 0 && chars[i] === '\\'; i -= 1) slashes += 1;
      return slashes % 2 === 1;
    };
    let start = -1;
    for (let i = 0; i < chars.length; i += 1) {
      if (start === -1) {
        if (chars[i] === open && !isEscaped(i) && (!apostropheAware || !isWord(chars[i - 1]))) start = i;
      } else if (chars[i] === close && !isEscaped(i) && (!apostropheAware || !isWord(chars[i + 1]))) {
        blankRange(chars, start, i + 1);
        start = -1;
      }
    }
  }

  // Additional protected spans for the hyphenation copyedit. Unlike general
  // AI-tell matching, this rule must not "correct" a literal spelling in a
  // quote, URL, path, filename, command flag, or Markdown blockquote.
  function maskHyphenationProtected(text) {
    const chars = maskCode(text).split('');
    maskTopLevelIndentedCode(chars);
    maskYamlFrontmatter(chars);
    maskYamlMetadata(chars);
    maskMarkdownTables(chars);
    maskDelimitedQuotes(chars, '"', '"');
    maskDelimitedQuotes(chars, '“', '”');
    maskDelimitedQuotes(chars, "'", "'", true);
    maskDelimitedQuotes(chars, '‘', '’', true);
    const maskMatches = (regex) => {
      const source = chars.join('');
      let match;
      while ((match = regex.exec(source)) !== null) {
        blankRange(chars, match.index, match.index + match[0].length);
      }
    };

    maskMatches(/^[ \t]*>[^\n]*$/gm);
    maskMatches(/\b(?:https?:\/\/|www\.)[^\s<>]+/gi);
    maskMatches(/<[!?/]?[a-z][^>\n]*>/gi);
    maskMatches(/(?<![a-z0-9_-])--?[a-z0-9][a-z0-9-]{0,127}/gi);

    // Paths and filenames use bounded components. Besides preventing
    // superlinear backtracking on long kebab blobs, the explicit prefix and
    // trailing-slash forms cover single-component paths such as C:\\code-base,
    // ~/code-base, /code-base, and code-base/.
    maskMatches(/(?:[a-z]:[\\/]|\.{1,2}[\\/]|~[\\/]|[\\/])[a-z0-9_.-]{1,255}/gi);
    maskMatches(/(?:[a-z]:[\\/]|(?:\.\.?[\\/])?)(?:[a-z0-9_.-]{1,64}[\\/]){1,128}[a-z0-9_.-]{1,64}/gi);
    maskMatches(/\b[a-z0-9_.]{0,63}-[a-z0-9_.-]{1,64}[\\/]/gi);
    maskMatches(/\b[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}\.[a-z0-9]{1,16}\b/gi);

    // Technical identifiers that are distinguishable from ordinary prose:
    // selectors, scoped packages, versioned tokens, assignments, and kebab
    // names next to an explicit identifier cue. Plain lowercase compounds
    // remain visible to the curated copyedit patterns below.
    maskMatches(/[.#][a-z_][a-z0-9_.-]{0,63}-[a-z0-9_.-]{1,64}\b/gi);
    maskMatches(/@[a-z0-9_.-]{1,64}\/[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}(?:@[^\s,;)\]}]{1,32})?/gi);
    maskMatches(/\b[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}@[~^]?v?\d[a-z0-9*_.+-]{0,31}\b/gi);
    maskMatches(/\b(?:[a-z0-9_.-]{0,64}\d[a-z0-9_.-]{0,64}-[a-z0-9_.-]{1,64}|[a-z0-9_.-]{1,64}-[a-z0-9_.-]{0,64}\d[a-z0-9_.-]{0,64})\b/gi);
    maskMatches(/\b[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}(?=[ \t]*[=:])/gi);
    maskMatches(/\b[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}(?=[ \t]+(?:npm[ \t]+)?(?:package|module|class|selector|config(?:uration)?[ \t]+key|key|identifier|property|setting|token|slug|command|option)\b)/gi);
    maskMatches(/\b(?:(?:file(?:name)?|directory|folder|package|module|class|selector|config(?:uration)?[ \t]+key|identifier|property|setting|token|slug|command|option)(?:[ \t]+(?:named|called|is|was))?|key[ \t]+(?:named|called|is|was))[ \t]+(?:@[a-z0-9_.-]{1,64}\/)?[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}\b/gi);
    maskMatches(/\b(?:npm|pnpm|yarn)[ \t]+(?:add|install)[ \t]+(?:@[a-z0-9_.-]{1,64}\/)?[a-z0-9_.]{1,64}(?:-[a-z0-9_.]{1,64}){1,8}/gi);

    return chars.join('');
  }

  function findUnnecessaryHyphenation(text) {
    const scanText = maskHyphenationProtected(text);
    const issues = [];
    for (const entry of UNNECESSARY_HYPHENATION) {
      const regex = new RegExp(entry.pattern.source, entry.pattern.flags);
      let match;
      while ((match = regex.exec(scanText)) !== null) {
        issues.push({
          type: 'unnecessary-hyphenation',
          text: match[0],
          severity: 'medium',
          suggestion: typeof entry.suggestion === 'function'
            ? entry.suggestion(match[0])
            : entry.suggestion,
        });
      }
    }
    return issues;
  }

  // ─── Forms that open with `#` but are not social tags ──────────────
  // `#` is overloaded in technical prose and the hashtag rule counts every
  // `#word` it sees, so these are subtracted before the threshold applies:
  //   #88, #1234         issue and PR references
  //   #1a2b3c            CSS hex colours, 6 or 8 chars AND containing a digit
  //   #include, #ifndef  C preprocessor directives
  // Deliberately NOT carving out 3- and 4-digit hex: #dad, #cafe, #b2b, #e2e,
  // #ace, #face and #bad are real tags, and subtracting them cost true
  // positives on exactly the stuffed-block shape this rule exists to catch.
  // Real palettes are dominated by 6-digit values, so a CSS paragraph still
  // lands under the threshold without the short forms.
  // `owner/repo#88` and URL fragments need no carve-out: the char before `#`
  // is a word char, so the rule's own anchor already rejects them.
  // Ambiguous word tags stay counted on purpose. `#general` as a channel and
  // `#general` as a tag are the same token, and separating them needs a guess
  // that costs more precision on real tag blocks than the carve-out buys.
  // Requires at least one digit: #decade, #facade, #deadbeef are a-f words and
  // real tags. Every actual palette value in the wild carries a digit.
  const HEX_COLOUR = /^(?=[0-9a-f]*\d)(?:[0-9a-f]{6}|[0-9a-f]{8})$/i;
  const CPP_DIRECTIVE = /^(?:include|define|undef|if|ifdef|ifndef|elif|else|endif|pragma|error|warning|line)$/;

  function isSocialTag(tag) {
    return !/^\d+$/.test(tag) && !HEX_COLOUR.test(tag) && !CPP_DIRECTIVE.test(tag);
  }

  // ─── Title Case Section Headers in non-technical prose ─────────────
  // "Strategic Negotiations And Key Partnerships" — every content word
  // capitalized. Acceptable in API docs, ML papers, news headlines. Tell
  // in marketing/personal/blog prose. Gated to "personal" / "marketing"
  // context modes (technical mode skips this check).
  //
  // The optional `#{1,6}` prefix is load-bearing (#62): without it the `^[A-Z]`
  // anchor required the line to START with a capital, so `## Benefits And
  // Strategic Considerations` never matched — the first character is `#`. The
  // rule missed the single most common way a heading is actually written, while
  // catching the bare-line form it is usually converted from. Reported by a
  // downstream vendoring the detector.
  //
  // Setext headings (`Title`/`=====`) need no prefix: their text line is bare
  // and already matched by this same pattern.
  const TITLE_CASE_HEADER = /^(?:#{1,6}[ \t]+)?([A-Z][a-z]+(?:\s+(?:[A-Z][a-z]+|and|or|of|the|in|for|to|a|an))+\s+[A-Z][a-z]+)\s*$/gm;

  // ─── Parenthetical hedging asides ──────────────────────────────────
  // "(and increasingly, X)", "(or more precisely, Y)", "(though to be
  // fair, Z)" — pseudo-aside that adds no information but performs
  // thoughtfulness. Different from genuine human parentheticals which
  // tend to be tangents or clarifications, not hedges.
  const PARENTHETICAL_HEDGE = [
    /\(\s*(?:and\s+)?(?:increasingly|notably|importantly|crucially|interestingly|perhaps)[,]?\s+[^)]{3,60}\)/gi,
    /\(\s*or\s+more\s+(?:precisely|accurately|specifically)[,]?\s+[^)]{3,60}\)/gi,
    /\(\s*though\s+to\s+be\s+fair[,]?\s+[^)]{3,60}\)/gi,
    /\(\s*at\s+least\s+(?:in\s+)?(?:theory|principle|part)[,]?\s+[^)]{0,60}\)/gi,
  ];

  // ─── Confidence calibration ────────────────────────────────────────
  const CONFIDENCE_CALIBRATION = [
    /\binterestingly\b/gi,
    /\bsurprisingly\b/gi,
    /\bimportantly\b/gi,
    /\bsignificantly\b/gi,
    /\bcertainly\b/gi,
    /\bundoubtedly\b/gi,
    /\bwithout\s+a\s+doubt\b/gi,
  ];

  // ─── Social endorsement / CTA closers ──────────────────────────────
  // The curatorial sign-off LLMs append to LinkedIn / X posts that share
  // or recommend something — usually a colon teeing up a link. Distinct
  // from the bare "worth reading" word-table entry (a single weak word)
  // and from infomercial hooks (mid-flow teasers): this is the
  // demonstrative-anchored endorsement — "THIS one is worth your time:",
  // "do yourself a favor and read this", "thank me later" — that performs
  // a recommendation without giving the reader a reason to click.
  //
  // Precision-first: each pattern carries an anchor so it stays off the
  // literal-verb prose a human writes. The demonstrative object ("read
  // THIS", not "read the runbook"), the trailing-terminal lookahead on
  // "miss this" / "bookmark this" (the closing-line shape, not "miss this
  // meeting" / "bookmark this page"), and the sentence-initial lookbehind
  // on "thank me later" / "save this for later" (the imperative CTA, not
  // "she will thank me later") all exist to suppress false positives on
  // ordinary instructional/conversational text. Apostrophe classes admit
  // the curly ' (U+2019) because LinkedIn / Word / macOS auto-curl it —
  // the straight-only form would miss the canonical "you won't" closer.
  const SOCIAL_CTA_CLOSER = [
    /\bthis\s+one['’]?s?\s+(?:is\s+)?(?:well\s+|totally\s+|absolutely\s+|definitely\s+|really\s+|truly\s+|easily\s+|more\s+than\s+)?worth\s+(?:your\s+time|the\s+read|a\s+read|every\s+(?:minute|second)|reading|watching|a\s+listen|a\s+watch|a\s+look|it)\b/gi,
    /\bthis\s+one['’]?s?\s+(?:is\s+)?a\s+must[-\s]?(?:read|watch|listen|see)\b/gi,
    /\b(?:highly|strongly|can['’]?t|cannot)\s+recommend\w*\s+(?:giving\s+)?(?:this|it)\s+(?:one\s+)?a\s+(?:read|listen|watch|look|go)\b/gi,
    /\bdo\s+yourself\s+a\s+favou?r\s+and\s+(?:read|watch|check\s+out)\s+(?:this|it)\b/gi,
    /\byou\s+(?:really\s+)?(?:won['’]?t|do\s*n['’]?t|will\s+not|do\s+not)\s+want\s+to\s+miss\s+this(?:\s+one)?(?=\s*(?:[:.!\n]|$))/gi,
    /(?<=^|[,.!?:\n]\s{0,4})(?:you\s+can\s+)?thank\s+me\s+later\b/gim,
    /(?<=^|[.!?:\n]\s{0,4})save\s+this\s+(?:one\s+)?for\s+later\b/gim,
    /\bbookmark\s+this(?:\s+(?:one|post|thread))?(?=\s*(?:[:.!\n]|$))/gi,
    /\bdo\s*n['’]?t\s+sleep\s+on\s+this\b/gi,
    /\btrust\s+me,?\s+(?:on\s+this|you['’]?ll)\b/gi,
  ];

  // ─── Unnecessary hyphenation (#107) ───────────────────────────────
  // Precision-first subclasses only. The general question of whether a
  // compound modifier is established English needs editorial judgment and
  // remains in SKILL.md; the engine covers only curated open/closed forms and
  // compounds whose surrounding syntax makes the unhyphenated form clear.
  const UNNECESSARY_HYPHENATION = [
    // Welded open noun phrases reported in #107. Match the complete phrase so
    // a project-specific spelling of the pair in another role is not swept in.
    { pattern: /\bresearch-impact\s+aggregat(?:or|ion)s?\b/g, suggestion: (match) => match.replace('research-impact', 'research impact') },
    { pattern: /\bdata-source\s+strateg(?:y|ies)\b/g, suggestion: (match) => match.replace('data-source', 'data source') },
    { pattern: /\bPython-package\s+usage\b/g, suggestion: (match) => match.replace('Python-package', 'Python package') },
    { pattern: /\bRust-crate\s+usage\b/g, suggestion: (match) => match.replace('Rust-crate', 'Rust crate') },
    { pattern: /\bsingle-Project\s+Manifest\b/g, suggestion: (match) => match.replace('single-Project', 'single Project') },
    { pattern: /\btotal-downloads\s+figures?\b/g, suggestion: (match) => match.replace('total-downloads', 'total downloads') },
    { pattern: /\blife-sciences-native\s+citation\s+count\b/g, suggestion: 'citation count from a life sciences source' },

    // Compounds whose standard spelling is closed. Kept as a small curated
    // list rather than guessing that every noun-noun pair should close up.
    { pattern: /\bcode-base\b/g, suggestion: (match) => match.replace('-', '') },
    { pattern: /\bdata-set\b/g, suggestion: (match) => match.replace('-', '') },
    { pattern: /\btime-frame\b/g, suggestion: (match) => match.replace('-', '') },
    { pattern: /\broad-map\b/g, suggestion: (match) => match.replace('-', '') },

    // Attributive-only forms used adverbially or as nouns. The boundary after
    // real-time / long-term is intentionally narrow: "real-time analytics"
    // and "long-term plan" must not fire.
    {
      pattern: /\bin\s+real-time(?=\s*(?:[,.!?;:]|$)|\s+(?:(?:across|as|automatically|because|but|continuously|during|dynamically|every|for|from|immediately|instantly|on|simultaneously|through|throughout|until|via|when|while|with|without)\b))/gi,
      suggestion: 'in real time',
    },
    {
      pattern: /\b(?:for|over)\s+the\s+long-term(?=\s*(?:[,.!?;:]|$)|\s+(?:across|because|but|by|during|for|from|on|through|throughout|until|via|when|while|with|without)\b)/gi,
      suggestion: (match) => match.replace(/long-term/i, 'long term'),
    },
    {
      pattern: /\b(?:functions?|functioned|functioning|operates?|operated|operating|runs?|ran|running|works?|worked|working)\s+out-of-the-box\b/gi,
      suggestion: (match) => match.replace(/out-of-the-box/i, 'out of the box'),
    },
  ];

  // ═══ Helpers ═══════════════════════════════════════════════════════

  function tokenize(text) {
    return text.toLowerCase().match(/[\w'-]+/g) || [];
  }

  function countWords(text) {
    return (text.match(/\S+/g) || []).length;
  }

  function getParagraphs(text) {
    return text.split(/\n\s*\n/).filter(p => p.trim().length > 0);
  }

  function getSentences(text) {
    return text.split(/[.!?]+/).filter(s => s.trim().length > 5);
  }

  function matchPatterns(text, patterns, category, severity) {
    const issues = [];
    for (const pat of patterns) {
      const regex = new RegExp(pat.source, pat.flags);
      let match;
      while ((match = regex.exec(text)) !== null) {
        issues.push({
          type: category,
          text: match[0],
          index: match.index,
          severity,
          suggestion: null,
        });
      }
    }
    return issues;
  }

  // ═══ Main analysis ═════════════════════════════════════════════════

  // Upper bound for one scan. Above this we bail rather than running all
  // regex passes over a huge buffer — protects page perf on pasted novels.
  const MAX_WORDS = 10000;

  // V2 contract defaults so early-exit paths (Empty/tooShort/tooLong)
  // still return the same field shape a v2 consumer expects. Without
  // this, `result.document_classification === 'AI_ONLY'` is `undefined`
  // on edge inputs and fails open.
  // UNSCORED is returned on empty / too-short / too-long inputs where
  // we declined to score. Distinct from HUMAN_ONLY (which is a positive
  // classification) so a caller can't mistake a refused scan for a
  // confident human verdict — a 50k-word LLM-generated document is
  // not "human", it's just outside our scoring window.
  function buildV2Defaults(classification, confidence) {
    const probs = classification === 'HUMAN_ONLY'
      ? { human: 1, mixed: 0, ai: 0 }
      : classification === 'AI_ONLY'
        ? { human: 0, mixed: 0, ai: 1 }
        : { human: 0.333, mixed: 0.334, ai: 0.333 };
    return {
      document_classification: classification,
      class_probabilities: probs,
      confidence_category: confidence,
      highlight_sentence_for_ai: [],
    };
  }

  function analyzeText(text, options = {}) {
    if (!text || text.trim().length === 0) {
      return { ...buildV2Defaults('UNSCORED', 'low'), score: 0, label: 'Empty', issues: [], stats: {}, tooShort: true };
    }

    // Context mode gates rules that are noisy in technical writing. Modes:
    //   'general' (default) — full ruleset
    //   'technical' — skip title-case headers, formulaic openers gated to
    //                 prose-only structures; lower em-dash + formatting weights
    //   'marketing' — full ruleset + boost on formulaic-opener / future-narrative
    //   'personal'  — full ruleset, normal weights
    // Mode is purely a soft gate; nothing is silently suppressed without
    // being reflected in stats.contextMode for transparency.
    // Mode validation: an unknown string (e.g. typo "tecnical") would
    // otherwise silently downgrade to general-mode behavior. Coerce to
    // 'general' and surface the original value in stats for traceability.
    const VALID_CONTEXT_MODES = new Set(['general', 'technical', 'marketing', 'personal']);
    const requestedMode = options.contextMode || 'general';
    const contextMode = VALID_CONTEXT_MODES.has(requestedMode) ? requestedMode : 'general';
    const contextModeFallback = requestedMode !== contextMode ? requestedMode : null;

    // Source mode controls which parts of a Markdown file count as prose.
    // Plain remains the compatibility default. Rendered Markdown masks only
    // initial YAML frontmatter and HTML comments; source-hygiene checks for
    // hidden TODO/placeholder comments remain available through plain mode.
    const VALID_SOURCE_MODES = new Set(['plain', 'rendered-markdown']);
    const requestedSourceMode = options.sourceMode === undefined ? 'plain' : options.sourceMode;
    const sourceMode = VALID_SOURCE_MODES.has(requestedSourceMode) ? requestedSourceMode : 'plain';
    const sourceModeFallback = requestedSourceMode !== sourceMode ? requestedSourceMode : undefined;
    let maskedFrontmatter = 0;
    let maskedHtmlComments = 0;
    if (sourceMode === 'rendered-markdown') {
      const rendered = maskRenderedMarkdown(text);
      text = rendered.text;
      maskedFrontmatter = rendered.maskedFrontmatter;
      maskedHtmlComments = rendered.maskedHtmlComments;
    }

    // Pre-pass: mask Markdown blockquotes before scoring. A human
    // reacting to AI text by quoting it shouldn't have the quoted block
    // counted against their own writing. Requires ≥2 consecutive `> `
    // lines to count as a blockquote — single-line `> ls -la` shell
    // prompts in technical docs stay in the text. Masking instead of deleting
    // keeps later issue and highlight offsets aligned with the source file.
    const blockquotes = sourceMode === 'rendered-markdown'
      ? maskMultilineBlockquotes(text)
      : stripMultilineBlockquotes(text);
    text = blockquotes.text;
    const { quotedLines } = blockquotes;

    // Pre-pass: strip bypass-trick chars before pattern matching so
    // "delve" with a Cyrillic 'е' still hits Tier 1. Original text is
    // preserved so reported `match.index` values remain visually accurate.
    const norm = normalizeText(text);
    text = norm.text;

    const wordCount = countWords(text);
    if (wordCount < 10) {
      return {
        ...buildV2Defaults('UNSCORED', 'low'),
        score: 0,
        label: 'Too short',
        issues: [],
        stats: { wordCount, contextMode, contextModeFallback, sourceMode, sourceModeFallback, maskedFrontmatter, maskedHtmlComments },
        tooShort: true,
      };
    }
    if (wordCount > MAX_WORDS) {
      return {
        ...buildV2Defaults('UNSCORED', 'low'),
        score: 0,
        label: 'Text too long',
        issues: [],
        stats: { wordCount, contextMode, contextModeFallback, sourceMode, sourceModeFallback, maskedFrontmatter, maskedHtmlComments },
        tooLong: true,
      };
    }

    const tokens = tokenize(text);
    const paragraphs = getParagraphs(text);
    const sentences = getSentences(text);
    const issues = [];
    let rawScore = 0;

    // ── 1. Tier 1 words ──────────────────────────────────────────
    const tier1Found = new Set();
    for (const token of tokens) {
      if (Object.hasOwn(TIER1, token) && !tier1Found.has(token)) {
        tier1Found.add(token);
        issues.push({
          type: 'tier1',
          text: token,
          severity: 'high',
          suggestion: TIER1[token],
        });
      }
    }

    // Tier 1 multi-word phrases. Adds each distinct phrase (lowercased) to
    // `tier1Found` so the same phrase hit multiple times only produces one
    // issue — matches the downstream dedup behavior.
    for (const phrase of TIER1_PHRASES) {
      const regex = new RegExp(phrase.pattern.source, phrase.pattern.flags);
      let match;
      while ((match = regex.exec(text)) !== null) {
        const lower = match[0].toLowerCase();
        if (tier1Found.has(lower)) continue;
        tier1Found.add(lower);
        issues.push({
          // Clarity-band entries are wordiness edits, not frequency evidence.
          // Same fix, weaker claim — see the Tier 1A/1B split in SKILL.md.
          type: phrase.clarity ? 'tier1-clarity' : 'tier1',
          text: match[0],
          severity: phrase.clarity ? 'medium' : 'high',
          suggestion: phrase.replace,
        });
      }
    }

    // ── 2. Tier 2 clusters ───────────────────────────────────────
    let tier2Clusters = 0;
    for (const para of paragraphs) {
      const paraTokens = tokenize(para);
      const found = [];
      const suggestions = {};
      for (const token of paraTokens) {
        if (Object.hasOwn(TIER2, token) && !found.includes(token)) {
          found.push(token);
          suggestions[token] = TIER2[token];
        }
      }
      for (const cond of TIER2_CONDITIONAL) {
        if (!found.includes(cond.word) && cond.pattern.test(para)) {
          found.push(cond.word);
          suggestions[cond.word] = cond.suggestion;
        }
      }
      if (found.length >= 2) {
        tier2Clusters++;
        for (const word of found) {
          issues.push({
            type: 'tier2',
            text: word,
            severity: 'medium',
            suggestion: suggestions[word],
          });
        }
      }
    }

    // ── 3. Tier 3 density ────────────────────────────────────────
    const tier3Counts = {};
    for (const token of tokens) {
      const canonical = TIER3_LOOKUP.get(token);
      if (canonical) tier3Counts[canonical] = (tier3Counts[canonical] || 0) + 1;
    }
    // Flag at 3% of word count, but never below 3 occurrences. Previous
    // floor of 1 meant a 50-word text with one "significant" got flagged
    // as Tier 3 overuse, which was noise.
    const densityThreshold = Math.max(3, Math.floor(wordCount * 0.03));
    let tier3Flags = 0;
    for (const [word, count] of Object.entries(tier3Counts)) {
      if (count >= densityThreshold) {
        tier3Flags++;
        issues.push({
          type: 'tier3',
          text: `"${word}" x${count}`,
          severity: 'low',
          suggestion: `Overused (${count} times in ${wordCount} words)`,
        });
      }
    }

    // ── 4–21. Pattern categories ─────────────────────────────────
    issues.push(...matchPatterns(text, TRANSITIONS, 'transition', 'medium'));
    issues.push(...matchPatterns(text, CHATBOT_ARTIFACTS, 'chatbot', 'critical'));
    issues.push(...matchPatterns(text, SYCOPHANTIC, 'sycophantic', 'critical'));
    issues.push(...matchPatterns(text, FILLERS, 'filler', 'medium'));
    issues.push(...matchPatterns(text, GENERIC_CONCLUSIONS, 'generic-conclusion', 'medium'));
    issues.push(...matchPatterns(text, LETS_PATTERNS, 'lets-construction', 'medium'));
    issues.push(...matchPatterns(text, REASONING_ARTIFACTS, 'reasoning-artifact', 'critical'));
    issues.push(...matchPatterns(text, ACKNOWLEDGMENT_LOOPS, 'acknowledgment-loop', 'medium'));
    issues.push(...matchPatterns(text, SIGNIFICANCE_INFLATION, 'significance-inflation', 'high'));
    issues.push(...matchPatterns(text, VAGUE_ATTRIBUTIONS, 'vague-attribution', 'critical'));
    issues.push(...matchPatterns(text, HOLLOW_INTENSIFIERS, 'hollow-intensifier', 'medium'));
    issues.push(...matchPatterns(text, EMOTIONAL_FLATLINE, 'emotional-flatline', 'low'));
    issues.push(...matchPatterns(text, LINGERING_ATTENTION, 'lingering-attention', 'medium'));
    issues.push(...matchPatterns(text, NOVELTY_INFLATION, 'novelty-inflation', 'medium'));
    issues.push(...matchPatterns(text, CUTOFF_DISCLAIMERS, 'cutoff-disclaimer', 'critical'));
    issues.push(...matchPatterns(text, AI_PLACEHOLDERS, 'ai-placeholder', 'critical'));
    issues.push(...matchPatterns(text, AI_CITATION_MARKUP, 'ai-citation-markup', 'critical'));
    issues.push(...matchPatterns(text, AI_UTM_SOURCE, 'ai-utm-source', 'critical'));
    issues.push(...matchPatterns(text, TEMPLATE_PHRASES, 'template-phrase', 'high'));
    issues.push(...matchPatterns(text, FALSE_CONCESSION, 'false-concession', 'medium'));
    issues.push(...matchPatterns(text, RHETORICAL_QUESTIONS, 'rhetorical-question', 'medium'));
    issues.push(...matchPatterns(text, HEDGE_STACK, 'hedge-stack', 'high'));
    issues.push(...matchPatterns(text, FUTURE_NARRATIVE, 'future-narrative', 'high'));
    issues.push(...matchPatterns(text, REAL_ACTUAL_INFLATION, 'real-actual-inflation', 'medium'));
    issues.push(...matchPatterns(text, SOCIAL_CTA_CLOSER, 'social-cta-closer', 'high'));
    issues.push(...matchPatterns(text, PERFORMED_INSIGHT, 'performed-insight', 'medium'));
    issues.push(...matchPatterns(text, NEGATION_CHAIN, 'negation-chain', 'high'));
    issues.push(...matchPatterns(text, DEV_BLOG_BOILERPLATE, 'dev-blog-boilerplate', 'medium'));
    issues.push(...findUnnecessaryHyphenation(text));

    // ── Tier 1 v2: formulaic openers + parenthetical hedges ──────────
    issues.push(...matchPatterns(text, FORMULAIC_OPENERS, 'formulaic-opener', 'high'));
    issues.push(...matchPatterns(text, SPECULATIVE_OPENERS, 'speculative-opener', 'high'));
    issues.push(...matchPatterns(text, LAUNCH_INTROS, 'launch-intro', 'high'));
    issues.push(...matchPatterns(text, CROWD_CONTRAST, 'crowd-contrast', 'medium'));
    issues.push(...matchPatterns(text, FAKE_CASUAL_PROPS, 'fake-casual-prop', 'high'));
    issues.push(...matchPatterns(text, PARENTHETICAL_HEDGE, 'parenthetical-hedge', 'medium'));

    // Title-case headers — gated to marketing/personal/general modes
    // (technical mode legitimately uses Title Case section headers).
    if (contextMode !== 'technical') {
      const titleHits = matchPatterns(text, [TITLE_CASE_HEADER], 'title-case-header', 'medium');
      // Drop matches that look like proper-noun titles (single line, all
      // tokens capitalized incl. function words) — that's headline style,
      // not the AI-section-header tell which has mid-sentence "And".
      //
      // The prefix strip is load-bearing. matchPatterns reports match[0], so a
      // Markdown hit arrives as "## Terms Of Service" and `##` counts as a
      // token — silently lowering this guard from four content words to three
      // for headings only, which is exactly the class it exists to protect.
      // "## Terms Of Service", "## Bank Of America" and "## Table Of Contents"
      // all flagged as a result: ordinary human headings, on a detector whose
      // stated first priority is not firing on human writing.
      const filtered = titleHits.filter((h) => {
        const title = h.text.replace(MD_HEADING_PREFIX, '');
        const tokens = title.trim().split(/\s+/);
        if (tokens.length < 4) return false;

        // The function word must be MID-title, which is what the comment above
        // has always said and what the test never enforced. A leading "The"
        // satisfied a bare /\bThe\b/, so ordinary human headings flagged:
        // "## The New Security Landscape", "## The Microsoft Approach to
        // Identity", "### The Four Keys to a Successful and Secure Modern
        // Workplace". Measured across 81 files that provably predate LLMs
        // (2018-19 eBooks, 2020 posts): 13 false positives, every one opening
        // with "The", against zero on main.
        //
        // "## Benefits And Strategic Considerations" -- the actual tell, and
        // this rule's own fixture -- is untouched: its "And" is interior.
        return FUNCTION_WORD.test(tokens.slice(1).join(' '));
      });
      const fences = filtered.length ? fenceRanges(text) : [];
      issues.push(...filtered.filter((h) => !inFenceRange(fences, h.index)));
    }

    // ── Normalization-trigger flag ───────────────────────────────────
    // ZWSPs or homoglyphs in pasted prose are near-dispositive: humans
    // don't insert these into their own writing. Single roleplay marker
    // can be a false positive on Markdown emphasis (filtered to multi-
    // word inner already), so requires ≥2.
    if (norm.flags.zeroWidth > 0 || norm.flags.homoglyph >= 2) {
      issues.push({
        type: 'normalization-flag',
        text: `${norm.flags.zeroWidth} zero-width + ${norm.flags.homoglyph} homoglyph swap${norm.flags.homoglyph === 1 ? '' : 's'}`,
        severity: 'critical',
        suggestion: 'Text contains invisible/lookalike chars typical of AI-humanizer bypass tools. Re-type from your own keyboard.',
      });
    }
    if (norm.flags.roleplay >= 2) {
      issues.push({
        type: 'normalization-flag',
        text: `${norm.flags.roleplay} *roleplay-action* markers stripped`,
        severity: 'high',
        suggestion: 'Paired *action* markers are a chat-model artifact.',
      });
    }

    // Em dashes in list-item separator position — a bulleted or numbered
    // list item opening with a bolded lead term or markdown link, then the
    // dash ("- **Term** — desc", "- [label](url) — desc") — are
    // definition-list typography, not prose punctuation. Shared by the
    // smart-punct signature below and the em-dash frequency check (§22).
    // An optional parenthetical or inline-code span may sit between the bold
    // lead term and the dash — "- **Lingering-attention claims**
    // (`lingering-attention`) — the share-post frame…" is the same definition
    // typography as the bare form. Found by the self-scan (see PROOF.md, #67).
    const SEPARATOR_DASH_RE = /^\s*(?:[-*+]|\d+[.)])\s+(?:\*\*[^*\n]+\*\*|\[[^\]\n]+\]\([^)\n]*\))(?:[ \t]*(?:\([^)\n]*\)|`[^`\n]+`))?[ \t]*—/gm;

    // Keep-a-Changelog version headings (`## [3.21.0] — 2026-07-30`) join a
    // label to a value exactly as a list separator does. Deliberately narrow:
    // a bracketed or bare semver token, then a dash, then an ISO date, and
    // nothing else on the line. Ordinary prose dashes in headings still count,
    // because SKILL.md applies the em-dash rule to headings too.
    const VERSION_HEADING_DASH_RE = /^#{1,6}[ \t]+\[?v?\d+\.\d+\.\d+[^\]\n]*\]?[ \t]*—[ \t]*\d{4}-\d{2}-\d{2}[ \t]*$/gm;

    // ── Smart-punctuation co-occurrence signature ────────────────────
    // Curly quotes + em-dash + Oxford comma all present + zero typos
    // (no double-spaces, no missing apostrophes in common contractions)
    // is a near-dispositive paste-from-LLM signature: humans typing
    // directly into a textarea don't produce all four. Standalone any of
    // these is meaningless — co-occurrence is the signal. Separator-position
    // dashes are typography and don't corroborate it.
    {
      const hasCurly = /[“”‘’]/.test(text);
      const totalEmDashes = (text.match(/—/g) || []).length;
      const separatorEmDashes = (text.match(SEPARATOR_DASH_RE) || []).length
        + (text.match(VERSION_HEADING_DASH_RE) || []).length;
      const hasEmDash = totalEmDashes > separatorEmDashes;
      const oxfordHit = text.match(/\b\w+,\s+\w+,\s+and\s+\w+/g);
      const hasOxford = (oxfordHit?.length || 0) >= 1;
      const doubleSpaces = (text.match(/[^.!?]  +/g) || []).length;
      const missingApos = /\b(?:dont|wont|cant|isnt|wasnt|shouldnt|wouldnt|couldnt|youre|theyre|its\s+a\s+\w+ing)\b/i.test(text);
      const clean = doubleSpaces === 0 && !missingApos;
      const signals = [hasCurly, hasEmDash, hasOxford, clean].filter(Boolean).length;
      if (signals >= 4 && wordCount >= 80) {
        issues.push({
          type: 'smart-punct-signature',
          text: 'curly-quotes + em-dash + Oxford comma + zero typos',
          severity: 'high',
          suggestion: 'Smart-punctuation signature consistent with LLM output. Humans typing into textareas rarely produce all four.',
        });
      }
    }

    // ── Punctuation distribution mode ────────────────────────────────
    // Humans cluster trimodal across paragraphs (some paras heavy, some
    // light, some none). AI converges on a normal distribution. We can't
    // run a real modality test client-side, but we can flag the AI
    // signature: low variance of per-paragraph punctuation density.
    // Requires ≥4 paragraphs to be meaningful.
    if (paragraphs.length >= 4) {
      const densities = paragraphs.map((p) => {
        const words = (p.match(/\S+/g) || []).length;
        if (words < 5) return null;
        const puncts = (p.match(/[,;:—()]/g) || []).length;
        return puncts / words;
      }).filter((d) => d !== null);
      if (densities.length >= 4) {
        const mean = densities.reduce((a, b) => a + b, 0) / densities.length;
        const variance = densities.reduce((s, d) => s + (d - mean) ** 2, 0) / densities.length;
        const cv = mean > 0 ? Math.sqrt(variance) / mean : 0;
        // CV < 0.25 across paragraphs means each paragraph has the same
        // punctuation density — the AI signature. Humans usually swing
        // wider. Threshold derived from stylometry papers (arxiv 2507.00838).
        if (cv < 0.25 && mean >= 0.04) {
          issues.push({
            type: 'punct-distribution',
            text: `Punctuation density uniform across paragraphs (CV=${cv.toFixed(2)})`,
            severity: 'medium',
            suggestion: 'AI text holds punctuation density steady; human writers swing between dense and sparse paragraphs.',
          });
        }
      }
    }

    // ── Function-word trigram entropy ────────────────────────────────
    // POS-trigram entropy is the academic signal; function-word trigram
    // entropy approximates it without a tagger (function words ARE the
    // closed-class POS classes). AI text has lower entropy because LLM
    // sampling collapses onto a narrower set of grammatical templates.
    //
    // Method: extract function-word indicators per sentence, build
    // trigrams over the sequence, compute Shannon entropy. Bins below
    // threshold flag.
    if (wordCount >= 150) {
      const FUNC_WORDS = new Set([
        'the','a','an','and','or','but','of','to','in','on','at','by','for','with',
        'from','as','is','was','are','were','be','been','being','have','has','had',
        'do','does','did','will','would','should','could','may','might','must','can',
        'this','that','these','those','it','its','they','them','their','there','here',
        'we','our','us','i','you','your','he','she','his','her','him','not','no','so',
        'if','then','than','when','where','which','who','what','how','why','because',
      ]);
      const seq = tokens.map((t) => FUNC_WORDS.has(t) ? t : '_').filter((_, i, arr) => arr[i] !== '_' || (i > 0 && arr[i - 1] !== '_'));
      if (seq.length >= 50) {
        const trigrams = {};
        for (let i = 0; i < seq.length - 2; i++) {
          const tg = `${seq[i]}|${seq[i + 1]}|${seq[i + 2]}`;
          trigrams[tg] = (trigrams[tg] || 0) + 1;
        }
        const total = seq.length - 2;
        let entropy = 0;
        for (const c of Object.values(trigrams)) {
          const p = c / total;
          entropy -= p * Math.log2(p);
        }
        // Normalize by log2(distinct trigrams) so entropy ranges roughly
        // 0..1 and threshold is interpretable. Empirical threshold: human
        // prose ~0.85-0.95 normalized, AI prose ~0.70-0.82.
        const distinctCount = Object.keys(trigrams).length;
        const normalized = distinctCount > 1 ? entropy / Math.log2(distinctCount) : 1;
        if (normalized < 0.82 && total >= 50) {
          issues.push({
            type: 'fnword-trigram-entropy',
            text: `Function-word trigram entropy ${normalized.toFixed(2)} (low)`,
            severity: 'medium',
            suggestion: 'Grammatical structure is unusually repetitive. AI sampling collapses onto narrower templates than human writing.',
          });
        }
        // Degenerate case: single distinct trigram repeated across the
        // whole document is the strongest possible AI signal but the
        // normalized fallback returns 1.0 (= "fully human"), inverting
        // the signal. Catch it explicitly.
        if (distinctCount === 1 && total >= 50) {
          issues.push({
            type: 'fnword-trigram-entropy',
            text: 'Single function-word trigram repeated across document',
            severity: 'high',
            suggestion: 'Grammatical structure is fully degenerate — every clause uses the same function-word skeleton.',
          });
        }
      }
    }

    // ── Cross-paragraph burstiness ───────────────────────────────────
    // We already check within-paragraph sentence-length uniformity. AI
    // is also flat ACROSS paragraphs — every paragraph has roughly the
    // same sentence-length variance. Humans vary: terse paras next to
    // discursive paras. Measure variance of CV across paragraphs.
    if (paragraphs.length >= 4) {
      const cvs = paragraphs.map((p) => {
        const sents = getSentences(p);
        if (sents.length < 3) return null;
        const lens = sents.map(countWords);
        const m = lens.reduce((a, b) => a + b, 0) / lens.length;
        if (m === 0) return null;
        const v = lens.reduce((s, l) => s + (l - m) ** 2, 0) / lens.length;
        return Math.sqrt(v) / m;
      }).filter((c) => c !== null);
      if (cvs.length >= 4) {
        const cvMean = cvs.reduce((a, b) => a + b, 0) / cvs.length;
        const cvVar = cvs.reduce((s, c) => s + (c - cvMean) ** 2, 0) / cvs.length;
        const cvStd = Math.sqrt(cvVar);
        // Std-of-CV below 0.08 means every paragraph has roughly the same
        // internal rhythm — AI signature. Human prose typically swings
        // 0.15-0.40 across paragraphs of mixed purpose.
        if (cvStd < 0.08 && cvMean < 0.45) {
          issues.push({
            type: 'cross-para-burstiness',
            text: `Sentence-rhythm uniform across paragraphs (σCV=${cvStd.toFixed(2)})`,
            severity: 'medium',
            suggestion: 'Every paragraph has the same internal rhythm. Humans vary cadence between terse and discursive paragraphs.',
          });
        }
      }
    }

    // ── Tier 3 multi-word phrase density ─────────────────────────
    // Two complementary rules:
    //   (a) Per-phrase density — same gating as single-word Tier 3: each
    //       phrase fine alone, repetition is the tell. Threshold = 2.
    //   (b) Cross-phrase clustering — ≥3 *distinct* boilerplate phrases
    //       in one piece. LLMs varying their own boilerplate often use
    //       each phrase only once but stack 5-10 across the text. The
    //       per-phrase rule misses this; the cluster rule catches it.
    // Track non-overlapping match spans so a longer phrase swallowing a
    // shorter one (e.g., "designed for long-term sustainability" matches
    // both "designed for long-term" AND "long-term sustainability") only
    // contributes one distinct hit. Without dedup the cluster threshold
    // can be reached by a single sentence stacking overlapping regexes.
    const claimedSpans = [];
    function spanOverlaps(start, end) {
      for (const [s, e] of claimedSpans) {
        if (start < e && end > s) return true;
      }
      return false;
    }
    let distinctPhrasesHit = 0;
    for (const phrase of TIER3_PHRASES) {
      const regex = new RegExp(phrase.source, phrase.flags);
      const phraseSpans = [];
      let phraseMatch;
      while ((phraseMatch = regex.exec(text)) !== null) {
        const start = phraseMatch.index;
        const end = start + phraseMatch[0].length;
        if (!spanOverlaps(start, end)) {
          phraseSpans.push([start, end, phraseMatch[0]]);
        }
      }
      if (phraseSpans.length === 0) continue;
      for (const [s, e] of phraseSpans) claimedSpans.push([s, e]);
      distinctPhrasesHit++;
      if (phraseSpans.length >= 2) {
        issues.push({
          type: 'tier3-phrase',
          text: `"${phraseSpans[0][2].toLowerCase()}" x${phraseSpans.length}`,
          severity: 'medium',
          suggestion: `Boilerplate phrase repeated ${phraseSpans.length}× — replace at least one with specifics`,
        });
      }
    }
    if (distinctPhrasesHit >= 3) {
      issues.push({
        type: 'tier3-phrase-cluster',
        text: `${distinctPhrasesHit} distinct boilerplate phrases`,
        severity: 'high',
        suggestion: 'Several stock crypto/web3 phrases stacked in one piece. Rewrite around one specific claim or observation.',
      });
    }

    // ── Hashtag stuffing ─────────────────────────────────────────
    // 6+ hashtags in a single post is rare for thoughtful humans and
    // near-universal for LLM-generated social posts. Counted globally,
    // not per-paragraph, since the trailing hashtag block is the shape
    // we care about.
    // Match #tag at start of text or after any non-word char (whitespace,
    // punctuation, line breaks). URL fragments are already excluded
    // because the char immediately before `#` in a URL path is always
    // a word char (e.g. `example.com/page#section` — `e` before `#`).
    // Earlier char class `[\s\\]` had a literal backslash and silently
    // missed hashtags after sentence punctuation; an interim `[\s]` fix
    // on origin only caught whitespace-preceded tags.
    // Code is masked and non-tag `#` forms are subtracted first — see maskCode
    // and isSocialTag. Without them a changelog paragraph citing six issue
    // numbers, or a palette listing six hex colours, scored as a tag block.
    const hashtagMatches = [...maskCode(text).matchAll(/(?:^|\W)#(\w[\w-]*)/g)]
      .filter((m) => isSocialTag(m[1]));
    if (hashtagMatches.length >= 6) {
      issues.push({
        type: 'hashtag-stuff',
        text: `${hashtagMatches.length} hashtags`,
        severity: 'medium',
        suggestion: 'Cut to 2-3 specific tags or none. Long hashtag blocks read as bot output.',
      });
    }

    // ── Bullet list of bare noun phrases ─────────────────────────
    // ≥5 consecutive bullet items that are short (≤6 words) and contain
    // no finite-verb / modal token. Catches the "Stable mining efficiency
    // / Reliable pool connectivity / Optimized RandomX performance ..."
    // shape LLMs default to. Markdown bullets, escaped Markdown bullets,
    // unicode bullets, and dashes are all matched. Numbered lists are
    // excluded — those have a separate "numbered list inflation" rule.
    //
    // Note: verbRe covers auxiliaries and modals ("was", "will", "can",
    // etc.). Regular past-tense verbs ("fixed", "removed") are not
    // matched here; instead, the ≤6-word length gate excludes most
    // real-world changelog lines, which tend to read "fixed the X that
    // was doing Y" (>6 words). Short two-word action items ("* fixed
    // bug") would pass both gates — an acceptable trade-off to avoid
    // false-negative risk from adjectives ending in -ed ("skilled",
    // "advanced") that share the same surface form.
    const lines = text.split(/\r?\n/);
    const bulletRe = /^\s*(?:\*|-|•|\+)\s+(.+)$/;
    const verbRe = /\b(?:is|are|was|were|has|have|had|will|would|should|must|do|does|did|can|could|may|might|am|been|being)\b/i;
    const fenceRe = /^\s*(?:```|~~~)/;
    let run = [];
    let blankStreak = 0;
    let inFence = false;
    function flushRun() {
      if (run.length >= 5) {
        const bareNP = run.filter((it) => {
          const wc = (it.match(/\S+/g) || []).length;
          return wc > 0 && wc <= 6 && !verbRe.test(it);
        });
        if (bareNP.length >= 5 && bareNP.length / run.length >= 0.75) {
          issues.push({
            type: 'bullet-np-list',
            text: `${run.length}-item bullet list of bare noun phrases`,
            severity: 'high',
            suggestion: 'Convert to a prose paragraph or merge items. Long lists of bare adj+noun pairs read as AI scaffolding.',
          });
        }
      }
      run = [];
      blankStreak = 0;
    }
    for (const line of lines) {
      if (fenceRe.test(line)) {
        // Code-fence toggle. Bullets inside fences are CLI flag docs or
        // option dumps, not prose AI scaffolding — flush any prose run
        // we were tracking and skip until the fence closes.
        flushRun();
        inFence = !inFence;
        continue;
      }
      if (inFence) continue;
      const m = line.match(bulletRe);
      if (m) {
        run.push(m[1].trim());
        blankStreak = 0;
      } else if (line.trim() === '') {
        // A single blank line inside a list is normal Markdown spacing;
        // two or more blank lines break the run, since visually-disjoint
        // bullet sections shouldn't merge into one logical list.
        blankStreak++;
        if (blankStreak >= 2) flushRun();
      } else {
        flushRun();
      }
    }
    flushRun();

    // NOTE: "Wall-of-text replies" (SKILL.md) is deliberately NOT a
    // detector rule here. A first pass tried "reply-length text, >=4
    // sentences, zero newlines" as a structural gate — it broke the
    // "repeated Tier 1 phrase does not inflate score linearly" fixture
    // and, on reflection, would fire on any ordinary short paragraph
    // (a blog intro, a single-paragraph email) since "one paragraph with
    // no internal line break" is simply what continuous prose looks
    // like, not an AI-specific shape. The tell in SKILL.md depends on
    // knowing the text is conversational-reply register in the first
    // place, which the engine can't reliably infer from the bytes alone
    // (CONTRIBUTING.md: "a signal that fires on most normal prose is not
    // worth adding"). Left as an LLM-judgment rule; see CATEGORIES.md §C.

    // Confidence calibration is only flagged when it stacks (3+ instances).
    // Gating happens pre-dedup on raw match count, since that signals actual
    // stacking, not just vocabulary use.
    const confIssues = matchPatterns(text, CONFIDENCE_CALIBRATION, 'confidence-calibration', 'low');
    if (confIssues.length >= 3) issues.push(...confIssues);

    // ── 22. Em dash frequency ────────────────────────────────────
    // Match real em dashes, plus `--` only when surrounded by whitespace on at
    // least one side (skips CLI flags like --save-dev and YAML `---` blocks).
    // Separator-position em dashes (SEPARATOR_DASH_RE above) are excluded
    // from the rate. The list marker is required on purpose: a line-initial
    // "**Bold lead** — full sentence" outside a list is itself an AI tell
    // and still counts, as does a mid-sentence "**bold** — like this"
    // splice. Em dash only — the `--` substitute is never carved out.
    const rawEmDashCount = (text.match(/—|(?<=\s)--(?=\s|$)|(?<=^|\s)--(?=\s)/gm) || []).length;
    const separatorDashCount = (text.match(SEPARATOR_DASH_RE) || []).length
      + (text.match(VERSION_HEADING_DASH_RE) || []).length;
    const emDashCount = rawEmDashCount - separatorDashCount;
    const emDashRate = emDashCount / (wordCount / 1000);
    if (emDashRate > 1) {
      issues.push({
        type: 'em-dash',
        text: `${emDashCount} em dashes in ${wordCount} words`,
        severity: 'medium',
        suggestion: 'Replace with commas, periods, or rewrite',
      });
    }

    // ── 23. Sentence length uniformity ───────────────────────────
    if (sentences.length >= 5) {
      const lengths = sentences.map(s => countWords(s));
      const avg = lengths.reduce((a, b) => a + b, 0) / lengths.length;
      const variance = lengths.reduce((sum, l) => sum + Math.pow(l - avg, 2), 0) / lengths.length;
      const stdDev = Math.sqrt(variance);
      const cv = avg > 0 ? stdDev / avg : 0;

      if (cv < 0.25 && avg > 10) {
        issues.push({
          type: 'uniformity',
          text: `Sentence lengths cluster around ${Math.round(avg)} words (low variation)`,
          severity: 'medium',
          suggestion: 'Mix short punchy sentences with longer flowing ones',
        });
      }
    }

    // ── Type-token ratio (stylometric — vocabulary diversity) ────
    // TTR = distinct word types / total tokens. Human prose at 200+
    // words typically sits around 0.50–0.65 for English; AI prose
    // tends flatter (0.55–0.75 looks normal, but the lower end of the
    // *too-flat* tail at >=200 words is where the signal lives — too
    // FEW unique words for the length). This is the simplest of the
    // four stylometric signals identified in the May 2026 detection-
    // research review (docs/competitive/detection-research.md): no
    // POS tagger required, no model, pure JS.
    //
    // Threshold tuning: flag only when the sample is large enough
    // that low TTR is meaningfully suspicious (>=200 tokens) AND TTR
    // is below 0.40 (very vocabulary-poor). Conservative on purpose;
    // false positives on short or topic-narrow human prose are easy
    // to trigger and would drown out other signals. The detector-
    // research lens flagged TTR as one of four stylometric add-ons;
    // POS-bigram log-odds, function-word z-scores, and sentence-
    // length burstiness are still TODO.
    if (tokens.length >= 200) {
      const unique = new Set(tokens).size;
      const ttr = unique / tokens.length;
      if (ttr < 0.4) {
        issues.push({
          type: 'low-ttr',
          text: `Vocabulary diversity ${(ttr * 100).toFixed(1)}% (${unique} unique / ${tokens.length} tokens)`,
          severity: 'low',
          suggestion: 'Text reuses a narrow word set. Vary nouns and verbs deliberately, or check if the topic genuinely warrants the repetition.',
        });
      }
    }

    // ── 24. Paragraph length uniformity ──────────────────────────
    if (paragraphs.length >= 4) {
      const paraLengths = paragraphs.map(p => getSentences(p).length);
      const avg = paraLengths.reduce((a, b) => a + b, 0) / paraLengths.length;
      const allSimilar = paraLengths.every(l => Math.abs(l - avg) <= 1);
      if (allSimilar && avg >= 3) {
        issues.push({
          type: 'uniformity',
          text: `All paragraphs are ~${Math.round(avg)} sentences`,
          severity: 'low',
          suggestion: 'Vary paragraph length deliberately',
        });
      }
    }

    // ── 25. Bold overuse ─────────────────────────────────────────
    const boldMatches = text.match(/\*\*[^*]+\*\*/g) || [];
    if (boldMatches.length > 3) {
      issues.push({
        type: 'formatting',
        text: `${boldMatches.length} bold phrases`,
        severity: 'medium',
        suggestion: 'Strip bold from most; restructure to lead with key info',
      });
    }

    // ── Score from the deduped issue list ───────────────────────
    // Previously rawScore was accumulated inline per pattern hit, so
    // repeated hits of the same phrase (or overlapping matches) inflated
    // the score while the displayed issue list was deduplicated. That
    // produced the UX regression where a "heavy AI patterns" label sat
    // above a list of two items. Now the dedup runs first, then each
    // distinct issue contributes its category weight — so the number
    // reflects the same signals the user actually sees.
    const deduped = deduplicateIssues(issues);
    for (const issue of deduped) {
      rawScore += ISSUE_WEIGHTS[issue.type] ?? 2;
    }

    // Scale by text length: longer text gets more chances to trigger.
    const lengthFactor = Math.max(1, Math.log2(wordCount / 50));
    const normalizedScore = Math.min(100, Math.round(rawScore / lengthFactor));

    const label = getLabel(normalizedScore);

    // ── Sentence-region smoothing (HMM-style without an HMM) ─────────
    // Map each text-bearing issue back to its sentence indexes, then
    // merge adjacent flagged sentences into contiguous regions for UI
    // highlighting. Borrowed from GPTZero's sentence-highlighting model
    // — gives users "this paragraph is AI" rather than scattered hits.
    const regions = buildSentenceRegions(text, deduped, sourceMode === 'rendered-markdown');

    // Stats derived from the same deduped list so tier counts + patternCount
    // sum to `deduped.length`. Previously patternCount subtracted
    // `tier2Clusters` (a paragraph count) which produced inconsistent totals.
    const tier1Count = deduped.filter((i) => i.type === 'tier1').length;
    const tier2Count = deduped.filter((i) => i.type === 'tier2').length;
    const tier3Count = deduped.filter((i) => i.type === 'tier3').length;

    // ── Trinary classification (GPTZero-shaped) ──────────────────────
    // Decouples confidence from AI-proportion. Maps the 0-100 score plus
    // structural signals into HUMAN_ONLY / MIXED / AI_ONLY with a
    // confidence band. Thresholds are FN-biased: ambiguity routes to
    // MIXED, never AI_ONLY. Quote from GPTZero's design principle:
    // "biases the detector to prefer making less-harmful false-negative
    // errors over false-positive errors."
    // Dense-AI-vocab trifecta: ≥5 distinct tier1 hits + ≥2 tier2 cluster
    // paragraphs + ≥1 transition phrase, AND ≥150 words. Catches
    // saturated ChatGPT prose without firing on dense-jargon human
    // technical writing where the tier1 vocabulary (robust,
    // comprehensive, leverage, ecosystem) legitimately overlaps with
    // systems-programming idiom. Word-count gate prevents short ESL or
    // contrived adversarial sentences from tripping the corroborator.
    const tier1Distinct = new Set(deduped.filter((i) => i.type === 'tier1').map((i) => (i.text || '').toLowerCase())).size;
    const hasTier2Cluster = tier2Clusters >= 2;
    const hasTransition = deduped.some((i) => i.type === 'transition');
    const denseAIVocab = wordCount >= 150 && tier1Distinct >= 5 && hasTier2Cluster && hasTransition;

    const trinary = classifyTrinary({
      score: normalizedScore,
      issues: deduped,
      regions,
      normFlags: norm.flags,
      wordCount,
      denseAIVocab,
    });

    return {
      // Legacy fields preserved for existing callers.
      score: normalizedScore,
      label,
      issues: deduped,
      stats: {
        wordCount,
        tier1Count,
        tier2Count,
        tier2Clusters,
        tier3Count,
        tier3Flags,
        patternCount: deduped.length - tier1Count - tier2Count - tier3Count,
        contextMode,
        contextModeFallback,
        sourceMode,
        sourceModeFallback,
        maskedFrontmatter,
        maskedHtmlComments,
        normalization: norm.flags,
        quotedLines,
        unmappedHighlights: regions._unmapped ?? 0,
        denseAIVocab,
        tier1Distinct,
      },
      // Trinary API — shape mirrors GPTZero so integrators can swap.
      document_classification: trinary.classification,
      class_probabilities: trinary.probabilities,
      confidence_category: trinary.confidence,
      highlight_sentence_for_ai: regions,
    };
  }

  // ═══ Sentence regions + trinary classifier ═════════════════════════

  function buildSentenceRegions(text, issues, trimBoundaryWhitespace = false) {
    // Split text into sentences with source offsets preserved so the UI
    // can highlight spans accurately. Sentence boundaries are coarse
    // (.!?) — fine for highlighting, not for linguistic correctness.
    const sentences = [];
    const sentenceRe = /[^.!?]+[.!?]+|\S[^.!?]*$/g;
    let m;
    while ((m = sentenceRe.exec(text)) !== null) {
      const sentenceText = m[0].trim();
      if (sentenceText.length < 4) continue;
      let start = m.index;
      let end = m.index + m[0].length;
      if (trimBoundaryWhitespace) {
        start += m[0].search(/\S/);
        end -= m[0].match(/\s*$/)[0].length;
      }
      sentences.push({ start, end, text: sentenceText });
    }
    if (sentences.length === 0) return [];

    // Map issue.text back to sentence indexes via substring search. Two
    // kinds of issue stay out of the AI-highlight regions. Summary signals
    // like "Punctuation density uniform across paragraphs" have no sentence
    // anchor — they contribute to the document-level signal but not to
    // highlights. Zero-weight style copyedits (unnecessary-hyphenation) do
    // have an anchor, but they are P2 grammar cleanup rather than evidence
    // of machine authorship, so they belong in issues[] and nowhere near a
    // field reserved for AI sentence highlights.
    // Filter by issue TYPE not text-regex: text-based filtering used to
    // drop legitimate phrase issues containing "across" / "density".
    const NON_HIGHLIGHT_TYPES = new Set([
      'punct-distribution',
      'cross-para-burstiness',
      'fnword-trigram-entropy',
      'smart-punct-signature',
      'normalization-flag',
      'uniformity',
      'em-dash',
      'formatting',
      'tier3',
      'tier3-phrase',
      'tier3-phrase-cluster',
      'hashtag-stuff',
      'bullet-np-list',
      'unnecessary-hyphenation',
    ]);
    const hits = sentences.map(() => ({ count: 0, weight: 0 }));
    const lowerText = text.toLowerCase();
    let unmappedHighlights = 0;
    for (const issue of issues) {
      if (!issue.text || issue.text.length > 200) continue;
      if (NON_HIGHLIGHT_TYPES.has(issue.type)) continue;
      const needle = issue.text.toLowerCase();
      let idx = 0;
      let matched = false;
      while ((idx = lowerText.indexOf(needle, idx)) !== -1) {
        matched = true;
        for (let i = 0; i < sentences.length; i++) {
          if (idx >= sentences[i].start && idx < sentences[i].end) {
            hits[i].count++;
            hits[i].weight += ISSUE_WEIGHTS[issue.type] ?? 2;
            break;
          }
        }
        idx += needle.length;
      }
      if (!matched) unmappedHighlights++;
    }

    // Window-merge contiguous flagged sentences. Allow 1 unflagged
    // sentence gap between two flagged ones (the "smoothing" — keeps
    // a single boring sentence from breaking what's clearly an AI
    // passage). A sentence is "flagged" if it has ≥1 hit.
    const regions = [];
    let cur = null;
    for (let i = 0; i < sentences.length; i++) {
      if (hits[i].count > 0) {
        if (cur === null) {
          cur = { startSentence: i, endSentence: i, start: sentences[i].start, end: sentences[i].end, hitCount: hits[i].count, weight: hits[i].weight };
        } else {
          cur.endSentence = i;
          cur.end = sentences[i].end;
          cur.hitCount += hits[i].count;
          cur.weight += hits[i].weight;
        }
      } else if (cur !== null) {
        // Allow one-sentence gap.
        const next = hits[i + 1];
        if (next && next.count > 0) {
          cur.endSentence = i;
          cur.end = sentences[i].end;
          continue;
        }
        regions.push(finalizeRegion(cur));
        cur = null;
      }
    }
    if (cur !== null) regions.push(finalizeRegion(cur));
    // Expose the unmapped-highlight count via a non-enumerable property
    // so the array length still reads naturally for consumers; the
    // analyzer pulls it into stats.unmappedHighlights for diagnostics.
    Object.defineProperty(regions, '_unmapped', { value: unmappedHighlights, enumerable: false });
    return regions;
  }

  function finalizeRegion(r) {
    // Map cumulative weight inside the region to a 0-1 score. Cap at 20
    // weight = 1.0 (matches Heavy threshold density).
    const score = Math.min(1, r.weight / 20);
    return {
      startSentence: r.startSentence,
      endSentence: r.endSentence,
      start: r.start,
      end: r.end,
      hitCount: r.hitCount,
      score: Math.round(score * 100) / 100,
    };
  }

  // FN-biased: false positives damage trust more than false negatives,
  // so MIXED is wide and AI_ONLY requires multiple signals. Quote from
  // GPTZero: "biases the detector to prefer making less-harmful
  // false-negative errors over false-positive errors."
  function classifyTrinary({ score, issues, regions, normFlags, wordCount, denseAIVocab }) {
    // Strong corroborators — each is near-dispositive on its own:
    //   - cutoff-disclaimer (LLM self-identifies as an AI)
    //   - reasoning-artifact + chatbot-artifact co-occurrence
    //   - normalization-flag at threshold (≥2 ZWSP or homoglyphs).
    //     Threshold parity prevents a single stray ZWSP in copy-paste
    //     from Word/Notion from flipping to AI_ONLY at score 0.
    //   - denseAIVocab: ≥4 distinct tier1 hits AND ≥1 tier2 cluster AND
    //     ≥1 transition phrase — the trifecta that saturated ChatGPT
    //     prose triggers without needing whitelisted stylometric hits.
    const hasCutoff = issues.some((i) => i.type === 'cutoff-disclaimer');
    const hasNormFlag = normFlags.zeroWidth >= 2 || normFlags.homoglyph >= 2;
    const hasReasoning = issues.some((i) => i.type === 'reasoning-artifact');
    const hasChatbot = issues.some((i) => i.type === 'chatbot');
    const strongCorrob =
      (hasCutoff ? 1 : 0) +
      (hasNormFlag ? 1 : 0) +
      (hasReasoning && hasChatbot ? 1 : 0) +
      (denseAIVocab ? 1 : 0);

    // Weak (stylometric) corroborators — suggestive on their own,
    // dispositive in combination. Smart-punct-signature matches
    // Word-edited human prose so doesn't count without other support.
    const stylometricHits = ['punct-distribution', 'cross-para-burstiness', 'fnword-trigram-entropy']
      .filter((t) => issues.some((i) => i.type === t)).length;
    const hasSmartPunct = issues.some((i) => i.type === 'smart-punct-signature');
    const weakCorrob = (stylometricHits >= 2 ? 1 : 0) + (hasSmartPunct ? 1 : 0);

    // Thresholds:
    //   score < 15 with no strong → HUMAN_ONLY
    //   strong ≥ 1 OR score ≥ 70 → AI_ONLY (lowered from 80; high
    //     density of AI vocab is sufficient evidence)
    //   score ≥ 40 with any corroborator → AI_ONLY
    //   everything else with score ≥ 15 → MIXED
    const totalCorrob = strongCorrob + weakCorrob;
    let classification;
    if (score < 15 && strongCorrob === 0) classification = 'HUMAN_ONLY';
    else if (strongCorrob >= 1 || score >= 70) classification = 'AI_ONLY';
    else if (score >= 40 && totalCorrob >= 1) classification = 'AI_ONLY';
    else classification = 'MIXED';

    // Humanizer-flag escalation: presence of bypass-trick chars is
    // adversarial signal. If a normalization-flag fired we already
    // counted it in strongCorrob → AI_ONLY. Confidence also gets a
    // floor of 'medium' in that case (an adversary actively evading
    // detection should never read as low-confidence noise).

    // Soft probability distribution. Not calibrated against a labeled
    // corpus yet (TODO when corpus exists — see roadmap.md). Largest
    // class is computed as `1 - others` after rounding to guarantee
    // sum=1 exactly. Sub-1% drift would otherwise hide in toFixed.
    const aiSoft = Math.min(0.97, score / 100 + totalCorrob * 0.06 + strongCorrob * 0.08);
    let p;
    if (classification === 'HUMAN_ONLY') p = { human: Math.max(0.6, 1 - aiSoft), mixed: Math.min(0.35, aiSoft * 0.8), ai: Math.min(0.1, aiSoft * 0.3) };
    else if (classification === 'AI_ONLY') p = { human: Math.max(0.02, 1 - aiSoft - 0.05), mixed: 0.1, ai: aiSoft };
    else p = { human: Math.max(0.15, 0.6 - aiSoft * 0.5), mixed: 0.5, ai: aiSoft * 0.7 };
    const rawSum = p.human + p.mixed + p.ai;
    p.human = +(p.human / rawSum).toFixed(3);
    p.mixed = +(p.mixed / rawSum).toFixed(3);
    // Assign ai as the remainder so the three values sum to exactly 1.
    // Clamp to >= 0 in case rounding pushes human+mixed above 1 (the
    // remainder would otherwise show as -0 or -0.001 — surfaces as a
    // negative percentage in any UI doing Math.round(p.ai * 100)).
    p.ai = Math.max(0, +(1 - p.human - p.mixed).toFixed(3));
    const probabilities = p;

    // Confidence band:
    //   high   — strongCorrob ≥ 2, OR cutoff-disclaimer, OR score < 8 (clean long doc)
    //   medium — strongCorrob ≥ 1, OR score ≥ 45 with weak corroborator, OR score < 20
    //   low    — everything else
    let confidence;
    if (strongCorrob >= 2 || hasCutoff || (score < 8 && wordCount >= 100)) confidence = 'high';
    else if (strongCorrob >= 1 || (score >= 45 && weakCorrob >= 1) || score < 20) confidence = 'medium';
    else confidence = 'low';

    return { classification, probabilities, confidence };
  }

  function getLabel(score) {
    if (score === 0) return 'Clean';
    if (score <= 15) return 'Minimal AI signals';
    if (score <= 35) return 'Some AI patterns';
    if (score <= 60) return 'Moderate AI signals';
    if (score <= 80) return 'Strong AI signals';
    return 'Heavy AI patterns';
  }

  function getColor(score) {
    if (score <= 15) return '#44bb66';
    if (score <= 35) return '#88bb44';
    if (score <= 60) return '#ddaa00';
    if (score <= 80) return '#ff8833';
    return '#ff4444';
  }

  function deduplicateIssues(issues) {
    const seen = new Set();
    return issues.filter(issue => {
      const key = `${issue.type}:${issue.text.toLowerCase()}`;
      if (seen.has(key)) return false;
      seen.add(key);
      return true;
    });
  }

  // ─── Severity labels ──────────────────────────────────────────
  const SEVERITY_LABELS = {
    critical: 'P0',
    high: 'P1',
    medium: 'P2',
    low: 'P3',
  };

  const TYPE_LABELS = {
    'tier1': 'AI vocabulary',
    'tier1-clarity': 'Wordiness',
    'tier2': 'Word cluster',
    'tier3': 'Overused word',
    'transition': 'AI transition',
    'chatbot': 'Chatbot artifact',
    'sycophantic': 'Sycophantic tone',
    'filler': 'Filler phrase',
    'generic-conclusion': 'Generic conclusion',
    'lets-construction': '"Let\'s" opener',
    'reasoning-artifact': 'Reasoning artifact',
    'acknowledgment-loop': 'Acknowledgment loop',
    'significance-inflation': 'Significance inflation',
    'vague-attribution': 'Vague attribution',
    'hollow-intensifier': 'Hollow intensifier',
    'emotional-flatline': 'Emotional flatline',
    'lingering-attention': 'Lingering-attention claim',
    'novelty-inflation': 'Novelty inflation',
    'cutoff-disclaimer': 'Cutoff disclaimer',
    'template-phrase': 'Template phrase',
    'false-concession': 'False concession',
    'rhetorical-question': 'Rhetorical question',
    'confidence-calibration': 'Confidence stacking',
    'em-dash': 'Em dash overuse',
    'uniformity': 'Rhythm uniformity',
    'formatting': 'Formatting',
    'tier3-phrase': 'Boilerplate phrase',
    'tier3-phrase-cluster': 'Boilerplate cluster',
    'hashtag-stuff': 'Hashtag stuffing',
    'bullet-np-list': 'Bullet-NP list',
    'hedge-stack': 'Hedge-stacked prediction',
    'future-narrative': 'Generic future narrative',
    'real-actual-inflation': '"Real/actual" inflation',
    'social-cta-closer': 'Engagement-bait closer',
    'formulaic-opener': 'Formulaic opener',
    'speculative-opener': 'Speculative scenario opener',
    'launch-intro': 'Launch-copy introduction',
    'crowd-contrast': 'Dramatized crowd contrast',
    'fake-casual-prop': 'Fake-casual prop',
    'title-case-header': 'Title Case header',
    'parenthetical-hedge': 'Parenthetical hedge',
    'smart-punct-signature': 'Smart-punct signature',
    'punct-distribution': 'Punctuation distribution',
    'fnword-trigram-entropy': 'Grammar repetition',
    'cross-para-burstiness': 'Cross-paragraph rhythm',
    'normalization-flag': 'Bypass-trick chars',
    'low-ttr': 'Low vocabulary diversity',
    'ai-placeholder': 'Unfilled placeholder',
    'ai-citation-markup': 'Chatbot citation markup leak',
    'ai-utm-source': 'AI-tool URL parameter',
    'unnecessary-hyphenation': 'Unnecessary hyphenation',
    'performed-insight': 'Performed-insight phrase',
    'negation-chain': 'Negation chain',
    'dev-blog-boilerplate': 'Dev-blog boilerplate',
  };

  return {
    analyzeText,
    normalizeText,
    getLabel,
    getColor,
    SEVERITY_LABELS,
    TYPE_LABELS,
  };
})();

if (typeof module !== 'undefined' && module.exports) {
  module.exports = AIDetector;
}

SHA-256: 4fd45ea96359ca194de06ec39b009f20d86ffdc426a9997bf2084a71e8467db9