← Files Avoid AI WritingARCHIVED FILE
skills/preservation-verifier/scripts/patterns.js
127 KB · Oct 2, 2026 · 00:32 UTC
/**
* Avoid AI Writing — detection engine (canonical source of truth)
* Implements regex, structural, and stylometric pattern detection. This repo's SKILL.md
* catalogs the human-editable pattern rules; this engine is the executable
* expression of the regex-detectable subset and extends it with stylometric and
* AI-tool-fingerprint detectors that don't make sense as skill prose
* (cross-paragraph burstiness, smart-punct signatures, function-word
* trigram entropy, low type-token ratio, AI-tool URL parameters,
* chatbot citation markup leaks, unfilled placeholders).
*
* Scoring model:
* Each category has a weight in the ISSUE_WEIGHTS table. Detection runs
* produce raw (possibly duplicate) issues which are then deduplicated by
* (type, text) pair. rawScore is the sum of category weights across the
* deduped list — so the number reflects the same distinct signals the
* user sees in the issue list.
*
* Weights are deliberately non-flat across severity tags. Cutoff
* disclaimers (10) and chatbot artifacts (8) weigh more than vague
* attributions (5), even though all three are tagged `critical`, because
* the skill treats them as stronger or weaker AI-origin signals.
*
* rawScore is then normalized to 0-100 via `log2(wordCount/50)` so longer
* texts don't accumulate unboundedly on the same density of patterns.
*/
const AIDetector = (() => {
// ═══ Tier 1 pre-pass: normalize bypass tricks ══════════════════════
//
// Humanizer tools and prompt-injection bypass techniques insert
// invisible / lookalike chars to defeat exact-string detectors. Strip
// them BEFORE pattern matching so "delve" with a Cyrillic 'е' still
// hits the Tier 1 list. Unicode ranges sourced from
// It-s-AI/llm-detection/detection/attacks/.
//
// Tracks what was stripped so the trinary classifier can use
// "normalization triggered" as a corroborating AI signal — humans don't
// paste ZWSPs into their own writing.
const CYRILLIC_LOOKALIKES = {
'а': 'a', 'е': 'e', 'о': 'o', 'р': 'p', 'с': 'c', 'х': 'x',
'у': 'y', 'к': 'k', 'м': 'm', 'н': 'h', 'в': 'b', 'т': 't',
'А': 'A', 'Е': 'E', 'О': 'O', 'Р': 'P', 'С': 'C', 'Х': 'X',
'У': 'Y', 'К': 'K', 'М': 'M', 'Н': 'H', 'В': 'B', 'Т': 'T',
};
const GREEK_LOOKALIKES = { 'ο': 'o', 'Ο': 'O', 'α': 'a', 'Α': 'A', 'ρ': 'p', 'Ρ': 'P' };
function normalizeText(text) {
const flags = { zeroWidth: 0, homoglyph: 0, roleplay: 0 };
let out = text;
// 1. Strip zero-width chars (ZWSP U+200B, ZWNJ U+200C, ZWJ U+200D,
// BOM U+FEFF, word joiner U+2060).
out = out.replace(/[-]/g, () => { flags.zeroWidth++; return ''; });
// 2. Swap Cyrillic / Greek Latin-lookalike chars back to Latin so
// pattern matching catches obfuscated tokens.
out = out.replace(/[Ѐ-ӿͰ-Ͽ]/g, (m) => {
const swap = CYRILLIC_LOOKALIKES[m] ?? GREEK_LOOKALIKES[m];
if (swap) { flags.homoglyph++; return swap; }
return m;
});
// 3. Strip *roleplay-action* markers — paired *...* containing an
// action verb (nods, sighs, laughs, smiles, etc.) anchored to
// the start of the inner phrase. This is the actual chat-model
// artifact shape. Markdown `**bold**` is rejected by the
// lookbehind/lookahead; legitimate multi-word `*italic*` is
// preserved because the verb whitelist is narrow.
const ROLEPLAY_VERBS = /^(?:nods|sighs|laughs|smiles|frowns|shrugs|grins|winks|chuckles|gasps|pauses|thinks|wonders|whispers|shouts|gestures|raises|leans|turns|looks|glances|smirks|blinks|nodding|sighing|laughing|smiling|thinking|gesturing)\b/i;
out = out.replace(/(?<!\*)\*([^*\n]{1,80}?)\*(?!\*)/gu, (m, inner) => {
if (ROLEPLAY_VERBS.test(inner)) { flags.roleplay++; return ''; }
return m;
});
return { text: out, flags };
}
// ─── Tier 1: Always flag ───────────────────────────────────────────
const TIER1 = {
'delve': 'explore, dig into, look at',
'tapestry': 'describe the actual complexity',
'paradigm': 'model, approach, framework',
'beacon': 'rewrite entirely',
'robust': 'strong, reliable, solid',
'comprehensive': 'thorough, complete, full',
'cutting-edge': 'latest, newest, advanced',
'pivotal': 'important, key, critical',
'meticulous': 'careful, detailed, precise',
'meticulously': 'carefully, precisely',
'seamless': 'smooth, easy, without friction',
'seamlessly': 'smoothly, easily',
'game-changer': 'describe what changed',
'game-changing': 'describe what changed',
'nestled': 'is located, sits',
'vibrant': 'describe what makes it active',
'thriving': 'growing, active',
'bustling': 'busy, active',
'intricate': 'complex, detailed',
'intricacies': 'complexities, details',
'ever-evolving': 'changing, growing',
'enduring': 'lasting, long-running',
'daunting': 'hard, difficult',
'holistic': 'complete, full, whole',
'holistically': 'completely, fully',
'actionable': 'practical, useful, concrete',
'impactful': 'effective, significant',
'learnings': 'lessons, findings, takeaways',
'synergy': 'describe the combined effect',
'synergies': 'describe the combined effect',
'interplay': 'relationship, connection',
'symphony': 'describe the coordination',
'embrace': 'adopt, accept, use',
};
// Multi-word tier 1 phrases
const TIER1_PHRASES = [
{ pattern: /\bdelve\s+into\b/gi, replace: 'explore, dig into' },
{ pattern: /\blandscape\b/gi, replace: 'field, space, industry', filter: true },
{ pattern: /\brealm\b/gi, replace: 'area, field, domain' },
{ pattern: /\btestament\s+to\b/gi, replace: 'shows, proves' },
{ pattern: /\bleverag(?:e|es|ing|ed)\b/gi, replace: 'use' },
{ pattern: /\bwatershed\s+moment\b/gi, replace: 'turning point, shift' },
{ pattern: /\bmarking\s+a\s+pivotal\s+moment\b/gi, replace: 'state what happened' },
{ pattern: /\bthe\s+future\s+looks\s+bright\b/gi, replace: 'cut or say something specific' },
{ pattern: /\bonly\s+time\s+will\s+tell\b/gi, replace: 'cut or say something specific' },
{ pattern: /\bdespite\s+challenges[^.]*continues?\s+to\s+thrive\b/gi, replace: 'name the challenge and response' },
{ pattern: /\bdeep\s+dive\b/gi, replace: 'look at, examine' },
{ pattern: /\bdive\s+into\b/gi, replace: 'look at, examine' },
{ pattern: /\bunpack(?:ing)?\b/gi, replace: 'explain, break down' },
{ pattern: /\bcomplexities\b/gi, replace: 'name the actual problems' },
{ pattern: /\bthought\s+leader(?:ship)?\b/gi, replace: 'expert, authority' },
{ pattern: /\bbest\s+practices\b/gi, replace: 'what works, proven methods' },
{ pattern: /\bat\s+its\s+core\b/gi, replace: 'cut, just state it' },
{ pattern: /\bin\s+order\s+to\b/gi, replace: 'to', clarity: true },
{ pattern: /\bdue\s+to\s+the\s+fact\s+that\b/gi, replace: 'because', clarity: true },
{ pattern: /\bserves\s+as\b/gi, replace: 'is', clarity: true },
{ pattern: /\bfeatures\b/gi, replace: 'has, includes', filter: true, clarity: true },
{ pattern: /\bboasts\b/gi, replace: 'has', clarity: true },
{ pattern: /\butiliz(?:e|es|ing|ed)\b/gi, replace: 'use', clarity: true },
{ pattern: /\bshowcas(?:e|es|ing|ed)\b/gi, replace: 'show, demonstrate' },
{ pattern: /\bembark(?:s|ing|ed)?\b/gi, replace: 'start, begin' },
{ pattern: /\bcommenc(?:e|es|ing|ed)\b/gi, replace: 'start, begin', clarity: true },
{ pattern: /\bascertain(?:s|ing|ed)?\b/gi, replace: 'find out, determine', clarity: true },
{ pattern: /\bendeavou?r(?:s|ing|ed)?\b/gi, replace: 'effort, attempt, try', clarity: true },
{ pattern: /\bunderscor(?:es|ing|ed)\b/gi, replace: 'highlights, shows' },
// Hyphen required. The unhyphenated "load bearing" is ordinary English —
// "the load bearing down on the bridge" — where `bearing` is a participle,
// not part of a compound modifier. The tell is always hyphenated.
//
// Construction carve-out: exempt attributive `load-bearing` before a literal
// structural noun, with one optional material/position adjective in between
// ("load-bearing structural wall"). The noun list is limited to commonly
// physical nouns; the abstract-capable ones most likely to carry the
// metaphor (structure, element, frame, foundation) are omitted so "the
// load-bearing structure of the argument" still fires. Some listed nouns
// (member, column, partition) can still be used metaphorically and are
// silently exempt — a recall loss in the safe direction, tracked in #56.
// Predicative use ("the wall is load-bearing") is NOT exempt: the tell
// lives in the subject, which a lookahead cannot reach. Also #56.
{ pattern: /\bload-bearing\b(?!\s+(?:(?:structural|exterior|interior|internal|external|concrete|steel|timber|wooden|brick|masonry|perimeter|basement|main|primary|existing|original)\s+)?(?:walls?|beams?|columns?|joists?|truss(?:es)?|members?|footings?|slabs?|studs?|partitions?|masonry|lintels?|piers?|rafters?|girders?|capacity|capacities)\b)/gi, replace: 'essential, critical, or say what breaks if you remove it' },
];
// ─── Tier 2: Flag in clusters (2+ per paragraph) ──────────────────
const TIER2 = {
'harness': 'use, take advantage of',
'navigate': 'work through, handle',
'navigating': 'working through, handling',
'foster': 'encourage, support, build',
'elevate': 'improve, raise, strengthen',
'unleash': 'release, enable, unlock',
'streamline': 'simplify, speed up',
'empower': 'enable, let, allow',
'bolster': 'support, strengthen',
'spearhead': 'lead, drive, run',
'resonate': 'connect with, appeal to',
'resonates': 'connects with, appeals to',
'revolutionize': 'change, transform',
'facilitate': 'enable, help, allow',
'facilitates': 'enables, helps, allows',
'underpin': 'support, form the basis of',
'nuanced': 'specific, subtle, detailed',
'crucial': 'important, key, necessary',
'multifaceted': 'describe the actual facets',
'ecosystem': 'system, community, network',
'myriad': 'many, numerous',
'plethora': 'many, a lot of',
'encompass': 'include, cover, span',
'catalyze': 'start, trigger, accelerate',
'reimagine': 'rethink, redesign, rebuild',
'galvanize': 'motivate, rally, push',
'augment': 'add to, expand, supplement',
'cultivate': 'build, develop, grow',
'illuminate': 'clarify, explain, show',
'elucidate': 'explain, clarify',
'juxtapose': 'compare, contrast',
'transformative': 'describe what changed',
'transformation': 'describe what changed',
'cornerstone': 'foundation, basis, key part',
'paramount': 'most important, top priority',
'poised': 'ready, set, about to',
'burgeoning': 'growing, emerging',
'nascent': 'new, early-stage',
'quintessential': 'typical, classic, defining',
'overarching': 'main, central, broad',
'quietly': 'cut, or name the concrete contrast',
'underpinning': 'basis, foundation',
'underpinnings': 'basis, foundations',
'paradigm-shifting': 'describe what shifted',
};
// Conditional Tier 2 entries: everyday words whose AI tell is a specific
// significance collocation, not the word itself. They join a paragraph's
// Tier 2 cluster only when the collocation matches — bare uses ("deeply
// nested JSON", "cares deeply") never count, because the base rate of
// these words in innocent prose is far higher than the rest of the table
// and an unconditional entry measurably flags clean human writing.
const TIER2_CONDITIONAL = [
{
word: 'deeply',
pattern: /\bdeeply\s+(?:integrated|committed|rooted|personal|human|flawed|resonant|transformative|interconnected|ingrained|embedded|meaningful)\b/i,
suggestion: 'cut, or name what specifically runs deep',
},
];
// ─── Tier 3: Flag by density ───────────────────────────────────────
const TIER3 = [
'significant', 'significantly', 'innovative', 'innovation',
'effective', 'effectively', 'dynamic', 'dynamics',
'scalable', 'scalability', 'compelling', 'unprecedented',
'exceptional', 'exceptionally', 'remarkable', 'remarkably',
'sophisticated', 'instrumental',
'world-class', 'state-of-the-art', 'best-in-class',
// `verbatim` is usually redundant with the verb it modifies ("copies X
// verbatim" = "copies X"). It has a genuine term-of-art sense in legal,
// research, and QA registers ("verbatim transcript"), so it lives at Tier 3:
// density-gated, it only fires on overuse, not on a single legitimate use.
'verbatim',
];
// Multi-word Tier 3 phrases. Density-gated like single Tier 3 words because
// these legitimately show up in human crypto/web3/dev writing. Threshold is
// intentionally lower than single-word Tier 3 (≥2 occurrences of the same
// phrase) — repeating the same multi-word boilerplate is a stronger AI tell
// than re-using "significant."
const TIER3_PHRASES = [
/\bemerging\s+(?:sector|space|category|industry)\b/gi,
/\bthe\s+integration\s+of\b/gi,
/\bthe\s+intersection\s+of\b/gi,
/\bcommunity-?driven\b/gi,
/\blong-?term\s+sustainability\b/gi,
/\buser\s+engagement\b/gi,
/\bdecentralized\s+compute\b/gi,
/\b(?:sustainable\s+)?reward\s+emissions?\b/gi,
/\btokenized\s+incentive\s+structures?\b/gi,
/\bdesigned\s+for\s+long-?term\b/gi,
];
// O(1) lookup from any token form (hyphenated or dashless) to its canonical
// Tier 3 word. Counting originally did nested-loop word matching which was
// O(tokens × TIER3) — slow on long pastes.
const TIER3_LOOKUP = new Map();
for (const word of TIER3) {
TIER3_LOOKUP.set(word, word);
const dashless = word.replace(/-/g, '');
if (dashless !== word) TIER3_LOOKUP.set(dashless, word);
}
// Per-category score weights. Applied to distinct (deduplicated) issues so
// the score reflects the same signals the user sees in the issue list.
// Non-uniform on purpose: critical rules like cutoff disclaimers (×10) and
// chatbot artifacts (×8) weigh more than vague attributions (×5), even
// though all three are tagged `critical`.
const ISSUE_WEIGHTS = {
tier1: 5,
// Wordiness, not an AI-frequency marker. Weighted like tier2 so a
// clarity fix cannot push a document toward an AI classification.
'tier1-clarity': 3,
tier2: 3,
tier3: 2,
transition: 2,
chatbot: 8,
sycophantic: 8,
filler: 2,
'generic-conclusion': 3,
'lets-construction': 2,
'reasoning-artifact': 6,
'acknowledgment-loop': 3,
'significance-inflation': 4,
'vague-attribution': 5,
'hollow-intensifier': 2,
'emotional-flatline': 2,
'lingering-attention': 3,
'novelty-inflation': 3,
'cutoff-disclaimer': 10,
'template-phrase': 3,
'false-concession': 2,
'rhetorical-question': 2,
'confidence-calibration': 2,
'em-dash': 4,
uniformity: 5,
formatting: 3,
'tier3-phrase': 3,
// Structural / cluster signals are deliberately weighted high. Unlike
// single-vocabulary hits they're near-dispositive on social-length
// posts (a 15-hashtag block, a 6-item bullet-NP list, three distinct
// crypto-shill phrases stacked) and would otherwise be suppressed by
// the log2(words/50) length divisor on short pastes.
'tier3-phrase-cluster': 12,
'hashtag-stuff': 12,
'bullet-np-list': 10,
'hedge-stack': 6,
'future-narrative': 12,
'real-actual-inflation': 5,
// Social endorsement / CTA closer. Weighted like formulaic-opener: a
// strong single-hit social tell that the length divisor would
// otherwise wash out on a short LinkedIn-length post.
'social-cta-closer': 8,
// Performed-insight tics: single hits are common in human essays, so
// weighted like tier2 vocabulary — density does the classifying.
'performed-insight': 3,
// Negation chains are a strong single-hit structural tell.
'negation-chain': 5,
'dev-blog-boilerplate': 3,
'formulaic-opener': 8,
// Speculative scenario opener ("Imagine a world where…"). Weighted like
// formulaic-opener: a single strong opener tell the length divisor would
// otherwise wash out on a short post.
'speculative-opener': 8,
// Launch-copy introduction ("Enter X.", "Meet X, your new...").
// Weighted like the other single-hit opener tells: strong on the
// short launch posts where it actually appears.
'launch-intro': 8,
// Dramatized crowd contrast. Gated hard on the dismissive verb, so
// a hit is meaningful, but the surface shares words with ordinary
// narrative — weighted below the opener tells on purpose.
'crowd-contrast': 6,
// Fake-casual props (stage directions, wink asides). Near-costume
// when present; same class as the opener tells on short posts.
'fake-casual-prop': 8,
'title-case-header': 4,
'parenthetical-hedge': 3,
'smart-punct-signature': 6,
'punct-distribution': 6,
'fnword-trigram-entropy': 5,
'cross-para-burstiness': 5,
'normalization-flag': 9,
// Vocabulary-diversity signal (type-token ratio). Weighted modestly
// because the threshold (>=200 tokens AND TTR<0.4) is conservative;
// it stacks with structural signals to push borderline scores up.
'low-ttr': 3,
// AI-tool fingerprints. Weighted higher than statistical patterns
// because each is a near-definitive single-hit signal — the AI tool
// literally left its mark in the text. citation-markup ranks highest
// (smoking gun: the literal internal markup of ChatGPT/Grok/etc.),
// UTM tracking second (auto-appended by the tool to URLs it writes),
// placeholders third (strong but humans use bracketed slots in
// templates legitimately and forget them — still a publishing bug
// but slightly less definitive AI evidence).
'ai-placeholder': 10,
'ai-citation-markup': 15,
'ai-utm-source': 12,
// Curated P2 copyedits, not authorship evidence. Keep them visible in the
// issue list without moving the AI score, label, probabilities, or
// trinary classification on short documents.
'unnecessary-hyphenation': 0,
};
// ─── Transition phrases ────────────────────────────────────────────
const TRANSITIONS = [
/\bmoreover\b/gi,
/\bfurthermore\b/gi,
/\badditionally\b/gi,
/\bin\s+today'?s\b/gi,
/\bin\s+an\s+era\s+where\b/gi,
/\bit'?s\s+worth\s+noting\s+that\b/gi,
/\bnotably\b/gi,
/\bin\s+conclusion\b/gi,
/\bin\s+summary\b/gi,
/\bto\s+summarize\b/gi,
/\bwhen\s+it\s+comes\s+to\b/gi,
/\bat\s+the\s+end\s+of\s+the\s+day\b/gi,
/\bthat\s+(?:being\s+)?said\b/gi,
];
// ─── Chatbot artifacts ─────────────────────────────────────────────
const CHATBOT_ARTIFACTS = [
/\bi\s+hope\s+this\s+helps\b/gi,
/\bcertainly!\b/gi,
/\babsolutely!\b/gi,
/\bgreat\s+question!\b/gi,
/\bexcellent\s+point!\b/gi,
/\bfeel\s+free\s+to\s+reach\s+out\b/gi,
/\blet\s+me\s+know\s+if\s+you\s+need\s+anything\b/gi,
/\bin\s+this\s+article,?\s+we\s+will\s+explore\b/gi,
/\blet'?s\s+dive\s+in!?\b/gi,
];
// ─── Sycophantic tone ──────────────────────────────────────────────
const SYCOPHANTIC = [
/\byou'?re\s+absolutely\s+right\b/gi,
/\bthat'?s\s+a\s+really\s+insightful\b/gi,
/\bthat'?s\s+a\s+great\s+question\b/gi,
/\bexcellent\s+question\b/gi,
];
// ─── Filler phrases ────────────────────────────────────────────────
const FILLERS = [
/\bit\s+is\s+important\s+to\s+note\s+that\b/gi,
/\bin\s+terms\s+of\b/gi,
/\bthe\s+reality\s+is\s+that\b/gi,
/\bit'?s\s+important\s+to\s+note\s+that\b/gi,
];
// ─── Generic conclusions ───────────────────────────────────────────
const GENERIC_CONCLUSIONS = [
/\bthe\s+future\s+looks\s+bright\b/gi,
/\bonly\s+time\s+will\s+tell\b/gi,
/\bone\s+thing\s+is\s+certain\b/gi,
/\bas\s+we\s+move\s+forward\b/gi,
];
// ─── "Let's" constructions ─────────────────────────────────────────
const LETS_PATTERNS = [
/\blet'?s\s+explore\b/gi,
/\blet'?s\s+take\s+a\s+look\b/gi,
/\blet'?s\s+break\s+this\s+down\b/gi,
/\blet'?s\s+examine\b/gi,
/\blet'?s\s+(?:consider|discuss|delve|unpack|walk\s+through)\b/gi,
];
// ─── Reasoning chain artifacts ─────────────────────────────────────
const REASONING_ARTIFACTS = [
/\blet\s+me\s+think\s+step\s+by\s+step\b/gi,
/\bbreaking\s+this\s+down\b/gi,
/\bto\s+approach\s+this\s+systematically\b/gi,
/\bhere'?s\s+my\s+thought\s+process\b/gi,
/\bfirst,?\s+let'?s\s+consider\b/gi,
/\bworking\s+through\s+this\s+logically\b/gi,
];
// ─── Acknowledgment loops ──────────────────────────────────────────
const ACKNOWLEDGMENT_LOOPS = [
/\byou'?re\s+asking\s+about\b/gi,
/\bthe\s+question\s+of\s+whether\b/gi,
/\bto\s+answer\s+your\s+question\b/gi,
];
// ─── Significance inflation ────────────────────────────────────────
const SIGNIFICANCE_INFLATION = [
/\bmarking\s+a\s+(?:pivotal|significant|important)\s+moment\b/gi,
/\ba\s+watershed\s+moment\s+for\b/gi,
/\bin\s+the\s+evolution\s+of\b/gi,
/\ba\s+(?:pivotal|defining)\s+moment\s+in\b/gi,
];
// ─── Vague attributions ────────────────────────────────────────────
const VAGUE_ATTRIBUTIONS = [
/\bexperts\s+(?:believe|say|suggest|agree)\b/gi,
/\bstudies\s+(?:show|suggest|indicate)\b/gi,
/\bresearch\s+(?:shows|suggests|indicates)\b/gi,
/\bindustry\s+leaders\s+(?:agree|believe|say)\b/gi,
];
// ─── Hollow intensifiers ──────────────────────────────────────────
const HOLLOW_INTENSIFIERS = [
/\bgenuine(?:ly)?\b/gi,
/\btruly\b/gi,
/\bquite\s+frankly\b/gi,
/\bto\s+be\s+honest\b/gi,
/\blet'?s\s+be\s+clear\b/gi,
];
// ─── Emotional flatline ────────────────────────────────────────────
// The "interesting (part|thing|aspect|piece)" family is matched in two
// shapes: (1) "the most interesting X" inline (the canonical AI list
// intro), and (2) bare "Interesting X:" used as a section-header opener,
// which is the section-break variant that slipped past v3.3.x.
const EMOTIONAL_FLATLINE = [
/\bwhat\s+surprised\s+me\s+most\b/gi,
/\bi\s+was\s+fascinated\s+to\b/gi,
/\bwhat\s+struck\s+me\s+was\b/gi,
/\bi\s+was\s+excited\s+to\s+learn\b/gi,
/\bthe\s+most\s+interesting\s+(?:part|thing|aspect|piece)\b/gi,
// Multiline flag (/m) so `^` matches at every line start, including
// position 0 of a pasted text that has no leading newline. The earlier
// `(?:^|\n)` form silently missed bare openers at the very start of
// input — caught by silent-failure audit 2026-05-16.
/^\s*interesting\s+(?:part|thing|aspect|piece)(?:\s+of\s+(?:the\s+)?\w+)?\s*:/gim,
];
// ─── Lingering-attention claims ────────────────────────────────────
// The share-post opener that claims duration of attention instead of
// saying anything about the thing ("the line I keep coming back to").
//
// Precision note: the bare verb phrase "I keep coming back to X" is NOT
// matched on its own, because it is legitimate whenever a reason follows
// ("I keep coming back to Hirschman because it predicts who quits"), and
// the reason clause is not reliably detectable by regex. Only the
// noun-anchored frame ("the line/quote/bit ... I keep coming back to")
// fires, which is the shape that introduces a subject rather than
// asserting something about it. The bare form stays a skill-prose
// judgment call. See SKILL.md §Lingering-attention claims carve-out.
const LINGERING_ATTENTION = [
// "the line I keep coming back to", "the one quote that I keep coming back to"
// Noun-anchored frame only. "can't stop thinking about" is deliberately
// left to its own pattern below rather than folded into this alternation,
// so "the line I can't stop thinking about" scores once, not twice.
/\b(?:the|that|this)\s+(?:one\s+)?(?:line|quote|bit|part|idea|point|framing|comment|thing)\s+(?:that\s+)?i\s+keep\s+(?:coming\s+back\s+to|thinking\s+about)\b/gi,
/\bi\s+can'?t\s+stop\s+thinking\s+about\b/gi,
/\bstill\s+thinking\s+about\s+(?:this|that)\s+one\b/gi,
/\b(?:been|be)\s+rattling\s+around\s+(?:in\s+)?my\s+(?:head|brain)\b/gi,
/\bi'?ve\s+been\s+chewing\s+on\s+(?:this|that)\b/gi,
];
// ─── Novelty inflation ─────────────────────────────────────────────
const NOVELTY_INFLATION = [
/\bthe\s+failure\s+mode\s+nobody'?s?\s+naming\b/gi,
/\ba\s+problem\s+nobody\s+talks\s+about\b/gi,
/\bthe\s+insight\s+everyone'?s?\s+missing\b/gi,
/\bwhat\s+nobody\s+tells\s+you\b/gi,
];
// ─── Cutoff disclaimers ────────────────────────────────────────────
// Includes the canonical LLM self-identification phrases — these are
// near-dispositive on their own (real humans don't write "as an AI
// language model" in first person). Patterns cover the major model
// families' default disclaimer language.
const CUTOFF_DISCLAIMERS = [
/\bas\s+of\s+my\s+last\s+update\b/gi,
/\bas\s+of\s+my\s+(?:knowledge\s+)?(?:cut-?off|last\s+training)\b/gi,
/\bi\s+don'?t\s+have\s+access\s+to\s+real-?time\s+(?:data|information)\b/gi,
/\bbased\s+on\s+available\s+information\b/gi,
/\bas\s+an?\s+(?:ai|artificial\s+intelligence|large\s+language|ai\s+language)\s+(?:language\s+)?model\b/gi,
/\bi\s+(?:am|'m)\s+an?\s+(?:ai|artificial\s+intelligence|large\s+language)\s+(?:assistant|model)?\b/gi,
/\bi\s+cannot\s+(?:provide|give|offer)\s+(?:legal|medical|financial|professional)\s+advice\b/gi,
/\bmy\s+training\s+data\s+(?:only\s+)?(?:goes\s+up\s+to|extends\s+to|ends\s+(?:in|at))\b/gi,
];
// ─── AI-tool fingerprints ──────────────────────────────────────────
// Three near-definitive AI-origin signals adapted from
// Aboudjem/humanizer-skill P33-P35 (see docs/competitive/audits/
// 2026-05-17-aboudjem-humanizer-skill.md). Unlike the statistical
// patterns above, single hit on any of these is strong evidence —
// the AI tool literally left its fingerprint in the text.
// Unfilled slot-fill placeholders. Catches the canonical "[Your Name]"
// family plus dated stubs and HTML/MD comments with placeholder verbs.
const AI_PLACEHOLDERS = [
// Directive stubs ("[Your Name]", "[INSERT SOURCE URL]",
// "[Describe the specific section]") — verb-led, the user was
// told to fill in.
/\[(?:Your|Insert|Add|Enter|Describe|Specify|Choose|Pick)[^\]\n]{1,80}\]/gi,
// Noun-only template variables common in AI-generated email or
// letter boilerplate. Match the bare noun OR noun + qualifier.
// Conservative list: only nouns that almost never appear as
// bracketed real content (citation refs, code identifiers, etc.
// are excluded because they typically contain dots, slashes,
// hyphens, or version numbers).
/\[(?:Recipient|Sender|Topic|Subject|Salutation|Closing|Position|Department|Project Name|Company Name|Date)(?:\s+[^\]\n]{0,60})?\]/gi,
// All-caps directive forms ("[INSERT X]", "[FILL IN]") — the
// uppercase tells you it's a slot, not real content.
/\[(?:INSERT|FILL\s+IN|ADD|TODO|TBD|PLACEHOLDER)[^\]\n]{0,80}\]/g,
// Date stubs.
/\b(?:19|20)\d{2}-XX-XX\b/g,
/\bXX\/XX\/(?:19|20)\d{2}\b/g,
// HTML/Markdown comment placeholders with placeholder verbs.
/<!--\s*(?:add|fill\s+in|insert|todo|placeholder)[^>]{0,120}-->/gi,
];
// Chatbot citation/markup tokens that leak through copy-paste.
// `citeturn0search0` / `citeturn0news5` from ChatGPT, contentReference
// tokens, oai_citation, attached_file references, grok_card markers.
// Each is a near-definitive signature of a specific tool.
const AI_CITATION_MARKUP = [
/\bcite(?:turn|news|search|navigation)\d+(?:search|turn|news|navigation)\d+/gi,
/contentReference\s*\[oaicite:[^\]]+\]\s*\{[^}]*\}/gi,
/\boai_citation\b/gi,
/\[attached_file:\d+\]/gi,
/\bgrok_card\b/gi,
];
// UTM/tracking parameters auto-appended by AI tools to URLs they
// generate. Survives copy-paste even when nothing else does.
const AI_UTM_SOURCE = [
/[?&]utm_source=(?:chatgpt|openai|copilot|claude|grok|gemini|perplexity)(?:\.com|\.ai)?\b/gi,
/[?&]referrer=(?:chatgpt|copilot|grok|claude|gemini|perplexity)\.(?:com|ai)\b/gi,
];
// ─── Template phrases ──────────────────────────────────────────────
const TEMPLATE_PHRASES = [
/\ba\s+\w+\s+step\s+(?:towards?|forward\s+for)\b/gi,
/\bwhether\s+you'?re\s+\w+\s+or\s+\w+/gi,
/\bi\s+recently\s+had\s+the\s+pleasure\s+of\b/gi,
];
// ─── False concession ──────────────────────────────────────────────
const FALSE_CONCESSION = [
/\bwhile\s+\w+\s+is\s+impressive\b/gi,
/\balthough\s+\w+\s+has\s+made\s+strides\b/gi,
/\bdespite\s+\w+\s+challenges?\b/gi,
];
// ─── Rhetorical question openers ───────────────────────────────────
const RHETORICAL_QUESTIONS = [
/\bbut\s+what\s+does\s+this\s+mean\s+for\b/gi,
/\bso\s+why\s+should\s+you\s+care\b/gi,
/\bwhat'?s\s+next\?\s*/gi,
];
// ─── Hedge-stacked predictions ─────────────────────────────────────
// Stacks a modal with a hedge adverb: "could potentially create new
// opportunities", "may eventually unlock value." Either word alone is
// fine; the stack is the tell.
const HEDGE_STACK = [
// At most one intervening word, and never a negator. The old {0,2} gap
// matched ordinary English: "could not possibly" (plain emphatic
// negation) and inverted questions like "could a savage possibly" both
// fired. Measured on the human-control corpus, 3 of 4 hedge-stack flags
// were this over-match. See issue #69.
/\b(?:could|may|might)\s+(?:(?!not\b|never\b|hardly\b|scarcely\b|barely\b)\w+\s+)?(?:potentially|eventually|ultimately|possibly|conceivably)\b/gi,
/\b(?:potentially|eventually|ultimately)\s+(?:could|may|might)\b/gi,
];
// ─── Generic future-narrative closers ──────────────────────────────
// The "may become one of the most important narratives" template — vague
// future significance with no falsifiable claim. Covers narratives /
// stories / trends / themes / chapters / movements.
const FUTURE_NARRATIVE = [
/\b(?:may|could|will|is\s+(?:poised|set)\s+to)\s+become\s+(?:one\s+of\s+)?(?:the\s+)?(?:most\s+)?\w+\s+(?:narratives?|stories|developments?|trends?|movements?|chapters?|themes?|forces?)\b/gi,
/\bone\s+of\s+the\s+most\s+important\s+(?:narratives?|stories|trends?|themes?)\s+of\s+the\s+(?:next|coming)\s+\w+\b/gi,
];
// ─── "Real/actual" adjective inflation ─────────────────────────────
// "Real on-chain tokenomics", "actual reward sustainability" — using
// real/actual/genuine/true as an empty intensifier on an abstract noun
// to imply the rest of the field is fake/superficial.
const REAL_ACTUAL_INFLATION = [
/\b(?:real|actual|genuine|true)\s+(?:on-?chain\s+)?(?:tokenomics|economics|utility|adoption|sustainability|impact|revenue|fundamentals|demand|value|innovation|traction)\b/gi,
];
// ─── Formulaic openers ─────────────────────────────────────────────
// The "In the rapidly evolving world of X, Y has emerged as..." family
// — LLM-default essay openers.
const FORMULAIC_OPENERS = [
/\bin\s+the\s+(?:rapidly\s+|ever-?\s*)?(?:evolving|changing|expanding|growing|shifting)\s+(?:world|landscape|realm|space|field|domain|era)\s+of\b/gi,
/\bin\s+(?:an?|the)\s+(?:digital\s+)?age\s+(?:where|of)\b/gi,
/\bas\s+(?:we|the\s+world|society|industries?)\s+(?:continue|move|navigate|enter)\s+(?:to\s+)?(?:evolve|forward|into|through)\b/gi,
// "has emerged as a leader/force/category" — gated to the inflated
// nouns that signal pseudo-significance, since bare "has emerged as
// a" matches normal English ("Rust has emerged as a serious systems
// language"). Same gating for "has become increasingly".
/\bhas\s+emerged\s+as\s+(?:a|the|one\s+of)\s+(?:leading|key|major|critical|essential|fundamental|pivotal|prominent|dominant|important)\s+\w+/gi,
/\bhas\s+become\s+increasingly\s+(?:important|critical|popular|relevant|prominent|essential)\b/gi,
];
// ─── Speculative scenario openers ──────────────────────────────────
// "Imagine a world where...", "Picture a future in which...", "Envision
// a world where..." — the LLM habit of opening an argument with a
// hypothetical that lists desirable outcomes instead of making a claim.
// Gated to the world/future/reality object plus where/in-which so it
// stays off instructional "imagine you have an array" (a teaching device
// pointing at a concrete example, not a speculative world) and bare
// "imagine that" asides. "consider a scenario where…" is deliberately
// excluded: that is analytical framing common in technical reasoning,
// not the marketing-opener tell. An optional short comma interrupter
// catches the equally common "Imagine, for a moment, a world where…"
// cadence. Known accepted false positive: fiction openings and staged
// thought experiments match too — the engine has no fiction context
// mode, so that call is left to the skill's carve-out (highlight-only;
// a lone hit cannot flip a document's classification).
const SPECULATIVE_OPENERS = [
/\b(?:imagine|picture|envision)(?:\s*,[^,\n]{1,30},)?\s+a\s+(?:world|future|reality)\s+(?:where|in\s+which)\b/gi,
];
// ─── Launch-copy dramatic introductions ────────────────────────────
// "Meet Flowdesk, your new favorite treasury dashboard" / "Think
// Notion meets Figma" — the LLM-default product-introduction move
// in launch and announcement copy. Both surfaces are gated to the
// sentence-initial imperative followed by ONE capitalized token of 2
// to 30 characters, which is a recall limit: a two-token product name
// ("Meet North Star", "Think Google Docs meets Microsoft Word") is a
// deliberate miss. The Meet surface additionally requires one of four
// launch-copy heads — "your new favorite", "your new go-to", and
// "the new home/way/standard", the last three only when followed by
// "of" / "to" / "in|for" or by end-of-clause punctuation. Without
// that tail the head noun swallows a compound noun and ordinary prose
// fires: "Meet Rosa, the new home secretary" and "Meet Emma, the new
// way station manager" both matched before the tail was required.
// Bare "Meet Sarah, your new account manager" is how humans introduce
// colleagues, pets, and babies, so that form stays with the skill's
// judgment side. Two surfaces from
// the same family are deliberately NOT detected. "Say hello to X",
// because "Say hello to Grandma." is ordinary human prose. And bare
// "Enter X.", because the sentence-initial capitalized-noun form is
// how UI and doc instructions are written: "Enter Password.", "Enter
// Amount.", "Enter Username — your work email." Dropping the dash
// terminator does not reach the period-terminated class, and neither
// does a field-name denylist, so that surface stays with the skill's
// judgment side and the UI forms are pinned as must-not-fire
// fixtures. The anchors are lookbehinds so adjacent intros each
// count and the reported span starts at the tell itself.
const LAUNCH_INTROS = [
/(?<=^|[.!?]\s|\n)Meet\s+[A-Z][\w'-]{1,29}\s*,\s*(?:your\s+new\s+(?:favorite|go-to)\b|the\s+new\s+(?:home\s+of\b|way\s+to\b|standard\s+(?:in|for)\b|(?:standard|way|home)(?=\s*(?:[.!?,;:\u2013\u2014]|$))))/g,
/(?<=^|[.!?]\s|\n)[Tt]hink\s+[A-Z][\w'-]{1,29}\s+meets\s+[A-Z][\w'-]{1,29}\b/g,
];
// ─── Dramatized contrast against the crowd ─────────────────────────
// "shipped it in 2022, while everyone else was still debating
// timelines" — a claim propped on an implied lagging crowd. The gate
// is a dismissive verb PLUS the "was still" dramatization marker,
// because bare "while everyone else" is ordinary simultaneity ("she
// read while everyone else watched the movie") and even the
// dismissive verbs are ordinary English in literal use ("others
// debated the amendment" in wire copy). That gate is on this FIRST
// branch only. Its stems are restricted to the -ing form, so the
// adjective ("was still deliberate about the tradeoff"), the passive
// ("was still debated by pundits") and the bare present ("was still
// debates timelines") all stay clean — allowing e/es/ed let all
// three through. The other two branches carry no "was still"
// requirement: they match their own stereotyped wording ("writing
// think-pieces", "playing catch-up"), and the skill entry scopes the
// claim the same way. Verb stems carry explicit inflection tails so
// agent nouns ("the market speculators") and adverbs ("deliberately
// ignored") never match. Measured residue, all accepted: branch one
// fires on ANY literal progressive use of its verbs ("while the
// market was still speculating about the price"), not only on "was
// still debating"; branches two and three fire on literal contrasts
// of their own ("while everyone else wrote think-pieces from
// Washington", "while everyone else played catch-up in the spring").
// Recall is deliberately sacrificed: "was busy debating" without
// "still" stays a miss, per precision-over-recall.
const CROWD_CONTRAST = [
/\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was|were|is|are)\s+still\s+(?:busy\s+)?(?:(?:debat|deliberat|hesitat|theoriz|philosophiz|pontificat|speculat|argu)ing|(?:dither|bicker)ing)\b/gi,
/\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was\s+|were\s+)?(?:busy\s+)?(?:writing|wrote)\s+think-?\s?pieces\b/gi,
/\bwhile\s+(?:everyone\s+else|the\s+(?:industry|market|competition)|others)\s+(?:was\s+|were\s+|is\s+|are\s+)?(?:still\s+)?play(?:ed|ing|s)?\s+catch[-\s]?up\b/gi,
];
// ─── Fake-casual props (stage directions and wink asides) ──────────
// The regexable props from the fake-casual register: theatrical
// asterisk stage directions ("*checks notes*", "*chef's kiss*",
// "*mic drop*") and wink asides. Both lists are closed and short, and
// that is a recall limit: exactly six stage directions ("checks
// notes", "chef's kiss", "mic drop", "takes a deep breath", "sips
// coffee|tea", "nervous laughter") and exactly four parentheticals,
// the full (yes|no) x (really|seriously) grid. Neighbours in the same
// register are deliberate misses: "*checks calendar*" and "(yes,
// honestly)" do not fire. The kiss pattern requires the apostrophe —
// making it optional matched the ordinary sentence "At midnight,
// *chefs kiss* their spouses goodbye".
// The rest of the register (one-word verdict closers, label-prefix
// openers, the self-QA volley) needs register judgment and stays
// skill-only — "wild." is a word, not a regex target. "because of
// course …" joins them: a tense gate does not separate the wink from
// the ordinary human grumble, because "The build failed because of
// course it did." is that grumble in the same present-tense-plus-did
// form the wink uses. Under precision-over-recall the surface is
// judgment-only, and the grumble is pinned as a fixture. The kiss
// pattern accepts the curly apostrophe (U+2019) — the form smart
// punctuation and LLMs actually emit. Known accepted false positive:
// a human writer using a wink aside on purpose; the props are
// weighted as a strong single hit, not a classification by
// themselves.
const FAKE_CASUAL_PROPS = [
/\*\s?(?:checks\s+notes|chef['\u2019]s\s+kiss|mic\s+drop|takes\s+a\s+deep\s+breath|sips\s+(?:coffee|tea)|nervous\s+laughter)\s?\*/gi,
/\(\s?(?:yes|no)\s?,\s?(?:really|seriously)\s?\)/gi,
];
// ─── Performed-insight phrases ─────────────────────────────────────
// Essayist tics that announce profundity instead of delivering it.
// Curated noun/complement lists keep precision high: "the whole family"
// and "sit with him" are ordinary English and must not fire. Adapted
// from Simon Willison's LLM cliché highlighter
// (tools.simonwillison.net/llm-cliche-highlighter).
const PERFORMED_INSIGHT = [
/\bsit(?:s|ting)?\s+with\s+(?:that|this)(?=\s*(?:[.!?,;:)\u2013\u2014\u2019"']|for\s+a\s+(?:moment|minute|second|beat)\b|$))(?:\s+for\s+a\s+(?:moment|minute|second|beat))?/gi,
/\bsit(?:s|ting)?\s+with\s+(?:the|your)\s+(?:discomfort|tension|uncertainty|ambiguity|grief|unease)\b/gi,
/\b(?:that|this|it|which)(?:['\u2019]s|\s+(?:is|was))\s+not\s+nothing\b/gi,
/\byou\s+already\s+know\s+the\s+answer\b/gi,
/\b(?:do\s+not|don['\u2019]t)\s+(?:have\s+to\s+)?take\s+my\s+word\s+for\s+it\b/gi,
/(?<=^|[.!?]\s|\n)Turns\s+out\b/g,
/(?:['\u2019]s|\b(?:is|was|are|were))\s+the\s+(?:whole|entire)\s+(?:point|game|ballgame|trick|pitch|idea|play|business\s+model|value\s+proposition)\b/gi,
/\b(?:that|this)(?:['\u2019]s|\s+(?:is|was))\s+the\s+part\s+(?:that|I|you|we|nobody|no\s+one|most\s+people)\b/gi,
/\bthe\s+only\s+[\w'\u2019-]+\s+that\s+(?:matters|counts)\b/gi,
/\bis\s+dead\s*[.;,:\u2013\u2014]\s*long\s+live\b/gi,
/\b(?:that|this)(?:['\u2019]s|\s+(?:is|was))\s+why\s+[^.!?\n]{0,60}\s+mattered\b/gi,
];
// ─── Negation chains ───────────────────────────────────────────────
// "No fluff, no filler, no jargon" / "It didn't ask, didn't wait" /
// "Don't call it X. Call it Y." Precision guards, in order: the
// "no …" chain must open its sentence (mid-sentence inventories like
// "takes no arguments, no headers, and no body" are factual, not
// rhetorical); the "did not" chain must be comma-joined with the
// subject elided ("I did not sleep. I did not eat" is ordinary
// narration and stays clean); the stop-list keeps idiomatic pairs
// ("no more, no less", "no matter what") from firing. Adapted from
// Simon Willison's LLM cliché highlighter.
const NO_ITEM_STOP = "(?!matter\\b|one\\b|doubt\\b|longer\\b|way\\b|less\\b|more\\b|such\\b|other\\b|means\\b)";
const NO_ITEM_SECOND_STOP = "(?!(?:in|on|at|of|to|for|with|from|by|is|are|was|were|be|been|being|will|would|can|could|should|shall|may|might|must|have|has|had|do|does|did)\\b)";
const NEGATION_CHAIN = [
new RegExp(
"(?<=^|[.!?]\\s|\\n|[:\\u2013\\u2014]\\s)No\\s+" + NO_ITEM_STOP + "[a-z'\u2019-]+(?:\\s+" + NO_ITEM_SECOND_STOP + "[a-z'\u2019-]+)?" +
"(?:\\s*,\\s*(?:and\\s+|or\\s+|just\\s+)?no\\s+" + NO_ITEM_STOP + "[a-z'\u2019-]+(?:\\s+" + NO_ITEM_SECOND_STOP + "[a-z'\u2019-]+)?){2,}",
'gm'
),
/\b(?:did\s+not|didn['\u2019]t)\s+[a-z]+[^,.;!?\n]{0,20},\s*(?:did\s+not|didn['\u2019]t)\s+[a-z]+/gi,
/\b(?:do\s+not|don['\u2019]t)\s+(?:just\s+)?(\w+)\s+it\b[^.!?\n]{0,60}[.!?;:,][\s'"\u201d\u2019]*(?:just\s+)?\1\s+it\b/gi,
];
// ─── Dev-blog boilerplate ──────────────────────────────────────────
// Stock simplicity slogans from developer marketing. Adapted from
// Simon Willison's LLM cliché highlighter.
const DEV_BLOG_BOILERPLATE = [
/\bit\s+just\s+works\b(?!\s+out\b(?![-\s]+of[-\s]+the[-\s]+box\b))/gi,
/\bzero[-\s]config(?:uration)?\b/gi,
/\bsane\s+defaults\b/gi,
/\b(?:hold|fit|fits|holds)\s+in\s+your\s+head\b/gi,
];
// Function words whose presence MID-title marks the AI section-header shape.
// Word-anchored: without \b the "A" alternative matches inside any word and
// the guard silently degrades to "four tokens".
const FUNCTION_WORD = /\b(?:And|Or|Of|The|In|For|To|A|An)\b/;
// Must accept exactly what TITLE_CASE_HEADER accepts, or the prefix survives
// into the token count and reintroduces the ##-as-token bug.
const MD_HEADING_PREFIX = /^#{1,6}[ \t]+/;
/** Byte ranges covered by fenced code blocks, computed once per scan.
*
* A document that documents Markdown is the normal case for this rule -- a
* fenced `## Heading` example is illustration, not the author's own section
* header, and flagging it makes every docs page flag itself.
*
* This tracks the opening delimiter instead of counting them, because a
* parity count is wrong on the very case the rule exists for. CommonMark
* closes a fence only on the same character at the same length or longer, so
* a four-backtick fence wrapping a three-backtick example -- exactly how you
* document fences -- nests in practice, and counting delimiters inverts on
* it. Up to three spaces of indent are legal. An unclosed fence runs to end
* of document, matching how renderers treat it.
*
* Computed once per scan rather than rescanned per hit: the previous version
* sliced the whole document for every candidate, which is quadratic on a
* heading-dense file. */
function fenceRanges(text) {
const re = /^[ \t]{0,3}(`{3,}|~{3,})([^\n]*)$/gm;
const ranges = [];
let open = null;
let m;
while ((m = re.exec(text)) !== null) {
const marker = m[1];
if (!open) {
open = { char: marker[0], len: marker.length, start: m.index };
} else if (
marker[0] === open.char &&
marker.length >= open.len &&
/^[ \t]*\r?$/.test(m[2])
) {
ranges.push([open.start, m.index + m[0].length]);
open = null;
}
}
if (open) ranges.push([open.start, text.length]);
return ranges;
}
function inFenceRange(ranges, index) {
return typeof index === 'number' && ranges.some(([a, b]) => index >= a && index < b);
}
function blankRange(chars, start, end) {
for (let i = start; i < end && i < chars.length; i += 1) {
if (chars[i] !== '\n') chars[i] = ' ';
}
}
// Copy of the text with fenced blocks and inline code spans blanked out.
// Index-preserving: each masked character becomes a space and newlines are
// kept, so offsets into the result still address the same position in the
// original. For rules where a character inside code is something the author
// is quoting rather than using — a `#fff` in a CSS sample is not a tag.
// Fences are blanked first so their backticks cannot pair with a later
// inline span and swallow the prose between them.
function maskCode(text) {
const chars = text.split('');
for (const [a, b] of fenceRanges(text)) blankRange(chars, a, b);
// Indented code blocks are deliberately NOT masked. Four spaces is a code
// block only at top level; under a list marker it is a paragraph
// continuation, so blanking it silences real tag blocks. #90 reports
// fences and inline spans, and those are what this masks.
const withoutFences = chars.join('');
const inlineRe = /(`+)(?:(?!\1)[^\n])+\1/g;
let m;
while ((m = inlineRe.exec(withoutFences)) !== null) blankRange(chars, m.index, m.index + m[0].length);
return chars.join('');
}
function initialFrontmatterRange(text) {
const lines = [];
const lineRe = /[^\r\n]*(?:\r\n|\n|\r|$)/g;
let match;
while ((match = lineRe.exec(text)) !== null && match[0]) {
const body = match[0].replace(/(?:\r\n|\n|\r)$/, '');
lines.push({ body, start: match.index, end: match.index + body.length });
}
if (lines.length < 3 || !/^---[ \t]*$/.test(lines[0].body.replace(/^\uFEFF/, ''))) return null;
let closingLine = -1;
for (let i = 1; i < lines.length; i += 1) {
if (/^---[ \t]*$/.test(lines[i].body)) {
closingLine = i;
break;
}
}
if (closingLine === -1) return null;
// A pair of thematic breaks can also surround ordinary Markdown prose.
// Require the first substantive line to begin like a YAML mapping entry
// before treating the delimited block as frontmatter. Leading blank lines
// and YAML comments are valid, and the line parser accepts LF, CRLF, or CR.
const yamlKey = /^[ \t]*(?:[A-Za-z0-9_.-]+|"[^"\r\n]+"|'[^'\r\n]+')[ \t]*:/;
const firstContent = lines
.slice(1, closingLine)
.find((line) => line.body.trim() && !/^[ \t]*#/.test(line.body));
if (!firstContent || !yamlKey.test(firstContent.body)) return null;
return { start: 0, end: lines[closingLine].end };
}
// Mask source-only Markdown spans while preserving source offsets. The
// detector can then score what a reader sees without making later issue
// indexes or sentence highlights point at the wrong source location.
function maskRenderedMarkdown(text) {
const chars = text.split('');
let maskedFrontmatter = 0;
const frontmatter = initialFrontmatterRange(text);
if (frontmatter) {
blankRange(chars, frontmatter.start, frontmatter.end);
maskedFrontmatter = 1;
}
const maskCommentCode = () => {
const codeChars = maskCode(chars.join('')).split('');
maskTopLevelIndentedCode(codeChars, { listAware: true });
return codeChars.join('');
};
let maskedHtmlComments = 0;
let searchIndex = 0;
while (searchIndex < text.length) {
const openingIndex = maskCommentCode().indexOf('<!--', searchIndex);
if (openingIndex === -1) break;
const closingIndex = text.indexOf('-->', openingIndex + 2);
const end = closingIndex === -1 ? text.length : closingIndex + 3;
blankRange(chars, openingIndex, end);
maskedHtmlComments += 1;
searchIndex = end;
}
return { text: chars.join(''), maskedFrontmatter, maskedHtmlComments };
}
function maskMultilineBlockquotes(text) {
const chars = text.split('');
const lines = [];
const lineRe = /[^\r\n]*(?:\r\n|\n|\r|$)/g;
let match;
while ((match = lineRe.exec(text)) !== null && match[0]) {
const body = match[0].replace(/[\r\n]+$/, '');
lines.push({ text: body, start: match.index, end: match.index + body.length });
}
let quotedLines = 0;
const isQuote = lines.map((line) => /^\s*>\s/.test(line.text));
for (let i = 0; i < lines.length; i += 1) {
if (isQuote[i] && ((isQuote[i - 1] && i > 0) || isQuote[i + 1])) {
blankRange(chars, lines[i].start, lines[i].end);
quotedLines += 1;
}
}
return { text: chars.join(''), quotedLines };
}
// Keep the historical deletion behavior for default plain mode. Paragraph-
// scoped rules depend on the surrounding lines being rejoined exactly this
// way, so changing this prepass would change scores for existing callers.
function stripMultilineBlockquotes(text) {
const rawLines = text.split(/\r?\n/);
const isQuote = rawLines.map((line) => /^\s*>\s/.test(line));
const stripIndexes = new Set();
for (let i = 0; i < rawLines.length; i += 1) {
if (isQuote[i] && ((isQuote[i - 1] && i > 0) || isQuote[i + 1])) stripIndexes.add(i);
}
return {
text: rawLines.filter((_, i) => !stripIndexes.has(i)).join('\n'),
quotedLines: stripIndexes.size,
};
}
function maskTopLevelIndentedCode(chars, { listAware = false } = {}) {
const lines = chars.join('').split('\n');
let offset = 0;
let inBlock = false;
let previousBlank = true;
let listContext = false;
for (let i = 0; i < lines.length; i += 1) {
const line = lines[i];
const indented = /^(?: {4}|\t)\S/.test(line);
const blank = line.trim() === '';
const isCode = indented && (inBlock || (previousBlank && (!listAware || !listContext)));
if (isCode) {
blankRange(chars, offset, offset + line.length);
inBlock = true;
} else if (!blank) {
inBlock = false;
}
if (listAware && !blank && !isCode) {
if (/^ {0,3}(?:[-*+]|\d{1,9}[.)])(?:\s|$)/.test(line)) listContext = true;
else if (/^\S/.test(line)) listContext = false;
}
previousBlank = blank;
offset += line.length + 1;
}
}
function maskYamlFrontmatter(chars) {
const lines = chars.join('').split('\n');
const bare = (line) => line.replace(/\r$/, '');
const first = bare(lines[0]).replace(/^\uFEFF/, '');
if (first !== '---' || lines.length < 2 || /^\s*$/.test(bare(lines[1]))) return;
let closingLine = -1;
for (let i = 1; i < lines.length; i += 1) {
if (bare(lines[i]) === '---') {
closingLine = i;
break;
}
}
if (closingLine === -1) return;
let end = 0;
for (let i = 0; i <= closingLine; i += 1) {
end += lines[i].length;
if (i < lines.length - 1) end += 1;
}
blankRange(chars, 0, end);
}
function maskYamlMetadata(chars) {
const lines = chars.join('').split('\n');
let offset = 0;
let nestedAfterIndent = null;
for (const line of lines) {
const bare = line.replace(/\r$/, '');
// Lowercase keys are the common unfenced-YAML shape. Keeping this
// case-sensitive prevents prose labels such as "Note: ..." from being
// mistaken for metadata and silencing a real copyedit on the line.
const key = bare.match(/^([ \t]*)(?:-[ \t]+)?[a-z_][a-z0-9_.-]*[ \t]*:(.*)$/);
const indentation = (bare.match(/^[ \t]*/) || [''])[0].length;
let shouldMask = false;
if (key) {
shouldMask = true;
nestedAfterIndent = key[2].trim() === '' ? key[1].length : null;
} else if (nestedAfterIndent !== null && bare.trim() !== '' && indentation > nestedAfterIndent) {
shouldMask = true;
} else if (bare.trim() === '') {
nestedAfterIndent = null;
} else {
nestedAfterIndent = null;
}
if (shouldMask) blankRange(chars, offset, offset + line.length);
offset += line.length + 1;
}
}
function maskMarkdownTables(chars) {
const lines = chars.join('').split('\n');
const delimiter = /^\s*\|?\s*:?-{3,}:?\s*(?:\|\s*:?-{3,}:?\s*)+\|?\s*\r?$/;
const rows = new Set();
for (let i = 0; i < lines.length; i += 1) {
if (!delimiter.test(lines[i])) continue;
if (i > 0 && lines[i - 1].includes('|')) rows.add(i - 1);
rows.add(i);
for (let j = i + 1; j < lines.length && lines[j].includes('|'); j += 1) rows.add(j);
}
let offset = 0;
for (let i = 0; i < lines.length; i += 1) {
if (rows.has(i)) blankRange(chars, offset, offset + lines[i].length);
offset += lines[i].length + 1;
}
}
function maskDelimitedQuotes(chars, open, close, apostropheAware = false) {
const isWord = (char) => char !== undefined && /[a-z0-9]/i.test(char);
const isEscaped = (index) => {
let slashes = 0;
for (let i = index - 1; i >= 0 && chars[i] === '\\'; i -= 1) slashes += 1;
return slashes % 2 === 1;
};
let start = -1;
for (let i = 0; i < chars.length; i += 1) {
if (start === -1) {
if (chars[i] === open && !isEscaped(i) && (!apostropheAware || !isWord(chars[i - 1]))) start = i;
} else if (chars[i] === close && !isEscaped(i) && (!apostropheAware || !isWord(chars[i + 1]))) {
blankRange(chars, start, i + 1);
start = -1;
}
}
}
// Additional protected spans for the hyphenation copyedit. Unlike general
// AI-tell matching, this rule must not "correct" a literal spelling in a
// quote, URL, path, filename, command flag, or Markdown blockquote.
function maskHyphenationProtected(text) {
const chars = maskCode(text).split('');
maskTopLevelIndentedCode(chars);
maskYamlFrontmatter(chars);
maskYamlMetadata(chars);
maskMarkdownTables(chars);
maskDelimitedQuotes(chars, '"', '"');
maskDelimitedQuotes(chars, '“', '”');
maskDelimitedQuotes(chars, "'", "'", true);
maskDelimitedQuotes(chars, '‘', '’', true);
const maskMatches = (regex) => {
const source = chars.join('');
let match;
while ((match = regex.exec(source)) !== null) {
blankRange(chars, match.index, match.index + match[0].length);
}
};
maskMatches(/^[ \t]*>[^\n]*$/gm);
maskMatches(/\b(?:https?:\/\/|www\.)[^\s<>]+/gi);
maskMatches(/<[!?/]?[a-z][^>\n]*>/gi);
maskMatches(/(?<![a-z0-9_-])--?[a-z0-9][a-z0-9-]{0,127}/gi);
// Paths and filenames use bounded components. Besides preventing
// superlinear backtracking on long kebab blobs, the explicit prefix and
// trailing-slash forms cover single-component paths such as C:\\code-base,
// ~/code-base, /code-base, and code-base/.
maskMatches(/(?:[a-z]:[\\/]|\.{1,2}[\\/]|~[\\/]|[\\/])[a-z0-9_.-]{1,255}/gi);
maskMatches(/(?:[a-z]:[\\/]|(?:\.\.?[\\/])?)(?:[a-z0-9_.-]{1,64}[\\/]){1,128}[a-z0-9_.-]{1,64}/gi);
maskMatches(/\b[a-z0-9_.]{0,63}-[a-z0-9_.-]{1,64}[\\/]/gi);
maskMatches(/\b[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}\.[a-z0-9]{1,16}\b/gi);
// Technical identifiers that are distinguishable from ordinary prose:
// selectors, scoped packages, versioned tokens, assignments, and kebab
// names next to an explicit identifier cue. Plain lowercase compounds
// remain visible to the curated copyedit patterns below.
maskMatches(/[.#][a-z_][a-z0-9_.-]{0,63}-[a-z0-9_.-]{1,64}\b/gi);
maskMatches(/@[a-z0-9_.-]{1,64}\/[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}(?:@[^\s,;)\]}]{1,32})?/gi);
maskMatches(/\b[a-z0-9_.-]{1,64}-[a-z0-9_.-]{1,64}@[~^]?v?\d[a-z0-9*_.+-]{0,31}\b/gi);
maskMatches(/\b(?:[a-z0-9_.-]{0,64}\d[a-z0-9_.-]{0,64}-[a-z0-9_.-]{1,64}|[a-z0-9_.-]{1,64}-[a-z0-9_.-]{0,64}\d[a-z0-9_.-]{0,64})\b/gi);
maskMatches(/\b[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}(?=[ \t]*[=:])/gi);
maskMatches(/\b[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}(?=[ \t]+(?:npm[ \t]+)?(?:package|module|class|selector|config(?:uration)?[ \t]+key|key|identifier|property|setting|token|slug|command|option)\b)/gi);
maskMatches(/\b(?:(?:file(?:name)?|directory|folder|package|module|class|selector|config(?:uration)?[ \t]+key|identifier|property|setting|token|slug|command|option)(?:[ \t]+(?:named|called|is|was))?|key[ \t]+(?:named|called|is|was))[ \t]+(?:@[a-z0-9_.-]{1,64}\/)?[a-z_][a-z0-9_.]{0,63}(?:-[a-z0-9_.]{1,64}){1,8}\b/gi);
maskMatches(/\b(?:npm|pnpm|yarn)[ \t]+(?:add|install)[ \t]+(?:@[a-z0-9_.-]{1,64}\/)?[a-z0-9_.]{1,64}(?:-[a-z0-9_.]{1,64}){1,8}/gi);
return chars.join('');
}
function findUnnecessaryHyphenation(text) {
const scanText = maskHyphenationProtected(text);
const issues = [];
for (const entry of UNNECESSARY_HYPHENATION) {
const regex = new RegExp(entry.pattern.source, entry.pattern.flags);
let match;
while ((match = regex.exec(scanText)) !== null) {
issues.push({
type: 'unnecessary-hyphenation',
text: match[0],
severity: 'medium',
suggestion: typeof entry.suggestion === 'function'
? entry.suggestion(match[0])
: entry.suggestion,
});
}
}
return issues;
}
// ─── Forms that open with `#` but are not social tags ──────────────
// `#` is overloaded in technical prose and the hashtag rule counts every
// `#word` it sees, so these are subtracted before the threshold applies:
// #88, #1234 issue and PR references
// #1a2b3c CSS hex colours, 6 or 8 chars AND containing a digit
// #include, #ifndef C preprocessor directives
// Deliberately NOT carving out 3- and 4-digit hex: #dad, #cafe, #b2b, #e2e,
// #ace, #face and #bad are real tags, and subtracting them cost true
// positives on exactly the stuffed-block shape this rule exists to catch.
// Real palettes are dominated by 6-digit values, so a CSS paragraph still
// lands under the threshold without the short forms.
// `owner/repo#88` and URL fragments need no carve-out: the char before `#`
// is a word char, so the rule's own anchor already rejects them.
// Ambiguous word tags stay counted on purpose. `#general` as a channel and
// `#general` as a tag are the same token, and separating them needs a guess
// that costs more precision on real tag blocks than the carve-out buys.
// Requires at least one digit: #decade, #facade, #deadbeef are a-f words and
// real tags. Every actual palette value in the wild carries a digit.
const HEX_COLOUR = /^(?=[0-9a-f]*\d)(?:[0-9a-f]{6}|[0-9a-f]{8})$/i;
const CPP_DIRECTIVE = /^(?:include|define|undef|if|ifdef|ifndef|elif|else|endif|pragma|error|warning|line)$/;
function isSocialTag(tag) {
return !/^\d+$/.test(tag) && !HEX_COLOUR.test(tag) && !CPP_DIRECTIVE.test(tag);
}
// ─── Title Case Section Headers in non-technical prose ─────────────
// "Strategic Negotiations And Key Partnerships" — every content word
// capitalized. Acceptable in API docs, ML papers, news headlines. Tell
// in marketing/personal/blog prose. Gated to "personal" / "marketing"
// context modes (technical mode skips this check).
//
// The optional `#{1,6}` prefix is load-bearing (#62): without it the `^[A-Z]`
// anchor required the line to START with a capital, so `## Benefits And
// Strategic Considerations` never matched — the first character is `#`. The
// rule missed the single most common way a heading is actually written, while
// catching the bare-line form it is usually converted from. Reported by a
// downstream vendoring the detector.
//
// Setext headings (`Title`/`=====`) need no prefix: their text line is bare
// and already matched by this same pattern.
const TITLE_CASE_HEADER = /^(?:#{1,6}[ \t]+)?([A-Z][a-z]+(?:\s+(?:[A-Z][a-z]+|and|or|of|the|in|for|to|a|an))+\s+[A-Z][a-z]+)\s*$/gm;
// ─── Parenthetical hedging asides ──────────────────────────────────
// "(and increasingly, X)", "(or more precisely, Y)", "(though to be
// fair, Z)" — pseudo-aside that adds no information but performs
// thoughtfulness. Different from genuine human parentheticals which
// tend to be tangents or clarifications, not hedges.
const PARENTHETICAL_HEDGE = [
/\(\s*(?:and\s+)?(?:increasingly|notably|importantly|crucially|interestingly|perhaps)[,]?\s+[^)]{3,60}\)/gi,
/\(\s*or\s+more\s+(?:precisely|accurately|specifically)[,]?\s+[^)]{3,60}\)/gi,
/\(\s*though\s+to\s+be\s+fair[,]?\s+[^)]{3,60}\)/gi,
/\(\s*at\s+least\s+(?:in\s+)?(?:theory|principle|part)[,]?\s+[^)]{0,60}\)/gi,
];
// ─── Confidence calibration ────────────────────────────────────────
const CONFIDENCE_CALIBRATION = [
/\binterestingly\b/gi,
/\bsurprisingly\b/gi,
/\bimportantly\b/gi,
/\bsignificantly\b/gi,
/\bcertainly\b/gi,
/\bundoubtedly\b/gi,
/\bwithout\s+a\s+doubt\b/gi,
];
// ─── Social endorsement / CTA closers ──────────────────────────────
// The curatorial sign-off LLMs append to LinkedIn / X posts that share
// or recommend something — usually a colon teeing up a link. Distinct
// from the bare "worth reading" word-table entry (a single weak word)
// and from infomercial hooks (mid-flow teasers): this is the
// demonstrative-anchored endorsement — "THIS one is worth your time:",
// "do yourself a favor and read this", "thank me later" — that performs
// a recommendation without giving the reader a reason to click.
//
// Precision-first: each pattern carries an anchor so it stays off the
// literal-verb prose a human writes. The demonstrative object ("read
// THIS", not "read the runbook"), the trailing-terminal lookahead on
// "miss this" / "bookmark this" (the closing-line shape, not "miss this
// meeting" / "bookmark this page"), and the sentence-initial lookbehind
// on "thank me later" / "save this for later" (the imperative CTA, not
// "she will thank me later") all exist to suppress false positives on
// ordinary instructional/conversational text. Apostrophe classes admit
// the curly ' (U+2019) because LinkedIn / Word / macOS auto-curl it —
// the straight-only form would miss the canonical "you won't" closer.
const SOCIAL_CTA_CLOSER = [
/\bthis\s+one['’]?s?\s+(?:is\s+)?(?:well\s+|totally\s+|absolutely\s+|definitely\s+|really\s+|truly\s+|easily\s+|more\s+than\s+)?worth\s+(?:your\s+time|the\s+read|a\s+read|every\s+(?:minute|second)|reading|watching|a\s+listen|a\s+watch|a\s+look|it)\b/gi,
/\bthis\s+one['’]?s?\s+(?:is\s+)?a\s+must[-\s]?(?:read|watch|listen|see)\b/gi,
/\b(?:highly|strongly|can['’]?t|cannot)\s+recommend\w*\s+(?:giving\s+)?(?:this|it)\s+(?:one\s+)?a\s+(?:read|listen|watch|look|go)\b/gi,
/\bdo\s+yourself\s+a\s+favou?r\s+and\s+(?:read|watch|check\s+out)\s+(?:this|it)\b/gi,
/\byou\s+(?:really\s+)?(?:won['’]?t|do\s*n['’]?t|will\s+not|do\s+not)\s+want\s+to\s+miss\s+this(?:\s+one)?(?=\s*(?:[:.!\n]|$))/gi,
/(?<=^|[,.!?:\n]\s{0,4})(?:you\s+can\s+)?thank\s+me\s+later\b/gim,
/(?<=^|[.!?:\n]\s{0,4})save\s+this\s+(?:one\s+)?for\s+later\b/gim,
/\bbookmark\s+this(?:\s+(?:one|post|thread))?(?=\s*(?:[:.!\n]|$))/gi,
/\bdo\s*n['’]?t\s+sleep\s+on\s+this\b/gi,
/\btrust\s+me,?\s+(?:on\s+this|you['’]?ll)\b/gi,
];
// ─── Unnecessary hyphenation (#107) ───────────────────────────────
// Precision-first subclasses only. The general question of whether a
// compound modifier is established English needs editorial judgment and
// remains in SKILL.md; the engine covers only curated open/closed forms and
// compounds whose surrounding syntax makes the unhyphenated form clear.
const UNNECESSARY_HYPHENATION = [
// Welded open noun phrases reported in #107. Match the complete phrase so
// a project-specific spelling of the pair in another role is not swept in.
{ pattern: /\bresearch-impact\s+aggregat(?:or|ion)s?\b/g, suggestion: (match) => match.replace('research-impact', 'research impact') },
{ pattern: /\bdata-source\s+strateg(?:y|ies)\b/g, suggestion: (match) => match.replace('data-source', 'data source') },
{ pattern: /\bPython-package\s+usage\b/g, suggestion: (match) => match.replace('Python-package', 'Python package') },
{ pattern: /\bRust-crate\s+usage\b/g, suggestion: (match) => match.replace('Rust-crate', 'Rust crate') },
{ pattern: /\bsingle-Project\s+Manifest\b/g, suggestion: (match) => match.replace('single-Project', 'single Project') },
{ pattern: /\btotal-downloads\s+figures?\b/g, suggestion: (match) => match.replace('total-downloads', 'total downloads') },
{ pattern: /\blife-sciences-native\s+citation\s+count\b/g, suggestion: 'citation count from a life sciences source' },
// Compounds whose standard spelling is closed. Kept as a small curated
// list rather than guessing that every noun-noun pair should close up.
{ pattern: /\bcode-base\b/g, suggestion: (match) => match.replace('-', '') },
{ pattern: /\bdata-set\b/g, suggestion: (match) => match.replace('-', '') },
{ pattern: /\btime-frame\b/g, suggestion: (match) => match.replace('-', '') },
{ pattern: /\broad-map\b/g, suggestion: (match) => match.replace('-', '') },
// Attributive-only forms used adverbially or as nouns. The boundary after
// real-time / long-term is intentionally narrow: "real-time analytics"
// and "long-term plan" must not fire.
{
pattern: /\bin\s+real-time(?=\s*(?:[,.!?;:]|$)|\s+(?:(?:across|as|automatically|because|but|continuously|during|dynamically|every|for|from|immediately|instantly|on|simultaneously|through|throughout|until|via|when|while|with|without)\b))/gi,
suggestion: 'in real time',
},
{
pattern: /\b(?:for|over)\s+the\s+long-term(?=\s*(?:[,.!?;:]|$)|\s+(?:across|because|but|by|during|for|from|on|through|throughout|until|via|when|while|with|without)\b)/gi,
suggestion: (match) => match.replace(/long-term/i, 'long term'),
},
{
pattern: /\b(?:functions?|functioned|functioning|operates?|operated|operating|runs?|ran|running|works?|worked|working)\s+out-of-the-box\b/gi,
suggestion: (match) => match.replace(/out-of-the-box/i, 'out of the box'),
},
];
// ═══ Helpers ═══════════════════════════════════════════════════════
function tokenize(text) {
return text.toLowerCase().match(/[\w'-]+/g) || [];
}
function countWords(text) {
return (text.match(/\S+/g) || []).length;
}
function getParagraphs(text) {
return text.split(/\n\s*\n/).filter(p => p.trim().length > 0);
}
function getSentences(text) {
return text.split(/[.!?]+/).filter(s => s.trim().length > 5);
}
function matchPatterns(text, patterns, category, severity) {
const issues = [];
for (const pat of patterns) {
const regex = new RegExp(pat.source, pat.flags);
let match;
while ((match = regex.exec(text)) !== null) {
issues.push({
type: category,
text: match[0],
index: match.index,
severity,
suggestion: null,
});
}
}
return issues;
}
// ═══ Main analysis ═════════════════════════════════════════════════
// Upper bound for one scan. Above this we bail rather than running all
// regex passes over a huge buffer — protects page perf on pasted novels.
const MAX_WORDS = 10000;
// V2 contract defaults so early-exit paths (Empty/tooShort/tooLong)
// still return the same field shape a v2 consumer expects. Without
// this, `result.document_classification === 'AI_ONLY'` is `undefined`
// on edge inputs and fails open.
// UNSCORED is returned on empty / too-short / too-long inputs where
// we declined to score. Distinct from HUMAN_ONLY (which is a positive
// classification) so a caller can't mistake a refused scan for a
// confident human verdict — a 50k-word LLM-generated document is
// not "human", it's just outside our scoring window.
function buildV2Defaults(classification, confidence) {
const probs = classification === 'HUMAN_ONLY'
? { human: 1, mixed: 0, ai: 0 }
: classification === 'AI_ONLY'
? { human: 0, mixed: 0, ai: 1 }
: { human: 0.333, mixed: 0.334, ai: 0.333 };
return {
document_classification: classification,
class_probabilities: probs,
confidence_category: confidence,
highlight_sentence_for_ai: [],
};
}
function analyzeText(text, options = {}) {
if (!text || text.trim().length === 0) {
return { ...buildV2Defaults('UNSCORED', 'low'), score: 0, label: 'Empty', issues: [], stats: {}, tooShort: true };
}
// Context mode gates rules that are noisy in technical writing. Modes:
// 'general' (default) — full ruleset
// 'technical' — skip title-case headers, formulaic openers gated to
// prose-only structures; lower em-dash + formatting weights
// 'marketing' — full ruleset + boost on formulaic-opener / future-narrative
// 'personal' — full ruleset, normal weights
// Mode is purely a soft gate; nothing is silently suppressed without
// being reflected in stats.contextMode for transparency.
// Mode validation: an unknown string (e.g. typo "tecnical") would
// otherwise silently downgrade to general-mode behavior. Coerce to
// 'general' and surface the original value in stats for traceability.
const VALID_CONTEXT_MODES = new Set(['general', 'technical', 'marketing', 'personal']);
const requestedMode = options.contextMode || 'general';
const contextMode = VALID_CONTEXT_MODES.has(requestedMode) ? requestedMode : 'general';
const contextModeFallback = requestedMode !== contextMode ? requestedMode : null;
// Source mode controls which parts of a Markdown file count as prose.
// Plain remains the compatibility default. Rendered Markdown masks only
// initial YAML frontmatter and HTML comments; source-hygiene checks for
// hidden TODO/placeholder comments remain available through plain mode.
const VALID_SOURCE_MODES = new Set(['plain', 'rendered-markdown']);
const requestedSourceMode = options.sourceMode === undefined ? 'plain' : options.sourceMode;
const sourceMode = VALID_SOURCE_MODES.has(requestedSourceMode) ? requestedSourceMode : 'plain';
const sourceModeFallback = requestedSourceMode !== sourceMode ? requestedSourceMode : undefined;
let maskedFrontmatter = 0;
let maskedHtmlComments = 0;
if (sourceMode === 'rendered-markdown') {
const rendered = maskRenderedMarkdown(text);
text = rendered.text;
maskedFrontmatter = rendered.maskedFrontmatter;
maskedHtmlComments = rendered.maskedHtmlComments;
}
// Pre-pass: mask Markdown blockquotes before scoring. A human
// reacting to AI text by quoting it shouldn't have the quoted block
// counted against their own writing. Requires ≥2 consecutive `> `
// lines to count as a blockquote — single-line `> ls -la` shell
// prompts in technical docs stay in the text. Masking instead of deleting
// keeps later issue and highlight offsets aligned with the source file.
const blockquotes = sourceMode === 'rendered-markdown'
? maskMultilineBlockquotes(text)
: stripMultilineBlockquotes(text);
text = blockquotes.text;
const { quotedLines } = blockquotes;
// Pre-pass: strip bypass-trick chars before pattern matching so
// "delve" with a Cyrillic 'е' still hits Tier 1. Original text is
// preserved so reported `match.index` values remain visually accurate.
const norm = normalizeText(text);
text = norm.text;
const wordCount = countWords(text);
if (wordCount < 10) {
return {
...buildV2Defaults('UNSCORED', 'low'),
score: 0,
label: 'Too short',
issues: [],
stats: { wordCount, contextMode, contextModeFallback, sourceMode, sourceModeFallback, maskedFrontmatter, maskedHtmlComments },
tooShort: true,
};
}
if (wordCount > MAX_WORDS) {
return {
...buildV2Defaults('UNSCORED', 'low'),
score: 0,
label: 'Text too long',
issues: [],
stats: { wordCount, contextMode, contextModeFallback, sourceMode, sourceModeFallback, maskedFrontmatter, maskedHtmlComments },
tooLong: true,
};
}
const tokens = tokenize(text);
const paragraphs = getParagraphs(text);
const sentences = getSentences(text);
const issues = [];
let rawScore = 0;
// ── 1. Tier 1 words ──────────────────────────────────────────
const tier1Found = new Set();
for (const token of tokens) {
if (Object.hasOwn(TIER1, token) && !tier1Found.has(token)) {
tier1Found.add(token);
issues.push({
type: 'tier1',
text: token,
severity: 'high',
suggestion: TIER1[token],
});
}
}
// Tier 1 multi-word phrases. Adds each distinct phrase (lowercased) to
// `tier1Found` so the same phrase hit multiple times only produces one
// issue — matches the downstream dedup behavior.
for (const phrase of TIER1_PHRASES) {
const regex = new RegExp(phrase.pattern.source, phrase.pattern.flags);
let match;
while ((match = regex.exec(text)) !== null) {
const lower = match[0].toLowerCase();
if (tier1Found.has(lower)) continue;
tier1Found.add(lower);
issues.push({
// Clarity-band entries are wordiness edits, not frequency evidence.
// Same fix, weaker claim — see the Tier 1A/1B split in SKILL.md.
type: phrase.clarity ? 'tier1-clarity' : 'tier1',
text: match[0],
severity: phrase.clarity ? 'medium' : 'high',
suggestion: phrase.replace,
});
}
}
// ── 2. Tier 2 clusters ───────────────────────────────────────
let tier2Clusters = 0;
for (const para of paragraphs) {
const paraTokens = tokenize(para);
const found = [];
const suggestions = {};
for (const token of paraTokens) {
if (Object.hasOwn(TIER2, token) && !found.includes(token)) {
found.push(token);
suggestions[token] = TIER2[token];
}
}
for (const cond of TIER2_CONDITIONAL) {
if (!found.includes(cond.word) && cond.pattern.test(para)) {
found.push(cond.word);
suggestions[cond.word] = cond.suggestion;
}
}
if (found.length >= 2) {
tier2Clusters++;
for (const word of found) {
issues.push({
type: 'tier2',
text: word,
severity: 'medium',
suggestion: suggestions[word],
});
}
}
}
// ── 3. Tier 3 density ────────────────────────────────────────
const tier3Counts = {};
for (const token of tokens) {
const canonical = TIER3_LOOKUP.get(token);
if (canonical) tier3Counts[canonical] = (tier3Counts[canonical] || 0) + 1;
}
// Flag at 3% of word count, but never below 3 occurrences. Previous
// floor of 1 meant a 50-word text with one "significant" got flagged
// as Tier 3 overuse, which was noise.
const densityThreshold = Math.max(3, Math.floor(wordCount * 0.03));
let tier3Flags = 0;
for (const [word, count] of Object.entries(tier3Counts)) {
if (count >= densityThreshold) {
tier3Flags++;
issues.push({
type: 'tier3',
text: `"${word}" x${count}`,
severity: 'low',
suggestion: `Overused (${count} times in ${wordCount} words)`,
});
}
}
// ── 4–21. Pattern categories ─────────────────────────────────
issues.push(...matchPatterns(text, TRANSITIONS, 'transition', 'medium'));
issues.push(...matchPatterns(text, CHATBOT_ARTIFACTS, 'chatbot', 'critical'));
issues.push(...matchPatterns(text, SYCOPHANTIC, 'sycophantic', 'critical'));
issues.push(...matchPatterns(text, FILLERS, 'filler', 'medium'));
issues.push(...matchPatterns(text, GENERIC_CONCLUSIONS, 'generic-conclusion', 'medium'));
issues.push(...matchPatterns(text, LETS_PATTERNS, 'lets-construction', 'medium'));
issues.push(...matchPatterns(text, REASONING_ARTIFACTS, 'reasoning-artifact', 'critical'));
issues.push(...matchPatterns(text, ACKNOWLEDGMENT_LOOPS, 'acknowledgment-loop', 'medium'));
issues.push(...matchPatterns(text, SIGNIFICANCE_INFLATION, 'significance-inflation', 'high'));
issues.push(...matchPatterns(text, VAGUE_ATTRIBUTIONS, 'vague-attribution', 'critical'));
issues.push(...matchPatterns(text, HOLLOW_INTENSIFIERS, 'hollow-intensifier', 'medium'));
issues.push(...matchPatterns(text, EMOTIONAL_FLATLINE, 'emotional-flatline', 'low'));
issues.push(...matchPatterns(text, LINGERING_ATTENTION, 'lingering-attention', 'medium'));
issues.push(...matchPatterns(text, NOVELTY_INFLATION, 'novelty-inflation', 'medium'));
issues.push(...matchPatterns(text, CUTOFF_DISCLAIMERS, 'cutoff-disclaimer', 'critical'));
issues.push(...matchPatterns(text, AI_PLACEHOLDERS, 'ai-placeholder', 'critical'));
issues.push(...matchPatterns(text, AI_CITATION_MARKUP, 'ai-citation-markup', 'critical'));
issues.push(...matchPatterns(text, AI_UTM_SOURCE, 'ai-utm-source', 'critical'));
issues.push(...matchPatterns(text, TEMPLATE_PHRASES, 'template-phrase', 'high'));
issues.push(...matchPatterns(text, FALSE_CONCESSION, 'false-concession', 'medium'));
issues.push(...matchPatterns(text, RHETORICAL_QUESTIONS, 'rhetorical-question', 'medium'));
issues.push(...matchPatterns(text, HEDGE_STACK, 'hedge-stack', 'high'));
issues.push(...matchPatterns(text, FUTURE_NARRATIVE, 'future-narrative', 'high'));
issues.push(...matchPatterns(text, REAL_ACTUAL_INFLATION, 'real-actual-inflation', 'medium'));
issues.push(...matchPatterns(text, SOCIAL_CTA_CLOSER, 'social-cta-closer', 'high'));
issues.push(...matchPatterns(text, PERFORMED_INSIGHT, 'performed-insight', 'medium'));
issues.push(...matchPatterns(text, NEGATION_CHAIN, 'negation-chain', 'high'));
issues.push(...matchPatterns(text, DEV_BLOG_BOILERPLATE, 'dev-blog-boilerplate', 'medium'));
issues.push(...findUnnecessaryHyphenation(text));
// ── Tier 1 v2: formulaic openers + parenthetical hedges ──────────
issues.push(...matchPatterns(text, FORMULAIC_OPENERS, 'formulaic-opener', 'high'));
issues.push(...matchPatterns(text, SPECULATIVE_OPENERS, 'speculative-opener', 'high'));
issues.push(...matchPatterns(text, LAUNCH_INTROS, 'launch-intro', 'high'));
issues.push(...matchPatterns(text, CROWD_CONTRAST, 'crowd-contrast', 'medium'));
issues.push(...matchPatterns(text, FAKE_CASUAL_PROPS, 'fake-casual-prop', 'high'));
issues.push(...matchPatterns(text, PARENTHETICAL_HEDGE, 'parenthetical-hedge', 'medium'));
// Title-case headers — gated to marketing/personal/general modes
// (technical mode legitimately uses Title Case section headers).
if (contextMode !== 'technical') {
const titleHits = matchPatterns(text, [TITLE_CASE_HEADER], 'title-case-header', 'medium');
// Drop matches that look like proper-noun titles (single line, all
// tokens capitalized incl. function words) — that's headline style,
// not the AI-section-header tell which has mid-sentence "And".
//
// The prefix strip is load-bearing. matchPatterns reports match[0], so a
// Markdown hit arrives as "## Terms Of Service" and `##` counts as a
// token — silently lowering this guard from four content words to three
// for headings only, which is exactly the class it exists to protect.
// "## Terms Of Service", "## Bank Of America" and "## Table Of Contents"
// all flagged as a result: ordinary human headings, on a detector whose
// stated first priority is not firing on human writing.
const filtered = titleHits.filter((h) => {
const title = h.text.replace(MD_HEADING_PREFIX, '');
const tokens = title.trim().split(/\s+/);
if (tokens.length < 4) return false;
// The function word must be MID-title, which is what the comment above
// has always said and what the test never enforced. A leading "The"
// satisfied a bare /\bThe\b/, so ordinary human headings flagged:
// "## The New Security Landscape", "## The Microsoft Approach to
// Identity", "### The Four Keys to a Successful and Secure Modern
// Workplace". Measured across 81 files that provably predate LLMs
// (2018-19 eBooks, 2020 posts): 13 false positives, every one opening
// with "The", against zero on main.
//
// "## Benefits And Strategic Considerations" -- the actual tell, and
// this rule's own fixture -- is untouched: its "And" is interior.
return FUNCTION_WORD.test(tokens.slice(1).join(' '));
});
const fences = filtered.length ? fenceRanges(text) : [];
issues.push(...filtered.filter((h) => !inFenceRange(fences, h.index)));
}
// ── Normalization-trigger flag ───────────────────────────────────
// ZWSPs or homoglyphs in pasted prose are near-dispositive: humans
// don't insert these into their own writing. Single roleplay marker
// can be a false positive on Markdown emphasis (filtered to multi-
// word inner already), so requires ≥2.
if (norm.flags.zeroWidth > 0 || norm.flags.homoglyph >= 2) {
issues.push({
type: 'normalization-flag',
text: `${norm.flags.zeroWidth} zero-width + ${norm.flags.homoglyph} homoglyph swap${norm.flags.homoglyph === 1 ? '' : 's'}`,
severity: 'critical',
suggestion: 'Text contains invisible/lookalike chars typical of AI-humanizer bypass tools. Re-type from your own keyboard.',
});
}
if (norm.flags.roleplay >= 2) {
issues.push({
type: 'normalization-flag',
text: `${norm.flags.roleplay} *roleplay-action* markers stripped`,
severity: 'high',
suggestion: 'Paired *action* markers are a chat-model artifact.',
});
}
// Em dashes in list-item separator position — a bulleted or numbered
// list item opening with a bolded lead term or markdown link, then the
// dash ("- **Term** — desc", "- [label](url) — desc") — are
// definition-list typography, not prose punctuation. Shared by the
// smart-punct signature below and the em-dash frequency check (§22).
// An optional parenthetical or inline-code span may sit between the bold
// lead term and the dash — "- **Lingering-attention claims**
// (`lingering-attention`) — the share-post frame…" is the same definition
// typography as the bare form. Found by the self-scan (see PROOF.md, #67).
const SEPARATOR_DASH_RE = /^\s*(?:[-*+]|\d+[.)])\s+(?:\*\*[^*\n]+\*\*|\[[^\]\n]+\]\([^)\n]*\))(?:[ \t]*(?:\([^)\n]*\)|`[^`\n]+`))?[ \t]*—/gm;
// Keep-a-Changelog version headings (`## [3.21.0] — 2026-07-30`) join a
// label to a value exactly as a list separator does. Deliberately narrow:
// a bracketed or bare semver token, then a dash, then an ISO date, and
// nothing else on the line. Ordinary prose dashes in headings still count,
// because SKILL.md applies the em-dash rule to headings too.
const VERSION_HEADING_DASH_RE = /^#{1,6}[ \t]+\[?v?\d+\.\d+\.\d+[^\]\n]*\]?[ \t]*—[ \t]*\d{4}-\d{2}-\d{2}[ \t]*$/gm;
// ── Smart-punctuation co-occurrence signature ────────────────────
// Curly quotes + em-dash + Oxford comma all present + zero typos
// (no double-spaces, no missing apostrophes in common contractions)
// is a near-dispositive paste-from-LLM signature: humans typing
// directly into a textarea don't produce all four. Standalone any of
// these is meaningless — co-occurrence is the signal. Separator-position
// dashes are typography and don't corroborate it.
{
const hasCurly = /[“”‘’]/.test(text);
const totalEmDashes = (text.match(/—/g) || []).length;
const separatorEmDashes = (text.match(SEPARATOR_DASH_RE) || []).length
+ (text.match(VERSION_HEADING_DASH_RE) || []).length;
const hasEmDash = totalEmDashes > separatorEmDashes;
const oxfordHit = text.match(/\b\w+,\s+\w+,\s+and\s+\w+/g);
const hasOxford = (oxfordHit?.length || 0) >= 1;
const doubleSpaces = (text.match(/[^.!?] +/g) || []).length;
const missingApos = /\b(?:dont|wont|cant|isnt|wasnt|shouldnt|wouldnt|couldnt|youre|theyre|its\s+a\s+\w+ing)\b/i.test(text);
const clean = doubleSpaces === 0 && !missingApos;
const signals = [hasCurly, hasEmDash, hasOxford, clean].filter(Boolean).length;
if (signals >= 4 && wordCount >= 80) {
issues.push({
type: 'smart-punct-signature',
text: 'curly-quotes + em-dash + Oxford comma + zero typos',
severity: 'high',
suggestion: 'Smart-punctuation signature consistent with LLM output. Humans typing into textareas rarely produce all four.',
});
}
}
// ── Punctuation distribution mode ────────────────────────────────
// Humans cluster trimodal across paragraphs (some paras heavy, some
// light, some none). AI converges on a normal distribution. We can't
// run a real modality test client-side, but we can flag the AI
// signature: low variance of per-paragraph punctuation density.
// Requires ≥4 paragraphs to be meaningful.
if (paragraphs.length >= 4) {
const densities = paragraphs.map((p) => {
const words = (p.match(/\S+/g) || []).length;
if (words < 5) return null;
const puncts = (p.match(/[,;:—()]/g) || []).length;
return puncts / words;
}).filter((d) => d !== null);
if (densities.length >= 4) {
const mean = densities.reduce((a, b) => a + b, 0) / densities.length;
const variance = densities.reduce((s, d) => s + (d - mean) ** 2, 0) / densities.length;
const cv = mean > 0 ? Math.sqrt(variance) / mean : 0;
// CV < 0.25 across paragraphs means each paragraph has the same
// punctuation density — the AI signature. Humans usually swing
// wider. Threshold derived from stylometry papers (arxiv 2507.00838).
if (cv < 0.25 && mean >= 0.04) {
issues.push({
type: 'punct-distribution',
text: `Punctuation density uniform across paragraphs (CV=${cv.toFixed(2)})`,
severity: 'medium',
suggestion: 'AI text holds punctuation density steady; human writers swing between dense and sparse paragraphs.',
});
}
}
}
// ── Function-word trigram entropy ────────────────────────────────
// POS-trigram entropy is the academic signal; function-word trigram
// entropy approximates it without a tagger (function words ARE the
// closed-class POS classes). AI text has lower entropy because LLM
// sampling collapses onto a narrower set of grammatical templates.
//
// Method: extract function-word indicators per sentence, build
// trigrams over the sequence, compute Shannon entropy. Bins below
// threshold flag.
if (wordCount >= 150) {
const FUNC_WORDS = new Set([
'the','a','an','and','or','but','of','to','in','on','at','by','for','with',
'from','as','is','was','are','were','be','been','being','have','has','had',
'do','does','did','will','would','should','could','may','might','must','can',
'this','that','these','those','it','its','they','them','their','there','here',
'we','our','us','i','you','your','he','she','his','her','him','not','no','so',
'if','then','than','when','where','which','who','what','how','why','because',
]);
const seq = tokens.map((t) => FUNC_WORDS.has(t) ? t : '_').filter((_, i, arr) => arr[i] !== '_' || (i > 0 && arr[i - 1] !== '_'));
if (seq.length >= 50) {
const trigrams = {};
for (let i = 0; i < seq.length - 2; i++) {
const tg = `${seq[i]}|${seq[i + 1]}|${seq[i + 2]}`;
trigrams[tg] = (trigrams[tg] || 0) + 1;
}
const total = seq.length - 2;
let entropy = 0;
for (const c of Object.values(trigrams)) {
const p = c / total;
entropy -= p * Math.log2(p);
}
// Normalize by log2(distinct trigrams) so entropy ranges roughly
// 0..1 and threshold is interpretable. Empirical threshold: human
// prose ~0.85-0.95 normalized, AI prose ~0.70-0.82.
const distinctCount = Object.keys(trigrams).length;
const normalized = distinctCount > 1 ? entropy / Math.log2(distinctCount) : 1;
if (normalized < 0.82 && total >= 50) {
issues.push({
type: 'fnword-trigram-entropy',
text: `Function-word trigram entropy ${normalized.toFixed(2)} (low)`,
severity: 'medium',
suggestion: 'Grammatical structure is unusually repetitive. AI sampling collapses onto narrower templates than human writing.',
});
}
// Degenerate case: single distinct trigram repeated across the
// whole document is the strongest possible AI signal but the
// normalized fallback returns 1.0 (= "fully human"), inverting
// the signal. Catch it explicitly.
if (distinctCount === 1 && total >= 50) {
issues.push({
type: 'fnword-trigram-entropy',
text: 'Single function-word trigram repeated across document',
severity: 'high',
suggestion: 'Grammatical structure is fully degenerate — every clause uses the same function-word skeleton.',
});
}
}
}
// ── Cross-paragraph burstiness ───────────────────────────────────
// We already check within-paragraph sentence-length uniformity. AI
// is also flat ACROSS paragraphs — every paragraph has roughly the
// same sentence-length variance. Humans vary: terse paras next to
// discursive paras. Measure variance of CV across paragraphs.
if (paragraphs.length >= 4) {
const cvs = paragraphs.map((p) => {
const sents = getSentences(p);
if (sents.length < 3) return null;
const lens = sents.map(countWords);
const m = lens.reduce((a, b) => a + b, 0) / lens.length;
if (m === 0) return null;
const v = lens.reduce((s, l) => s + (l - m) ** 2, 0) / lens.length;
return Math.sqrt(v) / m;
}).filter((c) => c !== null);
if (cvs.length >= 4) {
const cvMean = cvs.reduce((a, b) => a + b, 0) / cvs.length;
const cvVar = cvs.reduce((s, c) => s + (c - cvMean) ** 2, 0) / cvs.length;
const cvStd = Math.sqrt(cvVar);
// Std-of-CV below 0.08 means every paragraph has roughly the same
// internal rhythm — AI signature. Human prose typically swings
// 0.15-0.40 across paragraphs of mixed purpose.
if (cvStd < 0.08 && cvMean < 0.45) {
issues.push({
type: 'cross-para-burstiness',
text: `Sentence-rhythm uniform across paragraphs (σCV=${cvStd.toFixed(2)})`,
severity: 'medium',
suggestion: 'Every paragraph has the same internal rhythm. Humans vary cadence between terse and discursive paragraphs.',
});
}
}
}
// ── Tier 3 multi-word phrase density ─────────────────────────
// Two complementary rules:
// (a) Per-phrase density — same gating as single-word Tier 3: each
// phrase fine alone, repetition is the tell. Threshold = 2.
// (b) Cross-phrase clustering — ≥3 *distinct* boilerplate phrases
// in one piece. LLMs varying their own boilerplate often use
// each phrase only once but stack 5-10 across the text. The
// per-phrase rule misses this; the cluster rule catches it.
// Track non-overlapping match spans so a longer phrase swallowing a
// shorter one (e.g., "designed for long-term sustainability" matches
// both "designed for long-term" AND "long-term sustainability") only
// contributes one distinct hit. Without dedup the cluster threshold
// can be reached by a single sentence stacking overlapping regexes.
const claimedSpans = [];
function spanOverlaps(start, end) {
for (const [s, e] of claimedSpans) {
if (start < e && end > s) return true;
}
return false;
}
let distinctPhrasesHit = 0;
for (const phrase of TIER3_PHRASES) {
const regex = new RegExp(phrase.source, phrase.flags);
const phraseSpans = [];
let phraseMatch;
while ((phraseMatch = regex.exec(text)) !== null) {
const start = phraseMatch.index;
const end = start + phraseMatch[0].length;
if (!spanOverlaps(start, end)) {
phraseSpans.push([start, end, phraseMatch[0]]);
}
}
if (phraseSpans.length === 0) continue;
for (const [s, e] of phraseSpans) claimedSpans.push([s, e]);
distinctPhrasesHit++;
if (phraseSpans.length >= 2) {
issues.push({
type: 'tier3-phrase',
text: `"${phraseSpans[0][2].toLowerCase()}" x${phraseSpans.length}`,
severity: 'medium',
suggestion: `Boilerplate phrase repeated ${phraseSpans.length}× — replace at least one with specifics`,
});
}
}
if (distinctPhrasesHit >= 3) {
issues.push({
type: 'tier3-phrase-cluster',
text: `${distinctPhrasesHit} distinct boilerplate phrases`,
severity: 'high',
suggestion: 'Several stock crypto/web3 phrases stacked in one piece. Rewrite around one specific claim or observation.',
});
}
// ── Hashtag stuffing ─────────────────────────────────────────
// 6+ hashtags in a single post is rare for thoughtful humans and
// near-universal for LLM-generated social posts. Counted globally,
// not per-paragraph, since the trailing hashtag block is the shape
// we care about.
// Match #tag at start of text or after any non-word char (whitespace,
// punctuation, line breaks). URL fragments are already excluded
// because the char immediately before `#` in a URL path is always
// a word char (e.g. `example.com/page#section` — `e` before `#`).
// Earlier char class `[\s\\]` had a literal backslash and silently
// missed hashtags after sentence punctuation; an interim `[\s]` fix
// on origin only caught whitespace-preceded tags.
// Code is masked and non-tag `#` forms are subtracted first — see maskCode
// and isSocialTag. Without them a changelog paragraph citing six issue
// numbers, or a palette listing six hex colours, scored as a tag block.
const hashtagMatches = [...maskCode(text).matchAll(/(?:^|\W)#(\w[\w-]*)/g)]
.filter((m) => isSocialTag(m[1]));
if (hashtagMatches.length >= 6) {
issues.push({
type: 'hashtag-stuff',
text: `${hashtagMatches.length} hashtags`,
severity: 'medium',
suggestion: 'Cut to 2-3 specific tags or none. Long hashtag blocks read as bot output.',
});
}
// ── Bullet list of bare noun phrases ─────────────────────────
// ≥5 consecutive bullet items that are short (≤6 words) and contain
// no finite-verb / modal token. Catches the "Stable mining efficiency
// / Reliable pool connectivity / Optimized RandomX performance ..."
// shape LLMs default to. Markdown bullets, escaped Markdown bullets,
// unicode bullets, and dashes are all matched. Numbered lists are
// excluded — those have a separate "numbered list inflation" rule.
//
// Note: verbRe covers auxiliaries and modals ("was", "will", "can",
// etc.). Regular past-tense verbs ("fixed", "removed") are not
// matched here; instead, the ≤6-word length gate excludes most
// real-world changelog lines, which tend to read "fixed the X that
// was doing Y" (>6 words). Short two-word action items ("* fixed
// bug") would pass both gates — an acceptable trade-off to avoid
// false-negative risk from adjectives ending in -ed ("skilled",
// "advanced") that share the same surface form.
const lines = text.split(/\r?\n/);
const bulletRe = /^\s*(?:\*|-|•|\+)\s+(.+)$/;
const verbRe = /\b(?:is|are|was|were|has|have|had|will|would|should|must|do|does|did|can|could|may|might|am|been|being)\b/i;
const fenceRe = /^\s*(?:```|~~~)/;
let run = [];
let blankStreak = 0;
let inFence = false;
function flushRun() {
if (run.length >= 5) {
const bareNP = run.filter((it) => {
const wc = (it.match(/\S+/g) || []).length;
return wc > 0 && wc <= 6 && !verbRe.test(it);
});
if (bareNP.length >= 5 && bareNP.length / run.length >= 0.75) {
issues.push({
type: 'bullet-np-list',
text: `${run.length}-item bullet list of bare noun phrases`,
severity: 'high',
suggestion: 'Convert to a prose paragraph or merge items. Long lists of bare adj+noun pairs read as AI scaffolding.',
});
}
}
run = [];
blankStreak = 0;
}
for (const line of lines) {
if (fenceRe.test(line)) {
// Code-fence toggle. Bullets inside fences are CLI flag docs or
// option dumps, not prose AI scaffolding — flush any prose run
// we were tracking and skip until the fence closes.
flushRun();
inFence = !inFence;
continue;
}
if (inFence) continue;
const m = line.match(bulletRe);
if (m) {
run.push(m[1].trim());
blankStreak = 0;
} else if (line.trim() === '') {
// A single blank line inside a list is normal Markdown spacing;
// two or more blank lines break the run, since visually-disjoint
// bullet sections shouldn't merge into one logical list.
blankStreak++;
if (blankStreak >= 2) flushRun();
} else {
flushRun();
}
}
flushRun();
// NOTE: "Wall-of-text replies" (SKILL.md) is deliberately NOT a
// detector rule here. A first pass tried "reply-length text, >=4
// sentences, zero newlines" as a structural gate — it broke the
// "repeated Tier 1 phrase does not inflate score linearly" fixture
// and, on reflection, would fire on any ordinary short paragraph
// (a blog intro, a single-paragraph email) since "one paragraph with
// no internal line break" is simply what continuous prose looks
// like, not an AI-specific shape. The tell in SKILL.md depends on
// knowing the text is conversational-reply register in the first
// place, which the engine can't reliably infer from the bytes alone
// (CONTRIBUTING.md: "a signal that fires on most normal prose is not
// worth adding"). Left as an LLM-judgment rule; see CATEGORIES.md §C.
// Confidence calibration is only flagged when it stacks (3+ instances).
// Gating happens pre-dedup on raw match count, since that signals actual
// stacking, not just vocabulary use.
const confIssues = matchPatterns(text, CONFIDENCE_CALIBRATION, 'confidence-calibration', 'low');
if (confIssues.length >= 3) issues.push(...confIssues);
// ── 22. Em dash frequency ────────────────────────────────────
// Match real em dashes, plus `--` only when surrounded by whitespace on at
// least one side (skips CLI flags like --save-dev and YAML `---` blocks).
// Separator-position em dashes (SEPARATOR_DASH_RE above) are excluded
// from the rate. The list marker is required on purpose: a line-initial
// "**Bold lead** — full sentence" outside a list is itself an AI tell
// and still counts, as does a mid-sentence "**bold** — like this"
// splice. Em dash only — the `--` substitute is never carved out.
const rawEmDashCount = (text.match(/—|(?<=\s)--(?=\s|$)|(?<=^|\s)--(?=\s)/gm) || []).length;
const separatorDashCount = (text.match(SEPARATOR_DASH_RE) || []).length
+ (text.match(VERSION_HEADING_DASH_RE) || []).length;
const emDashCount = rawEmDashCount - separatorDashCount;
const emDashRate = emDashCount / (wordCount / 1000);
if (emDashRate > 1) {
issues.push({
type: 'em-dash',
text: `${emDashCount} em dashes in ${wordCount} words`,
severity: 'medium',
suggestion: 'Replace with commas, periods, or rewrite',
});
}
// ── 23. Sentence length uniformity ───────────────────────────
if (sentences.length >= 5) {
const lengths = sentences.map(s => countWords(s));
const avg = lengths.reduce((a, b) => a + b, 0) / lengths.length;
const variance = lengths.reduce((sum, l) => sum + Math.pow(l - avg, 2), 0) / lengths.length;
const stdDev = Math.sqrt(variance);
const cv = avg > 0 ? stdDev / avg : 0;
if (cv < 0.25 && avg > 10) {
issues.push({
type: 'uniformity',
text: `Sentence lengths cluster around ${Math.round(avg)} words (low variation)`,
severity: 'medium',
suggestion: 'Mix short punchy sentences with longer flowing ones',
});
}
}
// ── Type-token ratio (stylometric — vocabulary diversity) ────
// TTR = distinct word types / total tokens. Human prose at 200+
// words typically sits around 0.50–0.65 for English; AI prose
// tends flatter (0.55–0.75 looks normal, but the lower end of the
// *too-flat* tail at >=200 words is where the signal lives — too
// FEW unique words for the length). This is the simplest of the
// four stylometric signals identified in the May 2026 detection-
// research review (docs/competitive/detection-research.md): no
// POS tagger required, no model, pure JS.
//
// Threshold tuning: flag only when the sample is large enough
// that low TTR is meaningfully suspicious (>=200 tokens) AND TTR
// is below 0.40 (very vocabulary-poor). Conservative on purpose;
// false positives on short or topic-narrow human prose are easy
// to trigger and would drown out other signals. The detector-
// research lens flagged TTR as one of four stylometric add-ons;
// POS-bigram log-odds, function-word z-scores, and sentence-
// length burstiness are still TODO.
if (tokens.length >= 200) {
const unique = new Set(tokens).size;
const ttr = unique / tokens.length;
if (ttr < 0.4) {
issues.push({
type: 'low-ttr',
text: `Vocabulary diversity ${(ttr * 100).toFixed(1)}% (${unique} unique / ${tokens.length} tokens)`,
severity: 'low',
suggestion: 'Text reuses a narrow word set. Vary nouns and verbs deliberately, or check if the topic genuinely warrants the repetition.',
});
}
}
// ── 24. Paragraph length uniformity ──────────────────────────
if (paragraphs.length >= 4) {
const paraLengths = paragraphs.map(p => getSentences(p).length);
const avg = paraLengths.reduce((a, b) => a + b, 0) / paraLengths.length;
const allSimilar = paraLengths.every(l => Math.abs(l - avg) <= 1);
if (allSimilar && avg >= 3) {
issues.push({
type: 'uniformity',
text: `All paragraphs are ~${Math.round(avg)} sentences`,
severity: 'low',
suggestion: 'Vary paragraph length deliberately',
});
}
}
// ── 25. Bold overuse ─────────────────────────────────────────
const boldMatches = text.match(/\*\*[^*]+\*\*/g) || [];
if (boldMatches.length > 3) {
issues.push({
type: 'formatting',
text: `${boldMatches.length} bold phrases`,
severity: 'medium',
suggestion: 'Strip bold from most; restructure to lead with key info',
});
}
// ── Score from the deduped issue list ───────────────────────
// Previously rawScore was accumulated inline per pattern hit, so
// repeated hits of the same phrase (or overlapping matches) inflated
// the score while the displayed issue list was deduplicated. That
// produced the UX regression where a "heavy AI patterns" label sat
// above a list of two items. Now the dedup runs first, then each
// distinct issue contributes its category weight — so the number
// reflects the same signals the user actually sees.
const deduped = deduplicateIssues(issues);
for (const issue of deduped) {
rawScore += ISSUE_WEIGHTS[issue.type] ?? 2;
}
// Scale by text length: longer text gets more chances to trigger.
const lengthFactor = Math.max(1, Math.log2(wordCount / 50));
const normalizedScore = Math.min(100, Math.round(rawScore / lengthFactor));
const label = getLabel(normalizedScore);
// ── Sentence-region smoothing (HMM-style without an HMM) ─────────
// Map each text-bearing issue back to its sentence indexes, then
// merge adjacent flagged sentences into contiguous regions for UI
// highlighting. Borrowed from GPTZero's sentence-highlighting model
// — gives users "this paragraph is AI" rather than scattered hits.
const regions = buildSentenceRegions(text, deduped, sourceMode === 'rendered-markdown');
// Stats derived from the same deduped list so tier counts + patternCount
// sum to `deduped.length`. Previously patternCount subtracted
// `tier2Clusters` (a paragraph count) which produced inconsistent totals.
const tier1Count = deduped.filter((i) => i.type === 'tier1').length;
const tier2Count = deduped.filter((i) => i.type === 'tier2').length;
const tier3Count = deduped.filter((i) => i.type === 'tier3').length;
// ── Trinary classification (GPTZero-shaped) ──────────────────────
// Decouples confidence from AI-proportion. Maps the 0-100 score plus
// structural signals into HUMAN_ONLY / MIXED / AI_ONLY with a
// confidence band. Thresholds are FN-biased: ambiguity routes to
// MIXED, never AI_ONLY. Quote from GPTZero's design principle:
// "biases the detector to prefer making less-harmful false-negative
// errors over false-positive errors."
// Dense-AI-vocab trifecta: ≥5 distinct tier1 hits + ≥2 tier2 cluster
// paragraphs + ≥1 transition phrase, AND ≥150 words. Catches
// saturated ChatGPT prose without firing on dense-jargon human
// technical writing where the tier1 vocabulary (robust,
// comprehensive, leverage, ecosystem) legitimately overlaps with
// systems-programming idiom. Word-count gate prevents short ESL or
// contrived adversarial sentences from tripping the corroborator.
const tier1Distinct = new Set(deduped.filter((i) => i.type === 'tier1').map((i) => (i.text || '').toLowerCase())).size;
const hasTier2Cluster = tier2Clusters >= 2;
const hasTransition = deduped.some((i) => i.type === 'transition');
const denseAIVocab = wordCount >= 150 && tier1Distinct >= 5 && hasTier2Cluster && hasTransition;
const trinary = classifyTrinary({
score: normalizedScore,
issues: deduped,
regions,
normFlags: norm.flags,
wordCount,
denseAIVocab,
});
return {
// Legacy fields preserved for existing callers.
score: normalizedScore,
label,
issues: deduped,
stats: {
wordCount,
tier1Count,
tier2Count,
tier2Clusters,
tier3Count,
tier3Flags,
patternCount: deduped.length - tier1Count - tier2Count - tier3Count,
contextMode,
contextModeFallback,
sourceMode,
sourceModeFallback,
maskedFrontmatter,
maskedHtmlComments,
normalization: norm.flags,
quotedLines,
unmappedHighlights: regions._unmapped ?? 0,
denseAIVocab,
tier1Distinct,
},
// Trinary API — shape mirrors GPTZero so integrators can swap.
document_classification: trinary.classification,
class_probabilities: trinary.probabilities,
confidence_category: trinary.confidence,
highlight_sentence_for_ai: regions,
};
}
// ═══ Sentence regions + trinary classifier ═════════════════════════
function buildSentenceRegions(text, issues, trimBoundaryWhitespace = false) {
// Split text into sentences with source offsets preserved so the UI
// can highlight spans accurately. Sentence boundaries are coarse
// (.!?) — fine for highlighting, not for linguistic correctness.
const sentences = [];
const sentenceRe = /[^.!?]+[.!?]+|\S[^.!?]*$/g;
let m;
while ((m = sentenceRe.exec(text)) !== null) {
const sentenceText = m[0].trim();
if (sentenceText.length < 4) continue;
let start = m.index;
let end = m.index + m[0].length;
if (trimBoundaryWhitespace) {
start += m[0].search(/\S/);
end -= m[0].match(/\s*$/)[0].length;
}
sentences.push({ start, end, text: sentenceText });
}
if (sentences.length === 0) return [];
// Map issue.text back to sentence indexes via substring search. Two
// kinds of issue stay out of the AI-highlight regions. Summary signals
// like "Punctuation density uniform across paragraphs" have no sentence
// anchor — they contribute to the document-level signal but not to
// highlights. Zero-weight style copyedits (unnecessary-hyphenation) do
// have an anchor, but they are P2 grammar cleanup rather than evidence
// of machine authorship, so they belong in issues[] and nowhere near a
// field reserved for AI sentence highlights.
// Filter by issue TYPE not text-regex: text-based filtering used to
// drop legitimate phrase issues containing "across" / "density".
const NON_HIGHLIGHT_TYPES = new Set([
'punct-distribution',
'cross-para-burstiness',
'fnword-trigram-entropy',
'smart-punct-signature',
'normalization-flag',
'uniformity',
'em-dash',
'formatting',
'tier3',
'tier3-phrase',
'tier3-phrase-cluster',
'hashtag-stuff',
'bullet-np-list',
'unnecessary-hyphenation',
]);
const hits = sentences.map(() => ({ count: 0, weight: 0 }));
const lowerText = text.toLowerCase();
let unmappedHighlights = 0;
for (const issue of issues) {
if (!issue.text || issue.text.length > 200) continue;
if (NON_HIGHLIGHT_TYPES.has(issue.type)) continue;
const needle = issue.text.toLowerCase();
let idx = 0;
let matched = false;
while ((idx = lowerText.indexOf(needle, idx)) !== -1) {
matched = true;
for (let i = 0; i < sentences.length; i++) {
if (idx >= sentences[i].start && idx < sentences[i].end) {
hits[i].count++;
hits[i].weight += ISSUE_WEIGHTS[issue.type] ?? 2;
break;
}
}
idx += needle.length;
}
if (!matched) unmappedHighlights++;
}
// Window-merge contiguous flagged sentences. Allow 1 unflagged
// sentence gap between two flagged ones (the "smoothing" — keeps
// a single boring sentence from breaking what's clearly an AI
// passage). A sentence is "flagged" if it has ≥1 hit.
const regions = [];
let cur = null;
for (let i = 0; i < sentences.length; i++) {
if (hits[i].count > 0) {
if (cur === null) {
cur = { startSentence: i, endSentence: i, start: sentences[i].start, end: sentences[i].end, hitCount: hits[i].count, weight: hits[i].weight };
} else {
cur.endSentence = i;
cur.end = sentences[i].end;
cur.hitCount += hits[i].count;
cur.weight += hits[i].weight;
}
} else if (cur !== null) {
// Allow one-sentence gap.
const next = hits[i + 1];
if (next && next.count > 0) {
cur.endSentence = i;
cur.end = sentences[i].end;
continue;
}
regions.push(finalizeRegion(cur));
cur = null;
}
}
if (cur !== null) regions.push(finalizeRegion(cur));
// Expose the unmapped-highlight count via a non-enumerable property
// so the array length still reads naturally for consumers; the
// analyzer pulls it into stats.unmappedHighlights for diagnostics.
Object.defineProperty(regions, '_unmapped', { value: unmappedHighlights, enumerable: false });
return regions;
}
function finalizeRegion(r) {
// Map cumulative weight inside the region to a 0-1 score. Cap at 20
// weight = 1.0 (matches Heavy threshold density).
const score = Math.min(1, r.weight / 20);
return {
startSentence: r.startSentence,
endSentence: r.endSentence,
start: r.start,
end: r.end,
hitCount: r.hitCount,
score: Math.round(score * 100) / 100,
};
}
// FN-biased: false positives damage trust more than false negatives,
// so MIXED is wide and AI_ONLY requires multiple signals. Quote from
// GPTZero: "biases the detector to prefer making less-harmful
// false-negative errors over false-positive errors."
function classifyTrinary({ score, issues, regions, normFlags, wordCount, denseAIVocab }) {
// Strong corroborators — each is near-dispositive on its own:
// - cutoff-disclaimer (LLM self-identifies as an AI)
// - reasoning-artifact + chatbot-artifact co-occurrence
// - normalization-flag at threshold (≥2 ZWSP or homoglyphs).
// Threshold parity prevents a single stray ZWSP in copy-paste
// from Word/Notion from flipping to AI_ONLY at score 0.
// - denseAIVocab: ≥4 distinct tier1 hits AND ≥1 tier2 cluster AND
// ≥1 transition phrase — the trifecta that saturated ChatGPT
// prose triggers without needing whitelisted stylometric hits.
const hasCutoff = issues.some((i) => i.type === 'cutoff-disclaimer');
const hasNormFlag = normFlags.zeroWidth >= 2 || normFlags.homoglyph >= 2;
const hasReasoning = issues.some((i) => i.type === 'reasoning-artifact');
const hasChatbot = issues.some((i) => i.type === 'chatbot');
const strongCorrob =
(hasCutoff ? 1 : 0) +
(hasNormFlag ? 1 : 0) +
(hasReasoning && hasChatbot ? 1 : 0) +
(denseAIVocab ? 1 : 0);
// Weak (stylometric) corroborators — suggestive on their own,
// dispositive in combination. Smart-punct-signature matches
// Word-edited human prose so doesn't count without other support.
const stylometricHits = ['punct-distribution', 'cross-para-burstiness', 'fnword-trigram-entropy']
.filter((t) => issues.some((i) => i.type === t)).length;
const hasSmartPunct = issues.some((i) => i.type === 'smart-punct-signature');
const weakCorrob = (stylometricHits >= 2 ? 1 : 0) + (hasSmartPunct ? 1 : 0);
// Thresholds:
// score < 15 with no strong → HUMAN_ONLY
// strong ≥ 1 OR score ≥ 70 → AI_ONLY (lowered from 80; high
// density of AI vocab is sufficient evidence)
// score ≥ 40 with any corroborator → AI_ONLY
// everything else with score ≥ 15 → MIXED
const totalCorrob = strongCorrob + weakCorrob;
let classification;
if (score < 15 && strongCorrob === 0) classification = 'HUMAN_ONLY';
else if (strongCorrob >= 1 || score >= 70) classification = 'AI_ONLY';
else if (score >= 40 && totalCorrob >= 1) classification = 'AI_ONLY';
else classification = 'MIXED';
// Humanizer-flag escalation: presence of bypass-trick chars is
// adversarial signal. If a normalization-flag fired we already
// counted it in strongCorrob → AI_ONLY. Confidence also gets a
// floor of 'medium' in that case (an adversary actively evading
// detection should never read as low-confidence noise).
// Soft probability distribution. Not calibrated against a labeled
// corpus yet (TODO when corpus exists — see roadmap.md). Largest
// class is computed as `1 - others` after rounding to guarantee
// sum=1 exactly. Sub-1% drift would otherwise hide in toFixed.
const aiSoft = Math.min(0.97, score / 100 + totalCorrob * 0.06 + strongCorrob * 0.08);
let p;
if (classification === 'HUMAN_ONLY') p = { human: Math.max(0.6, 1 - aiSoft), mixed: Math.min(0.35, aiSoft * 0.8), ai: Math.min(0.1, aiSoft * 0.3) };
else if (classification === 'AI_ONLY') p = { human: Math.max(0.02, 1 - aiSoft - 0.05), mixed: 0.1, ai: aiSoft };
else p = { human: Math.max(0.15, 0.6 - aiSoft * 0.5), mixed: 0.5, ai: aiSoft * 0.7 };
const rawSum = p.human + p.mixed + p.ai;
p.human = +(p.human / rawSum).toFixed(3);
p.mixed = +(p.mixed / rawSum).toFixed(3);
// Assign ai as the remainder so the three values sum to exactly 1.
// Clamp to >= 0 in case rounding pushes human+mixed above 1 (the
// remainder would otherwise show as -0 or -0.001 — surfaces as a
// negative percentage in any UI doing Math.round(p.ai * 100)).
p.ai = Math.max(0, +(1 - p.human - p.mixed).toFixed(3));
const probabilities = p;
// Confidence band:
// high — strongCorrob ≥ 2, OR cutoff-disclaimer, OR score < 8 (clean long doc)
// medium — strongCorrob ≥ 1, OR score ≥ 45 with weak corroborator, OR score < 20
// low — everything else
let confidence;
if (strongCorrob >= 2 || hasCutoff || (score < 8 && wordCount >= 100)) confidence = 'high';
else if (strongCorrob >= 1 || (score >= 45 && weakCorrob >= 1) || score < 20) confidence = 'medium';
else confidence = 'low';
return { classification, probabilities, confidence };
}
function getLabel(score) {
if (score === 0) return 'Clean';
if (score <= 15) return 'Minimal AI signals';
if (score <= 35) return 'Some AI patterns';
if (score <= 60) return 'Moderate AI signals';
if (score <= 80) return 'Strong AI signals';
return 'Heavy AI patterns';
}
function getColor(score) {
if (score <= 15) return '#44bb66';
if (score <= 35) return '#88bb44';
if (score <= 60) return '#ddaa00';
if (score <= 80) return '#ff8833';
return '#ff4444';
}
function deduplicateIssues(issues) {
const seen = new Set();
return issues.filter(issue => {
const key = `${issue.type}:${issue.text.toLowerCase()}`;
if (seen.has(key)) return false;
seen.add(key);
return true;
});
}
// ─── Severity labels ──────────────────────────────────────────
const SEVERITY_LABELS = {
critical: 'P0',
high: 'P1',
medium: 'P2',
low: 'P3',
};
const TYPE_LABELS = {
'tier1': 'AI vocabulary',
'tier1-clarity': 'Wordiness',
'tier2': 'Word cluster',
'tier3': 'Overused word',
'transition': 'AI transition',
'chatbot': 'Chatbot artifact',
'sycophantic': 'Sycophantic tone',
'filler': 'Filler phrase',
'generic-conclusion': 'Generic conclusion',
'lets-construction': '"Let\'s" opener',
'reasoning-artifact': 'Reasoning artifact',
'acknowledgment-loop': 'Acknowledgment loop',
'significance-inflation': 'Significance inflation',
'vague-attribution': 'Vague attribution',
'hollow-intensifier': 'Hollow intensifier',
'emotional-flatline': 'Emotional flatline',
'lingering-attention': 'Lingering-attention claim',
'novelty-inflation': 'Novelty inflation',
'cutoff-disclaimer': 'Cutoff disclaimer',
'template-phrase': 'Template phrase',
'false-concession': 'False concession',
'rhetorical-question': 'Rhetorical question',
'confidence-calibration': 'Confidence stacking',
'em-dash': 'Em dash overuse',
'uniformity': 'Rhythm uniformity',
'formatting': 'Formatting',
'tier3-phrase': 'Boilerplate phrase',
'tier3-phrase-cluster': 'Boilerplate cluster',
'hashtag-stuff': 'Hashtag stuffing',
'bullet-np-list': 'Bullet-NP list',
'hedge-stack': 'Hedge-stacked prediction',
'future-narrative': 'Generic future narrative',
'real-actual-inflation': '"Real/actual" inflation',
'social-cta-closer': 'Engagement-bait closer',
'formulaic-opener': 'Formulaic opener',
'speculative-opener': 'Speculative scenario opener',
'launch-intro': 'Launch-copy introduction',
'crowd-contrast': 'Dramatized crowd contrast',
'fake-casual-prop': 'Fake-casual prop',
'title-case-header': 'Title Case header',
'parenthetical-hedge': 'Parenthetical hedge',
'smart-punct-signature': 'Smart-punct signature',
'punct-distribution': 'Punctuation distribution',
'fnword-trigram-entropy': 'Grammar repetition',
'cross-para-burstiness': 'Cross-paragraph rhythm',
'normalization-flag': 'Bypass-trick chars',
'low-ttr': 'Low vocabulary diversity',
'ai-placeholder': 'Unfilled placeholder',
'ai-citation-markup': 'Chatbot citation markup leak',
'ai-utm-source': 'AI-tool URL parameter',
'unnecessary-hyphenation': 'Unnecessary hyphenation',
'performed-insight': 'Performed-insight phrase',
'negation-chain': 'Negation chain',
'dev-blog-boilerplate': 'Dev-blog boilerplate',
};
return {
analyzeText,
normalizeText,
getLabel,
getColor,
SEVERITY_LABELS,
TYPE_LABELS,
};
})();
if (typeof module !== 'undefined' && module.exports) {
module.exports = AIDetector;
}
SHA-256: 4fd45ea96359ca194de06ec39b009f20d86ffdc426a9997bf2084a71e8467db9