← Files note Workspace|ROGNALIAARCHIVED FILE

skills/note-draft-quality/scripts/check_draft_style.py

7.65 KB · Oct 2, 2026 · 00:33 UTC

↓ Download file

#!/usr/bin/env python3
"""Find review signals in Japanese note drafts without modifying the draft."""

from __future__ import annotations

import argparse
import hashlib
import json
import re
import sys
from pathlib import Path
from typing import Any, Dict, List, Sequence


MAX_INPUT_BYTES = 2_000_000
MAX_COMMAS_PER_SENTENCE = 3

METHOD_TERMS = re.compile(
    r"置換テスト|温度確認|記号lint|文体チェック|品質ゲート|静的signal"
)
MARKDOWN_LINK = re.compile(r"!??\[([^\]]*)\]\([^)]*\)")
INLINE_CODE = re.compile(r"`[^`]*`")
URL = re.compile(r"https?://\S+")
SENTENCE = re.compile(r"[^。!?!?]+[。!?!?]?")


def _result(status: str, **fields: Any) -> Dict[str, Any]:
    result: Dict[str, Any] = {"schema_version": 1, "status": status}
    result.update(fields)
    return result


def _read_draft(value: str) -> str:
    if value == "-":
        text = sys.stdin.read()
    else:
        path = Path(value)
        if not path.is_file():
            raise ValueError("draft fileを読めません。")
        if path.stat().st_size > MAX_INPUT_BYTES:
            raise ValueError("draft fileが大きすぎます。")
        text = path.read_text(encoding="utf-8")
    if len(text.encode("utf-8")) > MAX_INPUT_BYTES:
        raise ValueError("draft inputが大きすぎます。")
    if not text.strip():
        raise ValueError("draft inputが空です。")
    return text


def _clean_markdown(line: str) -> str:
    cleaned = line.strip()
    if not cleaned or cleaned.startswith("<!--"):
        return ""
    if re.fullmatch(r"(?:#[^\s#]+\s*){2,}", cleaned):
        return ""
    cleaned = re.sub(r"^#{1,6}\s+", "", cleaned)
    cleaned = re.sub(r"^>\s?", "", cleaned)
    cleaned = re.sub(r"^(?:[-*+]\s+|\d+[.)]\s+)", "", cleaned)
    cleaned = MARKDOWN_LINK.sub(lambda match: match.group(1), cleaned)
    cleaned = INLINE_CODE.sub("", cleaned)
    cleaned = URL.sub("", cleaned)
    cleaned = cleaned.replace("**", "").replace("__", "")
    return cleaned.strip()


def _prose_units(text: str) -> List[Dict[str, Any]]:
    units: List[Dict[str, Any]] = []
    paragraph_index = -1
    in_paragraph = False
    fence: str = ""

    for line_number, raw_line in enumerate(text.splitlines(), start=1):
        stripped = raw_line.strip()
        fence_match = re.match(r"^(```|~~~)", stripped)
        if fence_match:
            marker = fence_match.group(1)
            if not fence:
                fence = marker
            elif fence == marker:
                fence = ""
            in_paragraph = False
            continue
        if fence:
            continue
        if not stripped:
            in_paragraph = False
            continue

        if stripped.startswith(">") or re.match(
            r"^(?:#{1,6}\s+|\[(?:大見出し|小見出し)\])", stripped
        ):
            in_paragraph = False
            continue
        cleaned = _clean_markdown(raw_line)
        if not cleaned:
            continue
        if not in_paragraph:
            paragraph_index += 1
            in_paragraph = True
        units.append(
            {
                "line": line_number,
                "paragraph": paragraph_index,
                "text": cleaned,
            }
        )
    return units


def _sentences(units: Sequence[Dict[str, Any]]) -> List[Dict[str, Any]]:
    sentences: List[Dict[str, Any]] = []
    for unit in units:
        for match in SENTENCE.finditer(unit["text"]):
            text = match.group(0).strip()
            if text:
                sentences.append(
                    {
                        "line": unit["line"],
                        "paragraph": unit["paragraph"],
                        "text": text,
                    }
                )
    return sentences


def _ending_kind(sentence: str) -> str:
    body = re.sub(r"[。!?!?]+$", "", sentence).strip()
    patterns = (
        "と考えます",
        "と思います",
        "になります",
        "しています",
        "していました",
        "できません",
        "できます",
        "でしょう",
        "ではありません",
        "ません",
        "でした",
        "ました",
        "です",
        "ます",
        "である",
        "だった",
    )
    for pattern in patterns:
        if body.endswith(pattern):
            return pattern
    return ""


def _collect_issues(text: str) -> List[Dict[str, Any]]:
    units = _prose_units(text)
    sentences = _sentences(units)
    issues: List[Dict[str, Any]] = []
    seen = set()

    def add(code: str, line: int, message: str, **details: Any) -> None:
        key = (code, line)
        if key in seen:
            return
        seen.add(key)
        issue: Dict[str, Any] = {"code": code, "line": line, "message": message}
        if details:
            issue["details"] = details
        issues.append(issue)

    for unit in units:
        line = int(unit["line"])
        value = str(unit["text"])
        if METHOD_TERMS.search(value):
            add(
                "internal_method_leak",
                line,
                "制作・確認の用語があります。記事の題材として必要な説明か、作業報告の混入かを確認します。",
            )

    for sentence in sentences:
        value = str(sentence["text"])
        line = int(sentence["line"])
        comma_count = value.count("、")
        if comma_count > MAX_COMMAS_PER_SENTENCE:
            add(
                "comma_density",
                line,
                "一文に読点が四つ以上あります。数だけでは直さず、読みづらさや主語・述語の離れすぎがないか確認します。",
                comma_count=comma_count,
                threshold=MAX_COMMAS_PER_SENTENCE,
            )
    endings = [_ending_kind(str(sentence["text"])) for sentence in sentences]
    for index in range(2, len(endings)):
        same_paragraph = (
            sentences[index]["paragraph"]
            == sentences[index - 1]["paragraph"]
            == sentences[index - 2]["paragraph"]
        )
        if (
            same_paragraph
            and endings[index]
            and endings[index] == endings[index - 1] == endings[index - 2]
        ):
            add(
                "repeated_sentence_ending",
                int(sentences[index - 2]["line"]),
                "同じ段落で同じ語尾が三文以上続いています。単調さがあるか確認し、意図的な反復は保ちます。",
                ending=endings[index],
            )

    issues.sort(key=lambda issue: (int(issue["line"]), str(issue["code"])))
    return issues


def parse_args() -> argparse.Namespace:
    parser = argparse.ArgumentParser(
        description="日本語のnote原稿から文体上のreview signalをJSONで返します。"
    )
    parser.add_argument("draft", help="UTF-8のdraft file。stdinは-を指定します。")
    return parser.parse_args()


def main() -> int:
    args = parse_args()
    try:
        text = _read_draft(args.draft)
    except (OSError, UnicodeError, ValueError) as error:
        print(
            json.dumps(
                _result("error", errors=[str(error)]),
                ensure_ascii=False,
                sort_keys=True,
            )
        )
        return 2

    issues = _collect_issues(text)
    status = "review" if issues else "pass"
    print(
        json.dumps(
            _result(
                status,
                input_sha256=hashlib.sha256(text.encode("utf-8")).hexdigest(),
                issue_count=len(issues),
                issues=issues,
            ),
            ensure_ascii=False,
            sort_keys=True,
        )
    )
    return 1 if issues else 0


if __name__ == "__main__":
    sys.exit(main())

SHA-256: 256a80cc92eedc5b3797f91edc950ea2f1d680bd8936bd676de64fdc5e979f46