← Files SyntheiaARCHIVED FILE
skills/redline-issues-list/scripts/extract_changes.py
48.7 KB · Oct 2, 2026 · 00:19 UTC
#!/usr/bin/env python3
"""
extract_changes.py -- Walk a redlined .docx's tracked changes in document
order and produce structured JSON for building a legal issues list.
Outputs four lists:
content_changes substantive edits: text inserted/deleted/substituted,
counter-edits (text one party inserted that another
then deleted), content moved, paragraphs split or
merged, table rows/cells structurally added or removed.
formatting_changes formatting-only revisions (rPrChange, pPrChange,
tblPrChange, tcPrChange, sectPrChange). Excluded from
the issues table, but reported so you can confirm
nothing substantive is hiding among them -- notably
numbering-level changes, which can mean a clause was
promoted or demoted rather than merely restyled.
comments Word comments and replies, with the text they anchor
to. In a counterparty markup these usually carry the
negotiating rationale, so they belong in the review
even though they aren't revisions.
unanchored_comments comments whose anchor markers weren't found in the
parts scanned (e.g. anchored in a part not read).
Why a script rather than reading the XML directly: Word records
formatting-only edits with dedicated *Change elements that read like
content edits when scanning by eye; revisions nest (an <w:ins> containing
a <w:del> is a counter-edit, not a plain deletion); a substitution is
routinely split by a stray space into what looks like two unrelated
edits; clause numbers usually live in numbering.xml rather than in the
paragraph text; and changes hide in table cells, footnotes, and deleted
paragraph marks. One deterministic pass applies the same rules to all of
it and doesn't tire out on page 40.
Usage:
python extract_changes.py redline.docx > changes.json
python extract_changes.py redline.docx --all-parts > changes.json
python extract_changes.py unpacked_dir/ --markdown > table.md
Inputs: a .docx, an already-unzipped folder, or a bare document.xml
(clause numbering and comments are unavailable for a bare XML file).
By default the body, footnotes, and endnotes are scanned -- legal drafting
puts substantive terms in footnotes often enough that skipping them is a
real risk. --all-parts adds headers and footers. --markdown prints a
starter issues table instead of JSON.
Known limitations, so they aren't discovered the hard way:
- Clause numbers are computed by replaying Word's numbering counters over
the marked-up document. Inserted or deleted paragraphs shift the
numbering Word displays, and lvlRestart/lvlOverride edge cases aren't
modelled, so treat a number as a locator to verify rather than a
citation to quote. Paragraphs numbered by a style rather than direct
formatting are resolved via styles.xml; anything Word renumbers
dynamically in ways not captured here falls back to the nearest
heading.
- Table cell insert/delete/merge revisions are located and flagged but
not diffed cell-by-cell -- inspect the flagged row directly.
- Field codes (cross-reference fields, TOC) are read as their cached
display text; the field's target is not resolved.
"""
import sys
import re
import json
import zipfile
import argparse
from pathlib import Path
import xml.etree.ElementTree as ET
W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
# Wrappers we recurse straight through: they neither carry revision state
# nor contribute text of their own.
TRANSPARENT = {
"hyperlink", "smartTag", "smartTagPr", "fldSimple", "sdt", "sdtContent",
"bookmarkStart", "bookmarkEnd", "customXml", "customXmlPr", "dir", "bdo",
}
HEADING_STYLE_RE = re.compile(r"^(heading|title|berschrift)", re.IGNORECASE)
# Fallback for documents that hard-type their clause numbers into the text.
LITERAL_CLAUSE_RE = re.compile(
r"^\s*("
r"\(?\d+(\.\d+)+\)?[.)]?" # 2.1, 8.2.1, (2.1) -- trailing delimiter optional
r"|\(?\d+\)?[.)]" # 3. / 3) / (3) -- a lone number needs a delimiter,
# otherwise "12 months from..." reads as clause 12
r"|\([a-zA-Z]\)|\([ivxlcdm]+\)"
r"|(Section|Clause|Article|Schedule|Exhibit|Annex|Appendix)\s+[\w.]+"
r")(?!%)(?=\s+\S)", # reject "1.5% per month"; require following text
re.IGNORECASE,
)
ROMAN = [
(1000, "m"), (900, "cm"), (500, "d"), (400, "cd"), (100, "c"), (90, "xc"),
(50, "l"), (40, "xl"), (10, "x"), (9, "ix"), (5, "v"), (4, "iv"), (1, "i"),
]
def tag(el):
return el.tag.split("}", 1)[-1] if "}" in el.tag else el.tag
def attr(el, name):
return el.get(W + name) if el is not None else None
# --------------------------------------------------------------------------
# Package access: .docx zip, unzipped folder, or bare document.xml
# --------------------------------------------------------------------------
class Package:
def __init__(self, path):
self.path = Path(path)
self._zip = None
self.bare_xml = None
if self.path.is_dir():
self.mode = "dir"
elif self.path.suffix.lower() == ".xml":
self.mode = "bare"
self.bare_xml = self.path.read_bytes()
else:
self.mode = "zip"
self._zip = zipfile.ZipFile(self.path)
def read(self, name):
"""Return bytes for a package part, or None if absent."""
if self.mode == "bare":
return self.bare_xml if name == "word/document.xml" else None
if self.mode == "dir":
p = self.path / name
return p.read_bytes() if p.exists() else None
try:
return self._zip.read(name)
except KeyError:
return None
def names(self):
if self.mode == "bare":
return ["word/document.xml"]
if self.mode == "dir":
return [
str(p.relative_to(self.path)).replace("\\", "/")
for p in self.path.rglob("*.xml")
]
return self._zip.namelist()
def part_list(self, all_parts=False):
"""Parts to scan, in a sensible reading order."""
parts = [("body", "word/document.xml")]
for label, pattern in (("footnotes", r"word/footnotes\.xml$"),
("endnotes", r"word/endnotes\.xml$")):
for n in sorted(self.names()):
if re.match(pattern, n):
parts.append((label, n))
if all_parts:
for n in sorted(self.names()):
if re.match(r"word/(header|footer)\d*\.xml$", n):
parts.append((Path(n).stem, n))
return [(label, n) for label, n in parts if self.read(n) is not None]
# --------------------------------------------------------------------------
# Numbering: replay Word's counters to recover real clause numbers
# --------------------------------------------------------------------------
def format_counter(value, numfmt):
if numfmt in (None, "decimal"):
return str(value)
if numfmt == "decimalZero":
return f"{value:02d}"
if numfmt in ("lowerLetter", "upperLetter"):
# 1->a, 26->z, 27->aa (Word's alphabetic overflow)
s, n = "", value
while n > 0:
n, rem = divmod(n - 1, 26)
s = chr(ord("a") + rem) + s
return s.upper() if numfmt == "upperLetter" else s
if numfmt in ("lowerRoman", "upperRoman"):
s, n = "", value
for v, sym in ROMAN:
while n >= v:
s += sym
n -= v
return s.upper() if numfmt == "upperRoman" else s
return str(value)
class Numbering:
"""Resolves a paragraph's displayed clause number (e.g. '3.2(a)')."""
def __init__(self, pkg):
self.abstract = {} # abstractNumId -> {ilvl: {fmt, text, start}}
self.num_to_abstract = {} # numId -> abstractNumId
self.overrides = {} # (numId, ilvl) -> start override
self.style_numpr = {} # styleId -> (numId, ilvl)
self.style_outline = {} # styleId -> outline level int
self.style_names = {} # styleId -> name
self.counters = {} # (abstractNumId, ilvl) -> current value
self._load_numbering(pkg)
self._load_styles(pkg)
def _load_numbering(self, pkg):
raw = pkg.read("word/numbering.xml")
if not raw:
return
root = ET.fromstring(raw)
for an in root.findall(W + "abstractNum"):
aid = attr(an, "abstractNumId")
levels = {}
for lvl in an.findall(W + "lvl"):
ilvl = int(attr(lvl, "ilvl") or 0)
levels[ilvl] = {
"fmt": attr(lvl.find(W + "numFmt"), "val"),
"text": attr(lvl.find(W + "lvlText"), "val") or "",
"start": int(attr(lvl.find(W + "start"), "val") or 1),
"pStyle": attr(lvl.find(W + "pStyle"), "val"),
}
self.abstract[aid] = levels
for num in root.findall(W + "num"):
nid = attr(num, "numId")
ref = num.find(W + "abstractNumId")
if ref is not None:
self.num_to_abstract[nid] = attr(ref, "val")
for ov in num.findall(W + "lvlOverride"):
ilvl = int(attr(ov, "ilvl") or 0)
sw = ov.find(W + "startOverride")
if sw is not None:
self.overrides[(nid, ilvl)] = int(attr(sw, "val") or 1)
def _load_styles(self, pkg):
raw = pkg.read("word/styles.xml")
if not raw:
return
root = ET.fromstring(raw)
for st in root.findall(W + "style"):
sid = attr(st, "styleId")
name = attr(st.find(W + "name"), "val")
if name:
self.style_names[sid] = name
ppr = st.find(W + "pPr")
if ppr is None:
continue
numpr = ppr.find(W + "numPr")
if numpr is not None:
nid = attr(numpr.find(W + "numId"), "val")
ilvl = attr(numpr.find(W + "ilvl"), "val")
if nid:
self.style_numpr[sid] = (nid, int(ilvl or 0))
outline = ppr.find(W + "outlineLvl")
if outline is not None:
self.style_outline[sid] = int(attr(outline, "val") or 0)
def style_label(self, style_id):
if not style_id:
return None
return self.style_names.get(style_id, style_id)
def is_heading(self, style_id):
if not style_id:
return False
if HEADING_STYLE_RE.search(style_id):
return True
name = self.style_names.get(style_id, "")
return bool(HEADING_STYLE_RE.search(name)) or style_id in self.style_outline
def resolve(self, ppr, style_id):
"""Return (numId, ilvl) from direct formatting or the paragraph style."""
if ppr is not None:
numpr = ppr.find(W + "numPr")
if numpr is not None:
nid = attr(numpr.find(W + "numId"), "val")
ilvl = attr(numpr.find(W + "ilvl"), "val")
if nid and nid != "0":
if ilvl is None and style_id in self.style_numpr:
ilvl = str(self.style_numpr[style_id][1])
return nid, int(ilvl or 0)
if nid == "0":
return None # numbering explicitly removed
if style_id in self.style_numpr:
return self.style_numpr[style_id]
return None
def advance(self, numid, ilvl):
"""Increment counters for this paragraph and render its number."""
aid = self.num_to_abstract.get(numid)
if aid is None or aid not in self.abstract:
return None
levels = self.abstract[aid]
lvl = levels.get(ilvl)
if lvl is None:
return None
fmt = lvl["fmt"]
if fmt in ("bullet", "none"):
return None
key = (aid, ilvl)
if key not in self.counters:
start = self.overrides.get((numid, ilvl), lvl["start"])
self.counters[key] = start
else:
self.counters[key] += 1
# A new item at this level restarts everything below it.
for (a, l) in list(self.counters):
if a == aid and l > ilvl:
del self.counters[(a, l)]
text = lvl["text"]
def sub(m):
target = int(m.group(1)) - 1
tlvl = levels.get(target)
if tlvl is None:
return ""
val = self.counters.get((aid, target))
if val is None:
val = self.overrides.get((numid, target), tlvl["start"])
return format_counter(val, tlvl["fmt"])
rendered = re.sub(r"%(\d+)", sub, text).strip()
return rendered or None
# --------------------------------------------------------------------------
# Segment collection: revision-aware walk of a paragraph
# --------------------------------------------------------------------------
REV_TAGS = {"ins", "del", "moveFrom", "moveTo"}
SEG_LABEL = {
frozenset(): "unchanged",
frozenset({"ins"}): "ins",
frozenset({"del"}): "del",
frozenset({"ins", "del"}): "ins_del",
frozenset({"moveFrom"}): "move_out",
frozenset({"moveTo"}): "move_in",
frozenset({"moveTo", "del"}): "move_in_del",
frozenset({"moveFrom", "ins"}): "move_out_ins",
}
RENDER = {
"ins": "INSERTED",
"del": "DELETED",
"ins_del": "INSERTED THEN DELETED",
"move_out": "MOVED OUT",
"move_in": "MOVED IN",
"move_in_del": "MOVED IN THEN DELETED",
"move_out_ins": "MOVED OUT THEN RESTORED",
}
class Seg:
__slots__ = ("kind", "text", "revs")
def __init__(self, kind, text, revs):
self.kind = kind
self.text = text
self.revs = revs # list of (rev_tag, author, date)
@property
def authors(self):
return [a for _, a, _ in self.revs if a]
@property
def dates(self):
return [d for _, _, d in self.revs if d]
def collect_segments(el, revs=()):
"""Collect Seg objects from a paragraph in document order, tracking
nested revision context so an <w:ins> wrapping a <w:del> is recognised
as a counter-edit rather than a plain deletion."""
out = []
t = tag(el)
if t in REV_TAGS:
nested = tuple(revs) + ((t, attr(el, "author"), attr(el, "date")),)
for child in el:
out.extend(collect_segments(child, nested))
return out
if t == "r":
rpr = el.find(W + "rPr")
rprchange = rpr.find(W + "rPrChange") if rpr is not None else None
if rprchange is not None and not revs:
out.append(Seg("formatting_run", None,
[("rPrChange", attr(rprchange, "author"), attr(rprchange, "date"))]))
parts = []
for child in el:
ct = tag(child)
if ct in ("t", "delText", "instrText", "delInstrText"):
parts.append(child.text or "")
elif ct == "tab":
parts.append("\t")
elif ct in ("br", "cr"):
parts.append("\n")
elif ct in ("footnoteReference", "endnoteReference"):
parts.append(f"[{ct.replace('Reference','')} {attr(child, 'id')}]")
text = "".join(parts)
if text:
kinds = frozenset(r for r, _, _ in revs)
out.append(Seg(SEG_LABEL.get(kinds, "complex"), text, list(revs)))
return out
if t == "tbl":
return out # table content is walked separately, cell by cell
for child in el:
out.extend(collect_segments(child, revs))
return out
def merge_adjacent(segs):
"""Coalesce consecutive segments with identical kind and revision authorship."""
merged = []
for s in segs:
if s.kind == "formatting_run":
merged.append(s)
continue
prev = merged[-1] if merged else None
if (prev is not None and prev.kind == s.kind
and prev.kind != "formatting_run"
and prev.revs == s.revs):
prev.text += s.text
else:
merged.append(Seg(s.kind, s.text, list(s.revs)))
return merged
def render_paragraph(segs, limit=None):
out = []
for s in segs:
if s.kind == "formatting_run" or s.text is None:
continue
if s.kind == "unchanged":
out.append(s.text)
else:
out.append(f"\u27e6{RENDER.get(s.kind, s.kind.upper())}: {s.text}\u27e7")
text = "".join(out)
if limit and len(text) > limit:
text = text[: limit - 3].rstrip() + "..."
return text
def visible_text(segs):
"""Text as it reads with changes accepted (deletions dropped)."""
return "".join(
s.text for s in segs
if s.text and s.kind in ("unchanged", "ins", "move_in")
)
# Unchanged text that shouldn't break a single logical edit in two: a stray
# space or a short run of punctuation between a deletion and its replacement.
CONNECTIVE_RE = re.compile(r"^(\s*|[^\w\s]{1,3}\s*)$")
def group_edits(segs):
"""Group contiguous revision segments into single edit events, bridging
trivial unchanged text so a del+space+ins reads as one substitution.
Returns dicts carrying the group plus the partial words either side of it.
Comparison tools split runs mid-token (a change to "three (3)" arrives as
"three (3" with the ")" left in an unchanged run), so the reported text is
snapped out to the nearest whitespace to stay readable."""
results, current, pending = [], [], []
prev_unchanged = ""
def close():
if current:
results.append({"segs": list(current), "prefix": _tail_word(prev_unchanged)})
for s in segs:
if s.kind == "formatting_run":
continue
if s.kind == "unchanged":
if current and s.text and CONNECTIVE_RE.match(s.text):
pending.append(s)
continue
if current:
close()
results[-1]["suffix"] = _head_word(s.text or "")
current, pending = [], []
prev_unchanged = s.text or ""
continue
if pending:
current.extend(pending)
pending = []
current.append(s)
if current:
close()
results[-1]["suffix"] = ""
for r in results:
r.setdefault("suffix", "")
return results
def _tail_word(text, cap=30):
"""Trailing characters of `text` since the last whitespace."""
if not text or text[-1].isspace():
return ""
return text[-cap:].split()[-1] if text.split() else ""
def _head_word(text, cap=30):
"""Leading characters of `text` up to the first whitespace."""
if not text or text[0].isspace():
return ""
head = text[:cap]
return head.split()[0] if head.split() else ""
def classify_group(group):
kinds = {s.kind for s in group if s.kind != "unchanged"}
if "ins_del" in kinds or "move_in_del" in kinds or "move_out_ins" in kinds:
return "counter-edit"
if kinds == {"del"}:
return "deletion"
if kinds == {"ins"}:
return "insertion"
# Comparison tools (LibreOffice's Compare Documents, Word's Combine) often
# tag a word as "moved" when it merely also occurs elsewhere, so a
# deletion sitting against a moved-in run is a plain substitution in
# substance. Reading these as moves would scatter one edit across two rows.
if kinds in ({"del", "ins"}, {"del", "move_in"}, {"move_out", "ins"},
{"del", "ins", "move_in"}, {"del", "ins", "move_out"}):
return "substitution"
if kinds == {"move_out"}:
return "move-out"
if kinds == {"move_in"}:
return "move-in"
if kinds == {"move_out", "move_in"}:
return "move"
return "mixed"
def group_text(group, kinds):
return "".join(s.text for s in group if s.kind in kinds and s.text) or None
def one_or_list(values):
uniq = sorted(set(values))
if not uniq:
return None
return uniq[0] if len(uniq) == 1 else uniq
# --------------------------------------------------------------------------
# Document walk
# --------------------------------------------------------------------------
class State:
def __init__(self, numbering, context_chars):
self.numbering = numbering
self.context_chars = context_chars
self.seq = 0
self.para_index = 0
self.heading_stack = [] # [(outline_level, text)]
self.last_clause = None
self.content = []
self.formatting = []
self.text_lines = []
# Set while walking inside a wholly inserted or deleted table row: the
# row is already reported as one change, so the per-cell insertions and
# paragraph marks underneath it are noise. A rebuilt 4-row table
# otherwise produces 35 rows for what a reader sees as two edits.
self.suppress = 0
self.table_states = []
self.comment_hits = {} # comment id -> anchor info
self.open_comments = {} # comment id -> list of text fragments
self.part = "body"
def next_seq(self):
if self.suppress:
return None
self.seq += 1
return self.seq
def add_content(self, rec):
if self.suppress:
return
self.content.append(rec)
def add_formatting(self, rec):
if self.suppress:
return
self.formatting.append(rec)
def section_path(self):
return " > ".join(t for _, t in self.heading_stack) or None
def locate(self, extra=None):
"""Human-readable locator: heading path, then the clause number if it
adds information the heading path doesn't already carry."""
bits = []
if self.part != "body":
bits.append(self.part)
path = self.section_path()
if path:
bits.append(path)
clause = self.last_clause
if clause and (not path or clause.rstrip(".)") not in path):
bits.append(f"clause {clause}" if path else clause)
if extra:
bits.append(extra)
if not bits:
return f"paragraph {self.para_index}"
return " — ".join(bits) if len(bits) > 1 else bits[0]
def base_record(self, extra_location=None):
return {
"part": self.part,
"paragraph_index": self.para_index,
"clause": self.last_clause,
"section_path": self.section_path(),
"location": self.locate(extra_location),
}
def para_style(p):
ppr = p.find(W + "pPr")
if ppr is None:
return None
return attr(ppr.find(W + "pStyle"), "val")
def update_location(state, p, ppr, style_id, segs):
"""Advance numbering counters and heading/clause context for this paragraph."""
num = state.numbering.resolve(ppr, style_id) if state.numbering else None
number = state.numbering.advance(*num) if num else None
text = visible_text(segs).strip()
if number:
state.last_clause = number
elif text and LITERAL_CLAUSE_RE.match(text):
state.last_clause = LITERAL_CLAUSE_RE.match(text).group(0).strip()
if state.numbering and state.numbering.is_heading(style_id) and text:
label = f"{number} {text}".strip() if number else text
label = label[:120]
level = state.numbering.style_outline.get(style_id)
if level is None:
m = re.search(r"(\d+)", style_id or "")
level = int(m.group(1)) - 1 if m else 0
state.heading_stack = [(l, t) for l, t in state.heading_stack if l < level]
state.heading_stack.append((level, label))
def process_paragraph(p, state, extra_location=None):
state.para_index += 1
ppr = p.find(W + "pPr")
style_id = para_style(p)
segs = merge_adjacent(collect_segments(p))
context = render_paragraph(segs, state.context_chars)
# Resolve this paragraph's own clause number and heading position *before*
# recording its changes. Doing it afterwards labels every change with the
# preceding paragraph's location, which sends the reader to the wrong clause.
update_location(state, p, ppr, style_id, segs)
# Removing deleted runs can leave doubled spaces behind; collapse them so
# the text reads normally and phrase searches still match.
accepted = re.sub(r"[ \t]{2,}", " ", visible_text(segs)).strip()
if accepted:
prefix = state.last_clause or (
state.heading_stack[-1][1] if state.heading_stack else f"p{state.para_index}")
note = "" if state.part == "body" else f"{state.part}: "
state.text_lines.append(f"[{note}{prefix}] {accepted}")
# Paragraph-mark revisions and paragraph-level formatting revisions.
if ppr is not None:
rpr = ppr.find(W + "rPr")
if rpr is not None:
for rev_tag, ctype, note in (
("del", "paragraph-mark-deleted",
"The paragraph mark was deleted, so this paragraph merges into the one that follows. "
"Check whether that silently combines two separate obligations, or removes a "
"list item's identity for cross-referencing purposes."),
("ins", "paragraph-mark-inserted",
"A paragraph mark was inserted, splitting this text into two paragraphs. "
"In a numbered clause this creates a new sub-clause and shifts the numbering "
"of everything after it, which can break cross-references elsewhere."),
):
rev = rpr.find(W + rev_tag)
if rev is not None:
rec = state.base_record(extra_location)
rec.update({
"sequence": state.next_seq(),
"type": ctype,
"old_text": None,
"new_text": None,
"note": note,
"author": attr(rev, "author"),
"date": attr(rev, "date"),
"paragraph_context": context,
})
state.add_content(rec)
pprchange = ppr.find(W + "pPrChange")
if pprchange is not None:
old_ppr = pprchange.find(W + "pPr")
old_num = old_ppr.find(W + "numPr") if old_ppr is not None else None
new_num = ppr.find(W + "numPr")
old_ilvl = attr(old_num.find(W + "ilvl"), "val") if old_num is not None else None
new_ilvl = attr(new_num.find(W + "ilvl"), "val") if new_num is not None else None
numbering_shift = (old_num is None) != (new_num is None) or old_ilvl != new_ilvl
rec = state.base_record(extra_location)
rec.update({
"type": "paragraph-formatting",
"author": attr(pprchange, "author"),
"date": attr(pprchange, "date"),
"numbering_changed": numbering_shift,
"note": (
"Numbering or list level changed with this formatting revision. "
"Review before dismissing: promoting or demoting a clause changes what "
"it is subordinate to, and renumbering can break cross-references."
if numbering_shift else
"Paragraph formatting only (alignment, spacing, indent, style)."
),
"paragraph_preview": visible_text(segs)[:120] or None,
})
state.add_formatting(rec)
for s in segs:
if s.kind == "formatting_run":
rec = state.base_record(extra_location)
rec.update({
"type": "run-formatting",
"author": s.authors[0] if s.authors else None,
"date": s.dates[0] if s.dates else None,
"numbering_changed": False,
"note": "Character formatting only (font, bold, italic, colour) with no wording change.",
"paragraph_preview": visible_text(segs)[:120] or None,
})
state.add_formatting(rec)
for g in group_edits(segs):
group, prefix, suffix = g["segs"], g["prefix"], g["suffix"]
gtype = classify_group(group)
def snap(text):
if text is None:
return None
return f"{prefix}{text}{suffix}"
rec = state.base_record(extra_location)
old_text = snap(group_text(group, {"del", "move_out"}))
new_text = snap(group_text(group, {"ins", "move_in", "move_out_ins"}))
# A genuine move relocates identical text. When the two sides differ,
# the comparison tool has tagged replaced wording as moved because it
# happens to occur elsewhere -- in substance it's a substitution.
if gtype == "move" and (old_text or "").strip() != (new_text or "").strip():
gtype = "substitution"
rec.update({
"sequence": state.next_seq(),
"type": gtype,
"old_text": old_text,
"new_text": new_text,
# Text that was inserted in an earlier round and then deleted in a
# later one: never in the original, won't survive either.
"inserted_then_deleted": group_text(group, {"ins_del", "move_in_del"}),
"author": one_or_list([a for s in group for a in s.authors]),
"date": one_or_list([d for s in group for d in s.dates]),
"paragraph_context": context,
})
if gtype == "counter-edit":
rec["note"] = (
"Nested revision: text one author inserted has been deleted by another "
"(or vice versa). This is negotiating history across markup rounds -- "
"check who did what and whether the net position is the original wording."
)
state.add_content(rec)
def cell_text(tc, include_deleted=False):
"""Plain text of a cell. include_deleted keeps struck-through content, which
is the only way to recover what a wholly deleted row used to say."""
segs = merge_adjacent(collect_segments(tc))
kinds = {"unchanged", "ins", "move_in"}
if include_deleted:
kinds |= {"del", "move_out", "ins_del", "move_in_del"}
return "".join(s.text for s in segs if s.text and s.kind in kinds).strip()
def process_table(tbl, state, extra_location=None):
# A table edited structurally is often recorded as every old row deleted
# and every new row inserted, even when one figure changed. Detect that so
# the rows can be reported as one rebuild to compare, not N separate edits.
rows = tbl.findall(W + "tr")
ins_rows = sum(1 for tr in rows
if (tr.find(W + "trPr") is not None
and tr.find(W + "trPr").find(W + "ins") is not None))
del_rows = sum(1 for tr in rows
if (tr.find(W + "trPr") is not None
and tr.find(W + "trPr").find(W + "del") is not None))
# A replaced table shows up two ways: one table with both inserted and
# deleted rows, or -- as LibreOffice's compare does it -- a wholly inserted
# table sitting next to a wholly deleted one. Either way the reviewer needs
# to diff the two row sets rather than read each row as its own change.
if rows and ins_rows == len(rows):
table_status = "wholly-inserted"
elif rows and del_rows == len(rows):
table_status = "wholly-deleted"
elif ins_rows and del_rows:
table_status = "partly-replaced"
else:
table_status = None
if table_status:
state.table_states.append(table_status)
tblpr = tbl.find(W + "tblPr")
if tblpr is not None and tblpr.find(W + "tblPrChange") is not None:
change = tblpr.find(W + "tblPrChange")
rec = state.base_record(extra_location)
rec.update({
"type": "table-formatting",
"author": attr(change, "author"),
"date": attr(change, "date"),
"numbering_changed": False,
"note": "Table formatting only (borders, width, shading, layout).",
"paragraph_preview": None,
})
state.add_formatting(rec)
for row_i, tr in enumerate(rows, start=1):
trpr = tr.find(W + "trPr")
row_loc = f"table row {row_i}"
row_wholly_revised = False
if trpr is not None:
for rev_tag, ctype, note in (
("ins", "table-row-inserted", "An entire table row was inserted."),
("del", "table-row-deleted", "An entire table row was deleted."),
):
rev = trpr.find(W + rev_tag)
if rev is not None:
row_wholly_revised = True
# A deleted row's content only survives in delText, so the
# struck-through text has to be included to report what the
# row used to say.
text = " | ".join(
cell_text(tc, include_deleted=(rev_tag == "del"))
for tc in tr.findall(W + "tc"))
rec = state.base_record(row_loc)
rec.update({
"sequence": state.next_seq(),
"type": ctype,
"old_text": text if rev_tag == "del" else None,
"new_text": text if rev_tag == "ins" else None,
"note": note,
"table_status": table_status,
"author": attr(rev, "author"),
"date": attr(rev, "date"),
"paragraph_context": text[: state.context_chars] if state.context_chars else text,
})
state.add_content(rec)
if trpr.find(W + "trPrChange") is not None:
change = trpr.find(W + "trPrChange")
rec = state.base_record(row_loc)
rec.update({
"type": "table-row-formatting",
"author": attr(change, "author"),
"date": attr(change, "date"),
"numbering_changed": False,
"note": "Row height or row formatting only.",
"paragraph_preview": cell_text(tr)[:120] or None,
})
state.add_formatting(rec)
if row_wholly_revised:
state.suppress += 1
row_text = " | ".join(cell_text(tc, include_deleted=True)
for tc in tr.findall(W + "tc"))
for col_i, tc in enumerate(tr.findall(W + "tc"), start=1):
cell_loc = f"table row {row_i}, cell {col_i}"
tcpr = tc.find(W + "tcPr")
if tcpr is not None:
for stag, label in (("cellIns", "inserted"), ("cellDel", "deleted"),
("cellMerge", "merged or split")):
node = tcpr.find(W + stag)
if node is not None:
rec = state.base_record(cell_loc)
rec.update({
"sequence": state.next_seq(),
"type": "table-cell-structure",
"old_text": None,
"new_text": None,
"note": f"A table cell or column was {label}. The affected text is not "
f"auto-diffed -- inspect this cell and its row directly.",
"author": attr(node, "author"),
"date": attr(node, "date"),
"paragraph_context": row_text[: state.context_chars] if state.context_chars else row_text,
})
state.add_content(rec)
if tcpr.find(W + "tcPrChange") is not None:
change = tcpr.find(W + "tcPrChange")
rec = state.base_record(cell_loc)
rec.update({
"type": "table-cell-formatting",
"author": attr(change, "author"),
"date": attr(change, "date"),
"numbering_changed": False,
"note": "Cell width, shading, or borders only.",
"paragraph_preview": None,
})
state.add_formatting(rec)
walk_container(tc, state, extra_location=cell_loc)
if row_wholly_revised:
state.suppress -= 1
def walk_container(container, state, extra_location=None):
for child in list(container):
t = tag(child)
if t == "p":
process_paragraph(child, state, extra_location)
elif t == "tbl":
process_table(child, state, extra_location)
elif t == "sectPr":
change = child.find(W + "sectPrChange")
if change is not None:
rec = state.base_record(extra_location)
rec.update({
"type": "section-formatting",
"author": attr(change, "author"),
"date": attr(change, "date"),
"numbering_changed": False,
"note": "Section properties only (margins, page size, columns).",
"paragraph_preview": None,
})
state.add_formatting(rec)
elif t in ("footnote", "endnote"):
# footnotes.xml / endnotes.xml wrap their paragraphs one level deeper.
note_id = attr(child, "id")
walk_container(child, state,
extra_location=f"note {note_id}" if note_id else None)
elif t in ("commentRangeStart", "commentRangeEnd", "commentReference"):
pass # handled by the comment pass
# --------------------------------------------------------------------------
# Comments
# --------------------------------------------------------------------------
def collect_comment_anchors(root, state_part, anchors):
"""Record, per comment id, the visible text between its range markers."""
open_ids = set()
for order, el in enumerate(root.iter()):
t = tag(el)
if t == "commentRangeStart":
cid = attr(el, "id")
open_ids.add(cid)
anchors.setdefault(cid, {"part": state_part, "text": "", "order": order})
elif t == "commentRangeEnd":
open_ids.discard(attr(el, "id"))
elif t == "commentReference":
cid = attr(el, "id")
anchors.setdefault(cid, {"part": state_part, "text": "", "order": order})
elif t in ("t", "delText") and open_ids:
for cid in open_ids:
anchors[cid]["text"] += el.text or ""
def load_comments(pkg, anchors, context_chars):
raw = pkg.read("word/comments.xml")
if not raw:
return [], []
root = ET.fromstring(raw)
parents = {}
ext = pkg.read("word/commentsExtended.xml")
para_to_comment = {}
if ext:
# commentsExtended links replies by paragraph id rather than comment id,
# so map each comment's last paragraph id first.
for c in root.findall(W + "comment"):
paras = c.findall(W + "p")
if paras:
pid = paras[-1].get(
"{http://schemas.microsoft.com/office/word/2010/wordml}paraId"
)
if pid:
para_to_comment[pid] = attr(c, "id")
eroot = ET.fromstring(ext)
for ce in eroot.iter():
if tag(ce) != "commentEx":
continue
own = ce.get("{http://schemas.microsoft.com/office/word/2010/wordml}paraId")
parent = ce.get("{http://schemas.microsoft.com/office/word/2010/wordml}paraIdParent")
if own and parent:
child_id = para_to_comment.get(own)
parent_id = para_to_comment.get(parent)
if child_id and parent_id:
parents[child_id] = parent_id
records = []
for c in root.findall(W + "comment"):
cid = attr(c, "id")
text = "\n".join(
"".join(t.text or "" for t in p.iter() if tag(t) in ("t", "delText"))
for p in c.findall(W + "p")
).strip()
records.append({
"comment_id": cid,
"author": attr(c, "author"),
"initials": attr(c, "initials"),
"date": attr(c, "date"),
"text": text,
"reply_to": parents.get(cid),
})
# A reply carries no range markers of its own -- it hangs off its parent's
# anchor, so inherit it rather than reporting the reply as unanchored.
by_id = {r["comment_id"]: r for r in records}
for r in records:
anchor = anchors.get(r["comment_id"])
if anchor is None:
parent = r.get("reply_to")
seen = set()
while parent and parent not in seen:
seen.add(parent)
anchor = anchors.get(parent)
if anchor:
break
parent = by_id.get(parent, {}).get("reply_to")
anchor_text = (anchor or {}).get("text", "").strip()
if context_chars and len(anchor_text) > context_chars:
anchor_text = anchor_text[: context_chars - 3].rstrip() + "..."
r["anchored_to"] = anchor_text or None
r["part"] = (anchor or {}).get("part")
r["_order"] = (anchor or {}).get("order", 10**9)
resolved = sorted((r for r in records if r["part"] is not None),
key=lambda r: (r["_order"], r["comment_id"]))
unanchored = [r for r in records if r["part"] is None]
for r in records:
r.pop("_order", None)
return resolved, unanchored
# --------------------------------------------------------------------------
# Markdown starter table
# --------------------------------------------------------------------------
def escape_cell(text):
if text is None:
return ""
return text.replace("|", "\\|").replace("\n", " ").strip()
def raw_change_text(rec):
if rec["type"] == "substitution":
return f'"{escape_cell(rec["old_text"])}" → "{escape_cell(rec["new_text"])}"'
if rec["type"] == "insertion":
return f'inserted: "{escape_cell(rec["new_text"])}"'
if rec["type"] == "deletion":
return f'deleted: "{escape_cell(rec["old_text"])}"'
if rec["type"] == "counter-edit":
bits = []
if rec.get("inserted_then_deleted"):
bits.append(f'inserted then struck out: "{escape_cell(rec["inserted_then_deleted"])}"')
if rec.get("new_text"):
bits.append(f'net insertion: "{escape_cell(rec["new_text"])}"')
if rec.get("old_text"):
bits.append(f'original text deleted: "{escape_cell(rec["old_text"])}"')
return "counter-edit — " + "; ".join(bits)
parts = [rec["type"]]
if rec.get("new_text"):
parts.append(f'"{escape_cell(rec["new_text"])}"')
elif rec.get("old_text"):
parts.append(f'"{escape_cell(rec["old_text"])}"')
elif rec.get("note"):
parts.append(escape_cell(rec["note"]).split(".")[0])
return " — ".join(parts)
def write_markdown(result, out):
s = result["summary"]
print(f"<!-- {s['content_change_count']} substantive changes, "
f"{s['formatting_change_count']} formatting-only (excluded), "
f"{s['comment_count']} comments. Raw extraction — the change column is "
f"unedited machine text; rewrite it concisely, then fill comments and "
f"impact per the skill workflow. -->", file=out)
print(file=out)
print("| # | location | change | comments | impact |", file=out)
print("|---|---|---|---|---|", file=out)
for rec in result["content_changes"]:
print(f'| {rec["sequence"]} | {escape_cell(rec["location"])} | '
f'{raw_change_text(rec)} | | |', file=out)
if result["comments"]:
print(file=out)
print("### Word comments in the markup", file=out)
print(file=out)
print("| author | anchored to | comment |", file=out)
print("|---|---|---|", file=out)
for c in result["comments"]:
print(f'| {escape_cell(c["author"])} | {escape_cell(c["anchored_to"])} | '
f'{escape_cell(c["text"])} |', file=out)
# --------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("path", help=".docx, unzipped folder, or a document.xml")
ap.add_argument("--all-parts", action="store_true",
help="also scan headers and footers")
ap.add_argument("--markdown", action="store_true",
help="print a starter issues table instead of JSON")
ap.add_argument("--text", action="store_true",
help="print the document as plain text with changes accepted, "
"one paragraph per line prefixed by its clause number -- "
"greppable for defined terms and cross-references. Use this "
"if pandoc can't parse the file.")
ap.add_argument("--context-chars", type=int, default=700,
help="truncate paragraph context (0 = no limit; lower it for "
"very large redlines to control output size)")
ap.add_argument("--brief", action="store_true",
help="omit paragraph context and formatting detail -- a compact "
"index of a large redline")
ap.add_argument("--seq",
help="only output these change sequence numbers, e.g. "
"'12,15-20'. Use with the full context to work through a "
"long redline in batches without reloading all of it.")
args = ap.parse_args()
wanted = None
if args.seq:
wanted = set()
for chunk in args.seq.split(","):
chunk = chunk.strip()
if not chunk:
continue
if "-" in chunk:
lo, hi = chunk.split("-", 1)
wanted.update(range(int(lo), int(hi) + 1))
else:
wanted.add(int(chunk))
pkg = Package(args.path)
numbering = Numbering(pkg)
limit = args.context_chars or None
state = State(numbering, limit)
anchors = {}
parts = pkg.part_list(args.all_parts)
for label, name in parts:
raw = pkg.read(name)
if raw is None:
continue
root = ET.fromstring(raw)
state.part = label
state.last_clause = None
if label != "body":
state.heading_stack = []
body = root.find(W + "body")
walk_container(body if body is not None else root, state)
collect_comment_anchors(root, label, anchors)
comments, unanchored = load_comments(pkg, anchors, limit)
authors = sorted({
a for rec in state.content + state.formatting
for a in ([rec["author"]] if isinstance(rec.get("author"), str)
else (rec.get("author") or []))
if a
})
type_counts = {}
for rec in state.content:
type_counts[rec["type"]] = type_counts.get(rec["type"], 0) + 1
result = {
"summary": {
"source": str(Path(args.path).name),
"parts_scanned": [n for _, n in parts],
"content_change_count": len(state.content),
"formatting_change_count": len(state.formatting),
"comment_count": len(comments) + len(unanchored),
"change_types": type_counts,
"authors": authors,
"formatting_changes_needing_review": sum(
1 for r in state.formatting if r.get("numbering_changed")),
"clause_numbers_resolved": bool(numbering.abstract),
"replaced_tables": (
min(state.table_states.count("wholly-inserted"),
state.table_states.count("wholly-deleted"))
+ state.table_states.count("partly-replaced")),
},
"content_changes": state.content,
"formatting_changes": state.formatting,
"comments": comments,
"unanchored_comments": unanchored,
}
# Filters are applied after the summary is computed, so the counts always
# describe the whole document even when only a slice is printed -- that way
# a batch never looks like the full picture.
if wanted is not None:
result["content_changes"] = [
r for r in result["content_changes"] if r["sequence"] in wanted]
result["summary"]["filtered_to_sequences"] = sorted(wanted)
if args.brief:
for r in result["content_changes"]:
r.pop("paragraph_context", None)
r.pop("section_path", None)
result["formatting_changes"] = [
{k: v for k, v in r.items() if k in ("type", "location", "numbering_changed")}
for r in result["formatting_changes"]
]
if args.text:
for line in state.text_lines:
print(line)
elif args.markdown:
write_markdown(result, sys.stdout)
else:
json.dump(result, sys.stdout, indent=2, ensure_ascii=False)
print()
if __name__ == "__main__":
main()
SHA-256: 55dd01134b484cf43def9c4aa6088a4addcc6e22fe5b426de466713c46de967d