← Files VeraARCHIVED FILE
vendor/modules/courseware/library.py
57.5 KB · Oct 2, 2026 · 00:29 UTC
"""Load reviewed course content and render it without model or network calls.
The checks here concern exact product membership, file identity and language
coverage. They do not judge professional relevance, correctness or understanding.
"""
from __future__ import annotations
import argparse
import csv
import hashlib
import html
import io
import json
import os
import re
import shutil
from pathlib import Path
from typing import Any
from .policy import local_unavailability, unavailable_local_workflows
__all__ = ["CourseError", "CourseLibrary", "main"]
_RESULT_COLUMNS = {
"check_results.csv": [
"movement_number",
"entry_date",
"amount_signed",
"currency",
"matched_support",
"amount_found",
"review_notes",
"source_row",
],
"journal_sample.csv": [
"entry_date",
"movement_number",
"line_number",
"account",
"line_desc",
"amount_signed",
"currency",
"source_row",
],
"fatture_summary.csv": [
"invoice_number",
"invoice_date",
"total_amount",
"currency",
"anomalies",
"file_name",
],
"structured_fiscal_fields.csv": [
"file_name",
"label",
"value",
"confidence",
"evidence",
"warnings",
],
}
class CourseError(ValueError):
"""A course cannot be safely reused with the current installed workflow."""
def _read(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8"))
def _digest(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()
def _inside(root: Path, relative: str) -> Path:
path = root / relative
if not relative or Path(relative).is_absolute() or ".." in Path(relative).parts:
raise CourseError("Course paths must be relative and contained")
if not path.is_file() or not path.resolve().is_relative_to(root.resolve()):
raise CourseError(f"Missing or foreign course file: {relative}")
return path
def _esc(value: Any) -> str:
return html.escape(str(value), quote=True)
def _pdf_pages(source: Path, destination: Path, prefix: str, ui: dict[str, Any]) -> str:
"""Render the original lesson PDF for hosts without an embedded PDF viewer."""
try:
import pymupdf
except ImportError as exc:
raise CourseError(
"PDF lesson previews require PyMuPDF in the selected runtime; "
"run this helper with the product's ready managed Python."
) from exc
figures = []
try:
with pymupdf.open(source) as document:
if document.needs_pass or not document.page_count:
raise CourseError("The lesson PDF cannot be opened for preview")
for number, page in enumerate(document, start=1):
filename = f"{prefix}-page-{number}.png"
page.get_pixmap(matrix=pymupdf.Matrix(2, 2), alpha=False).save(
destination / filename
)
label = ui["pdf_page"].format(page=number, total=document.page_count)
figures.append(
f"<figure class='source-pdf-page'><figcaption>{_esc(label)}</figcaption>"
f"<img src='{_esc(filename)}' alt='{_esc(source.name)} — {_esc(label)}'>"
"</figure>"
)
except (RuntimeError, ValueError) as exc:
raise CourseError(
f"The lesson PDF could not be rendered: {source.name}"
) from exc
return "".join(figures)
def _list(values: list[str]) -> str:
return (
"<ol class='method'>"
+ "".join(f"<li>{_esc(value)}</li>" for value in values)
+ "</ol>"
)
def _csv_input(text: str, ui: dict[str, Any] | None = None) -> str:
"""Label known teaching fields while preserving monetary and unknown text."""
try:
dialect = csv.Sniffer().sniff(text[:8192], delimiters=",;\t|")
except csv.Error:
dialect = csv.excel
rows = list(csv.reader(io.StringIO(text), dialect))
if not rows:
return ""
labels = (ui or {}).get("csv_input_fields", {})
values = (ui or {}).get("csv_input_values", {})
head = "".join(
f"<th scope='col'>{_esc(labels.get(value, value))}</th>" for value in rows[0]
)
body = "".join(
"<tr>"
+ "".join(
f"<td>{_esc(values.get(rows[0][index], {}).get(value, value) if index < len(rows[0]) else value)}</td>"
for index, value in enumerate(row)
)
+ "</tr>"
for row in rows[1:]
)
return (
"<div class='table-scroll'><table><thead><tr>"
+ head
+ "</tr></thead><tbody>"
+ body
+ "</tbody></table></div>"
)
def _xml_input(text: str, ui: dict[str, Any]) -> str:
"""Read invoice fields without validating, inferring or changing the XML."""
from defusedxml import ElementTree as ET
from defusedxml.common import DefusedXmlException
raw = f"<pre class='source-text'>{_esc(text)}</pre>"
# Unsupported or malformed XML remains inspectable as escaped source.
if (
len(text) > 1_000_000
or "<!DOCTYPE" in text.upper()
or "<!ENTITY" in text.upper()
):
return raw
try:
root = ET.fromstring(text)
except (ET.ParseError, DefusedXmlException):
return raw
if root.tag.rsplit("}", 1)[-1] != "FatturaElettronica":
return raw
for element in root.iter():
element.tag = element.tag.rsplit("}", 1)[-1]
labels = ui["invoice_fields"]
def row(key: str, value: str | None) -> str:
return (
f"<tr><th scope='row'>{_esc(labels[key])}</th><td>{_esc(value)}</td></tr>"
if value
else ""
)
header = root.find("FatturaElettronicaHeader")
parties = ""
if header is not None:
for key, node in [
("supplier", "CedentePrestatore"),
("customer", "CessionarioCommittente"),
]:
party = header.find(node + "/DatiAnagrafici/Anagrafica")
if party is not None:
name = (
party.findtext("Denominazione")
or " ".join(
party.findtext(field, "") for field in ("Nome", "Cognome")
).strip()
)
parties += row(key, name)
tables = []
for body in root.findall("FatturaElettronicaBody"):
details = body.find("DatiGenerali/DatiGeneraliDocumento")
if details is None:
continue
rows = parties + "".join(
row(key, details.findtext(tag))
for key, tag in [
("number", "Numero"),
("date", "Data"),
("type", "TipoDocumento"),
("currency", "Divisa"),
("total", "ImportoTotaleDocumento"),
]
)
for line in body.findall("DatiBeniServizi/DettaglioLinee"):
rows += row("description", line.findtext("Descrizione"))
tables.append(
"<div class='table-scroll'><table><tbody>" + rows + "</tbody></table></div>"
)
if not tables:
return raw
return (
f"<p>{_esc(ui['invoice_reading_note'])}</p>"
+ "".join(tables)
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{raw}</details>"
)
def _text_outline(
text: str,
images: dict[str, str] | None = None,
links: dict[str, str] | None = None,
) -> str:
"""Format Markdown and plain disclosures; never execute supplied HTML."""
blocks = []
paragraph = []
listing = None
lines = [*text.splitlines(), ""]
index = 0
disclosure_depth = 0
def inline(value: str) -> str:
def plain(part: str) -> str:
pieces = re.split(r"(`[^`]+`)", part)
return "".join(
(
f"<code>{_esc(piece[1:-1])}</code>"
if len(piece) > 2 and piece.startswith("`") and piece.endswith("`")
else re.sub(r"\*\*([^*]+)\*\*", r"<strong>\1</strong>", _esc(piece))
)
for piece in pieces
)
parts = []
end = 0
for match in re.finditer(r"(?<!!)\[([^\]]+)\]\((<[^>]+>|[^)]+)\)", value):
parts.append(plain(value[end : match.start()]))
label, target = match.groups()
if target.startswith("<") and target.endswith(">"):
target = target[1:-1]
if links and target in links:
parts.append(f"<a href='{_esc(links[target])}'>{_esc(label)}</a>")
else:
# Preserve an unrecorded reference as readable, inert text.
# Only recorded inputs above become navigable local links.
parts.append(f"{_esc(label)} ({_esc(target)})")
end = match.end()
parts.append(plain(value[end:]))
return "".join(parts)
def cells(line: str) -> list[str]:
return [
cell.strip().replace(r"\|", "|")
for cell in re.split(r"(?<!\\)\|", line.strip().strip("|"))
]
while index < len(lines):
line = lines[index].strip()
index += 1
disclosure = re.fullmatch(r"<details><summary>([^<>]+)</summary>", line)
close_disclosure = line == "</details>" and disclosure_depth > 0
if disclosure or close_disclosure or line == "<!-- Review Handoff -->":
if paragraph:
blocks.append("<p>" + inline(" ".join(paragraph)) + "</p>")
paragraph = []
if listing:
blocks.append(f"</{listing}>")
listing = None
if disclosure:
# Reconstruct only this passive form; attributes and HTML in
# the label are not accepted or forwarded to the browser.
blocks.append(
f"<details><summary>{_esc(disclosure.group(1))}</summary>"
)
disclosure_depth += 1
elif close_disclosure:
blocks.append("</details>")
disclosure_depth -= 1
continue
if "|" in line and index < len(lines):
header = cells(line)
separator = cells(lines[index])
if len(header) == len(separator) and all(
re.fullmatch(r":?-{3,}:?", cell) for cell in separator
):
if paragraph:
blocks.append("<p>" + inline(" ".join(paragraph)) + "</p>")
paragraph = []
if listing:
blocks.append(f"</{listing}>")
listing = None
index += 1
rows = []
while index < len(lines) and "|" in lines[index]:
row = cells(lines[index])
if len(row) != len(header):
break
rows.append(row)
index += 1
blocks.append(
"<div class='table-scroll markdown-table' tabindex='0'><table><thead><tr>"
+ "".join(f"<th scope='col'>{_esc(v)}</th>" for v in header)
+ "</tr></thead><tbody>"
+ "".join(
"<tr>" + "".join(f"<td>{inline(v)}</td>" for v in row) + "</tr>"
for row in rows
)
+ "</tbody></table></div>"
)
continue
prefix, _, content = line.partition(" ")
heading = 1 <= len(prefix) <= 6 and set(prefix) == {"#"}
bullet = line.startswith(("- ", "* "))
ordered = re.fullmatch(r"([0-9]{1,9})[.)]\s+(.+)", line)
list_kind = "ul" if bullet else "ol" if ordered else None
picture = re.fullmatch(r"!\[([^\]]*)\]\(([^)]+)\)", line)
if not line or heading or list_kind or picture:
if paragraph:
blocks.append("<p>" + inline(" ".join(paragraph)) + "</p>")
paragraph = []
if listing and listing != list_kind:
blocks.append(f"</{listing}>")
listing = None
if picture:
caption, target = picture.groups()
if images and target in images:
blocks.append(
f"<figure class='result-chart'><img src='{_esc(images[target])}' "
f"alt='{_esc(caption)}' loading='lazy'><figcaption>{_esc(caption)}</figcaption></figure>"
)
else:
blocks.append(
f"<p>{_esc(caption)} <span class='caption'>({_esc(target)})</span></p>"
)
elif heading:
level = min(len(prefix) + 2, 6)
blocks.append(f"<h{level}>{_esc(content)}</h{level}>")
elif list_kind:
if not listing:
start = (
f" start='{int(ordered.group(1))}'"
if ordered and int(ordered.group(1)) != 1
else ""
)
blocks.append(f"<{list_kind}{start}>")
listing = list_kind
item = ordered.group(2) if ordered else line[2:]
blocks.append(f"<li>{inline(item)}</li>")
elif line:
if listing:
blocks.append(f"</{listing}>")
listing = None
paragraph.append(line)
blocks.extend("</details>" for _ in range(disclosure_depth))
return "".join(blocks)
class CourseLibrary:
"""Read only courses belonging to the caller's current eligible catalog."""
def __init__(self, plugin_root: Path, eligible: set[str]) -> None:
self.root = plugin_root.resolve()
self.product = _read(self.root / ".codex-plugin/plugin.json")["name"]
self.assets = self.root / "assets/courses"
self.index = _read(self.assets / "index.json")
self.eligible = eligible - unavailable_local_workflows(self.product)
if self.index.get("product") != self.product:
raise CourseError("The course catalog belongs to another product")
def catalog(self) -> list[dict[str, Any]]:
"""Return only installed courses, including their explicit locales."""
return [
{"workflow": key, **value}
for key, value in sorted(self.index["courses"].items())
if key in self.eligible
]
def load(self, workflow: str, language: str) -> dict[str, Any]:
"""Fail closed on foreign, unavailable, altered or stale course content."""
if reason := local_unavailability(self.product, workflow):
raise CourseError(reason)
if workflow not in self.eligible or workflow not in self.index["courses"]:
raise CourseError("Choose a course from this product's installed catalog")
entry = self.index["courses"][workflow]
if language not in entry["languages"]:
raise CourseError(
"This course supports: "
+ ", ".join(entry["languages"])
+ "; ask the user to select one, without silently translating"
)
manifest = _inside(self.assets, entry["path"])
if _digest(manifest) != entry["sha256"]:
raise CourseError("Course content changed; rebuild the reviewed catalog")
course = _read(manifest)
if course.get("schema") == "mparanza.course.v1":
from . import legacy
try:
return legacy.CourseLibrary(self.root, self.eligible).load(
workflow, language
)
except legacy.CourseError as exc:
raise CourseError(str(exc)) from exc
if (
course.get("schema") != "mparanza.teaching_kit.v2"
or course.get("product") != self.product
or course.get("workflow") != workflow
or sorted(course["locales"]) != sorted(entry["languages"])
or len(course["seconds"]) != 6
or any(
type(seconds) is not int or seconds <= 0
for seconds in course["seconds"]
)
or not 300 <= sum(course["seconds"]) <= 480
):
raise CourseError("Invalid course identity, language coverage or duration")
if course["group"] not in {"workflow", "supporting_task"}:
raise CourseError("Invalid teaching-kit catalogue group")
content = course["locales"][language]
for field in (
"title",
"goal",
"scenario",
"scope",
"inputs",
"request",
"review",
"practice",
"success",
"repeat",
):
if not isinstance(content.get(field), str) or not content[field].strip():
raise CourseError(f"Teaching kit is missing its {field}")
for field in ("steps", "deliverables", "checkpoints"):
if (
not isinstance(content.get(field), list)
or not content[field]
or any(
not isinstance(value, str) or not value.strip()
for value in content[field]
)
):
raise CourseError(f"Teaching kit is missing its {field}")
for source in course["sources"]:
# The package contains its own component; source checkouts use the
# explicitly pinned sibling component, never an installed plugin.
relative = source["path"]
candidate = self.root / relative
if (
not candidate.is_file()
and "repository_path" in source
and (self.root.parent.parent / ".git").exists()
):
candidate = _inside(self.root.parent.parent, source["repository_path"])
else:
candidate = _inside(self.root, relative)
if _digest(candidate) != source["sha256"]:
raise CourseError(
f"Course needs editorial refresh after a workflow change: {relative}"
)
selected_roles = set()
seen_paths = set()
for asset in course.get("files", []):
path = _inside(manifest.parent, asset["path"])
if _digest(path) != asset["sha256"]:
raise CourseError("Teaching input changed; review and rebuild its kit")
if (
asset["path"] in seen_paths
or asset["role"] not in {"source", "practice"}
or not set(asset["languages"]) <= set(entry["languages"])
):
raise CourseError(
"Teaching kits may contain only uniquely identified source and practice files"
)
seen_paths.add(asset["path"])
if language in asset["languages"]:
selected_roles.add(asset["role"])
if selected_roles != {"source", "practice"}:
raise CourseError(
"Teaching kit needs source and practice files in the selected language"
)
return course
def render(self, workflow: str, language: str, destination: Path) -> dict[str, Any]:
"""Materialize prepared content in a fresh local directory; record no lesson."""
course = self.load(workflow, language)
if course["schema"] == "mparanza.course.v1":
from . import legacy
try:
return legacy.CourseLibrary(self.root, self.eligible).render(
workflow, language, destination
)
except legacy.CourseError as exc:
raise CourseError(str(exc)) from exc
destination = destination.expanduser().absolute()
if any(path.is_symlink() for path in (destination, *destination.parents)):
raise CourseError("Use an ordinary local destination, not a symlink")
if destination.exists():
raise CourseError("Use a fresh course directory; preserve existing work")
destination.mkdir(parents=True)
copy = course["locales"][language]
shared_assets = Path(__file__).parent / "assets"
ui = _read(shared_assets / "languages.json")[language]
for asset in (
"course.css",
"InstrumentSans-Regular.ttf",
"InstrumentSans-SemiBold.ttf",
"OFL.txt",
):
shutil.copyfile(_inside(shared_assets, asset), destination / asset)
source_dir = (self.assets / self.index["courses"][workflow]["path"]).parent
files = []
for asset in course.get("files", []):
if language not in asset["languages"]:
continue
target = destination / asset["path"]
target.parent.mkdir(parents=True, exist_ok=True)
shutil.copyfile(_inside(source_dir, asset["path"]), target)
displayed = dict(asset)
# A linked XML/Markdown file is not a browser document in every host.
# Provide an escaped reading view; the workflow still receives the
# original, byte-identical file, never this presentation wrapper.
if target.suffix.lower() in {
".xml",
".md",
".txt",
".pdf",
".csv",
".xlsx",
".xlsm",
".docx",
}:
preview = (
"input-"
+ hashlib.sha256(asset["path"].encode()).hexdigest()[:16]
+ ".html"
)
step = 1 if asset["role"] == "source" else 5
office_input = target.suffix.lower() in {".xlsx", ".xlsm", ".docx"}
if target.suffix.lower() in {".xlsx", ".docx"}:
from .document_view import office_body
content = office_body(target, ui)
elif office_input:
content = f"<p>{_esc(ui['office_input_note'].format(product=self.product.title()))}</p>"
elif target.suffix.lower() == ".pdf":
content = _pdf_pages(target, destination, Path(preview).stem, ui)
else:
content = f"<pre class='source-text'>{_esc(target.read_text(encoding='utf-8'))}</pre>"
if target.suffix.lower() == ".csv":
content = _csv_input(target.read_text(encoding="utf-8-sig"), ui)
elif target.suffix.lower() == ".xml":
content = _xml_input(target.read_text(encoding="utf-8"), ui)
elif target.suffix.lower() == ".md":
content = (
_text_outline(target.read_text(encoding="utf-8"))
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
)
body = (
f"<main class='report'><a href='course.html#step-{step}'>{_esc(ui['back_to_lesson'])}</a>"
f"<section><p class='eyebrow'>{_esc(ui['input_preview'])}</p>"
f"<h1>{_esc(target.name)}</h1><p class='caption'>{_esc(asset['path'])}</p>"
+ (
""
if office_input
else f"<p>{_esc(ui['input_preview_note'])}</p>"
)
+ f"<p><a href='{_esc(asset['path'])}' download>{_esc(ui['download_original'])}</a></p>"
f"{content}"
"</section></main>"
)
(destination / preview).write_text(
self._shell(target.name, language, body), encoding="utf-8"
)
displayed["preview_path"] = preview
files.append(displayed)
(destination / "course.html").write_text(
self._page(course, copy, ui, language, files), encoding="utf-8"
)
(destination / "teacher.md").write_text(
self._teacher(course, copy, ui), encoding="utf-8"
)
# This is an instruction handoff, never an execution or review receipt.
execution = {
"schema": "mparanza.teaching_execution_request.v1",
"product": self.product,
"workflow": workflow,
"language": language,
"skill": str(self.root / "skills" / workflow / "SKILL.md"),
"request": copy["request"],
"source_files": [
str(destination / f["path"]) for f in files if f["role"] == "source"
],
"practice_files": [
str(destination / f["path"]) for f in files if f["role"] == "practice"
],
"execution": course["execution"],
"steps": copy["steps"],
"checkpoints": copy["checkpoints"],
"deliverables": copy["deliverables"],
"practice": copy["practice"],
"success": copy["success"],
"sources": course["sources"],
"execution_receipt": False,
"rule": "Read the current own-product skill and delegated procedure completely. Prepare the real bound tutorial case from source_files. Execute one authorized working-thread step at a time. Open and explain the actual outputs. Rendering this kit completes no demo, practice or understanding checkpoint. Never replace a missing or blocked pipeline with authored output. Keep the tutorial local; a hosted step needs the user's separate explicit choice and normal workflow authority.",
}
(destination / "execution-request.json").write_text(
json.dumps(execution, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
receipt = {
"product": self.product,
"workflow": workflow,
"language": language,
"course_revision": course["revision"],
"prepared_material_only": True,
"execution_receipt": False,
"understanding_confirmed": False,
"course": str(destination / "course.html"),
"execution_request": str(destination / "execution-request.json"),
"source_files": execution["source_files"],
"practice_files": execution["practice_files"],
"teacher": str(destination / "teacher.md"),
"sources": course["sources"],
"prepared_artifacts": [
{
"path": path.relative_to(destination).as_posix(),
"sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
}
for path in sorted(destination.rglob("*"))
if path.is_file()
],
}
(destination / "course-provenance.json").write_text(
json.dumps(receipt, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return receipt
def _shell(self, title: str, language: str, body: str) -> str:
return (
"<!doctype html>\n" + f"<html lang='{_esc(language)}'><head>"
"<meta charset='utf-8'><meta name='google' content='notranslate'><meta name='viewport' content='width=device-width,initial-scale=1'>"
"<meta http-equiv='Content-Security-Policy' content=\"default-src 'none'; style-src 'self'; font-src 'self'; img-src 'self' data:; base-uri 'none'; form-action 'none'\">"
f"<title>{_esc(title)}</title><link rel='stylesheet' href='course.css'></head>"
f"<body>{body}</body></html>\n"
)
def results(
self,
workflow: str,
language: str,
lesson: Path,
record: str,
destination: Path,
columns: list[str] | None = None,
) -> dict[str, Any]:
"""Present verified outputs and original downloads without changing the run."""
from .execution import collect_execution
course = self.load(workflow, language)
lesson = lesson.resolve(strict=True)
execution = _read(_inside(lesson, record))
evidence = collect_execution(
root=lesson,
plugin_root=self.root,
product=self.product,
workflow=workflow,
phase=execution["phase"],
worker_thread_id=None,
record_path=record,
artifacts=execution["outputs"],
)
destination = destination.expanduser().absolute()
if destination.exists() or any(p.is_symlink() for p in destination.parents):
raise CourseError("Use a fresh ordinary result-view directory")
assets = Path(__file__).parent / "assets"
ui = _read(assets / "languages.json")[language]
sections = []
result_navigation = []
downloads = []
document_views = []
source_pdf_previews = []
source_links = {}
source_labels = {}
input_names = [Path(item["path"]).name for item in execution["inputs"]]
website = {}
decks = {}
if self.product == "clara" and workflow in {"deck-correction", "html-deck"}:
from .deck_view import deck_files
decks = deck_files(lesson, execution, self.root)
downloads.extend(
(source, relative, digest)
for source, (relative, digest) in decks.items()
)
if workflow == "presenza-digitale-studio":
from .website_view import website_files
website = website_files(lesson, execution)
downloads.extend(
(source, relative, digest)
for source, (relative, digest) in website.items()
)
# Citations may open only inputs already verified by this execution.
# Unknown paths and network URLs remain inert Markdown text.
for index, item in enumerate(execution["inputs"]):
source = _inside(lesson, item["path"])
if source in decks:
source_links[str(source)] = decks[source][0]
source_labels[str(source)] = ui["original_deck"]
continue
if source.suffix.lower() not in {
".md",
".txt",
".csv",
".xml",
".json",
".xlsx",
".docx",
".pdf",
}:
continue
view_path = f"source-{index + 1:02d}.html"
download = f"outputs/source-{index + 1:02d}-{source.name}"
source_links[str(source)] = view_path
source_label = source.name
if input_names.count(source.name) > 1:
relative = Path(item["path"])
phase = relative.parts[0].partition("-")[0]
context = (
ui[f"{phase}_source"]
if phase in {"demo", "practice"}
else relative.parent.as_posix()
)
source_label = f"{context} — {source.name}"
source_labels[str(source)] = source_label
downloads.append((source, download, item["sha256"]))
if source.suffix.lower() in {".xlsx", ".docx"}:
from .document_view import office_body
source_body = office_body(source, ui)
elif source.suffix.lower() == ".pdf":
source_body = f"<!--source-pdf-{index}-->"
source_pdf_previews.append((source, view_path, source_body))
else:
source_text = source.read_text(encoding="utf-8-sig")
source_body = f"<pre class='source-text'>{_esc(source_text)}</pre>"
if source.suffix.lower() == ".csv":
source_body = _csv_input(source_text, ui)
elif source.suffix.lower() == ".md":
source_body = _text_outline(source_text)
document_views.append(
(
view_path,
self._shell(
source_label,
language,
f"<main class='report'><a href='course.html'>{_esc(ui['back_to_results'])}</a>"
f"<h1>{_esc(source_label)}</h1><p>{_esc(item['path'])}</p>"
f"{source_body}"
f"<p><a href='{_esc(download)}' download='{_esc(source.name)}'>{_esc(ui['download_original'])}</a></p></main>",
),
)
)
# This workflow delivers its readable assignment plan in the standard
# run review; its JSON contract is the machine handoff, not the first read.
primary_result = (
"codex_run_review.md"
if (self.product, workflow) == ("clara", "advisory-brief-planner")
else (
"report.html"
if (self.product, workflow) == ("clara", "attribute-reporting")
else None
)
)
ordered_outputs = sorted(
execution["outputs"],
key=lambda item: (
0
if Path(item["path"]).name == primary_result
else (
2
if Path(item["path"]).name
in {
"artifact_card.md",
"codex_run_review.md",
"validation_package.md",
}
else 1
)
),
)
result_links = {
str(_inside(lesson, item["path"])): f"#result-{index + 1:02d}"
for index, item in enumerate(ordered_outputs)
}
for index, item in enumerate(ordered_outputs):
source = _inside(lesson, item["path"])
# Resolve relative citations against this actual output, and only
# to already verified input/output records. No link discovery.
verified_links = {**source_links, **result_links}
output_source_links = {
**verified_links,
**{
Path(os.path.relpath(path, source.parent)).as_posix(): view
for path, view in verified_links.items()
},
}
binary_document = source.suffix.lower() in {".xlsx", ".docx", ".pdf"}
binary_image = source.suffix.lower() in {".png", ".jpg", ".jpeg", ".webp"}
binary_package = bool(decks) and source.suffix.lower() == ".zip"
if source in website and source.suffix.lower() != ".html":
# Assets travel with the real site; do not display CSS as a lesson.
continue
if not (
binary_document or binary_image or binary_package
) and source.suffix.lower() not in {
".csv",
".md",
".txt",
".json",
".jsonl",
".html",
".xml",
}:
raise CourseError(
"Unsupported result format; use a reviewed document or text output"
)
download = f"outputs/{index + 1:02d}-{source.name}"
downloads.append((source, download, item["sha256"]))
link = f"<p><a href='{_esc(download)}' download='{_esc(source.name)}'>{_esc(ui['download_original'])}</a></p>"
text = (
""
if (binary_document or binary_image or binary_package)
else source.read_text(encoding="utf-8-sig")
)
content = (
f"<p>{_esc(ui['native_result_note'])}</p>"
if binary_document
else f"<pre class='source-text'>{_esc(text)}</pre>"
)
if binary_image:
from PIL import Image, UnidentifiedImageError
try:
with Image.open(source) as picture:
expected_format = {
".png": "PNG",
".jpg": "JPEG",
".jpeg": "JPEG",
".webp": "WEBP",
}[source.suffix.lower()]
if (
picture.format != expected_format
or picture.width * picture.height > 40_000_000
):
raise CourseError(
"Result image has an unsupported format or size"
)
picture.verify()
except (
UnidentifiedImageError,
OSError,
Image.DecompressionBombError,
) as exc:
raise CourseError("Result image is unreadable") from exc
label = ui["result_files"].get(source.name, source.name)
content = f"<figure class='result-chart'><img src='{_esc(download)}' alt='{_esc(label)}' loading='lazy'></figure>"
elif source in decks:
content = f"<p><a class='button' href='{_esc(decks[source][0])}'>{_esc(ui['open_result'])}</a></p>"
elif source in website:
content = f"<p><a class='button' href='{_esc(website[source][0])}'>{_esc(ui['open_result'])}</a></p>"
assets_links = "".join(
f"<li><a href='{_esc(relative)}' download='{_esc(asset.name)}'>{_esc(asset.name)}</a></li>"
for asset, (relative, _) in website.items()
if asset.suffix.lower() != ".html"
)
if assets_links:
content += f"<details><summary>{_esc(ui['website_assets'])}</summary><ul>{assets_links}</ul></details>"
elif source.suffix.lower() == ".html":
from .html_view import passive_html_body
reading_body = passive_html_body(
text,
omit_controls=workflow
in {"business-planning", "attribute-reporting"},
report_metadata=(
workflow == "business-planning"
or (self.product, workflow) == ("clara", "attribute-reporting")
),
)
view_path = f"result-{index + 1:02d}.html"
label = ui["result_files"].get(source.name, source.name)
input_links = "".join(
f"<li><a href='{_esc(view)}'>{_esc(source_labels[path])}</a></li>"
for path, view in source_links.items()
)
source_section = (
f"<section><h2>{_esc(ui['input_preview'])}</h2><ul>{input_links}</ul></section>"
if input_links
else ""
)
document_views.append(
(
view_path,
self._shell(
label,
language,
f"<main class='report'><a href='course.html'>{_esc(ui['back_to_results'])}</a>"
f"<h1>{_esc(label)}</h1><p>{_esc(ui['html_reading_note'])}</p>"
f"<article class='html-reading'>{reading_body}</article>"
+ source_section
+ link
+ "</main>",
),
)
)
content = f"<p><a class='button' href='{view_path}'>{_esc(ui['open_result'])}</a></p>"
link = ""
elif source.suffix.lower() in {".xlsx", ".docx"}:
from .document_view import office_body
view_path = f"result-{index + 1:02d}.html"
label = ui["result_files"].get(source.name, source.name)
document_body = (
f"<main class='report'><a href='course.html'>{_esc(ui['back_to_results'])}</a>"
f"<h1>{_esc(label)}</h1><p class='caption'>{_esc(source.name)}</p>"
+ office_body(
source,
ui,
first_sheet=(
"Exceptions"
if source.name == "exception_workpaper.xlsx"
else None
),
)
+ link
+ "</main>"
)
document_views.append(
(view_path, self._shell(label, language, document_body))
)
content = f"<p><a class='button' href='{view_path}'>{_esc(ui['open_result'])}</a></p>"
link = ""
elif source.suffix.lower() == ".csv":
if (
source.name
in {"scenario_summary.csv", "assumption_application_ledger.csv"}
and workflow == "sales-plan"
):
from .sales_plan_view import sales_plan_body
table = sales_plan_body(source.name, text, ui, language)
content = (
table
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
)
label = ui["result_files"].get(source.name, source.name)
sections.append(
f"<section id='result-{index + 1:02d}'><h2>{_esc(label)}</h2><p class='caption'>{_esc(source.name)}</p>{link}{content}</section>"
)
continue
reader = csv.DictReader(io.StringIO(text))
fields = reader.fieldnames or []
preferred = columns or _RESULT_COLUMNS.get(source.name, fields)
if workflow == "previdenza-inps" and not columns:
preferred = {
"timeline.csv": [
"date",
"description",
"source_fact_ids",
"review_status",
],
"evidence_matrix.csv": [
"fact_id",
"statement",
"review_status",
"document_id",
"locator_kind",
"locator_value",
"quote",
],
"file_inventory.csv": [
"document_id",
"relative_path",
"readability",
"limitations",
],
}.get(source.name, preferred)
shown = [c for c in preferred if c in fields] or fields
rows = list(reader)
head = "".join(
f"<th scope='col'>{_esc(ui['result_columns'].get(c, c))}</th>"
for c in shown
)
cells = "".join(
"<tr>"
+ "".join(
f"<td>{_esc(ui.get('result_values', {}).get(c, {}).get(row.get(c), row.get(c)) or '')}</td>"
for c in shown
)
+ "</tr>"
for row in rows
)
table = (
f"<div class='table-scroll'><table><thead><tr>{head}</tr></thead><tbody>{cells}</tbody></table></div>"
if rows
else f"<p>{_esc(ui['no_rows'])}</p>"
)
content = (
table
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
)
elif source.name == "client_email.txt":
content = (
_text_outline(text, links=output_source_links)
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
)
elif source.suffix.lower() == ".md":
# Resolve only sibling images already covered by execution evidence.
# Never fetch Markdown URLs or discover files outside the record.
images = {
Path(
other["path"]
).name: f"outputs/{n + 1:02d}-{Path(other['path']).name}"
for n, other in enumerate(ordered_outputs)
if Path(other["path"]).parent == Path(item["path"]).parent
and Path(other["path"]).suffix.lower()
in {".png", ".jpg", ".jpeg", ".webp"}
}
content = (
_text_outline(text, images, output_source_links)
+ f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
)
elif source.suffix.lower() in {".json", ".jsonl", ".xml"}:
content = f"<details><summary>{_esc(ui['full_file'])}</summary>{content}</details>"
technical_summary = (
source.name
in {
"artifact_card.md",
"run_summary.md",
"codex_run_review.md",
"validation_package.md",
}
or (
(self.product, workflow) == ("clara", "attribute-reporting")
and (
source.name == "execution_notes.md"
or source.suffix.lower() == ".csv"
)
)
or (
source.name == "review_dossier.md"
and any(
Path(other["path"]).name == "review_dossier.html"
and Path(other["path"]).parent == Path(item["path"]).parent
for other in ordered_outputs
)
)
)
if source.name == primary_result:
technical_summary = False
if technical_summary and len(ordered_outputs) > 1:
content = f"<details><summary>{_esc(ui['full_file'])}</summary><pre class='source-text'>{_esc(text)}</pre></details>"
label = ui["result_files"].get(source.name, source.name)
if source in website:
label = ui["website_result"]
if (
source.suffix.lower() == ".md"
and (label == source.name or source.name == primary_result)
and not technical_summary
):
# Content-addressed filenames identify versions, not documents.
# Prefer the document's own plain title in the reading view.
heading, _, body_text = text.partition("\n")
if heading.startswith("# ") and heading[2:].strip():
label = heading[2:].strip()
content = (
_text_outline(body_text, images, output_source_links)
+ f"<details><summary>{_esc(ui['full_file'])}</summary>"
+ f"<pre class='source-text'>{_esc(text)}</pre></details>"
)
if technical_summary and len(ordered_outputs) == 1:
# A summary can be the workflow's actual delivered result.
# Do not hide the only result because of its conventional filename.
heading, _, body_text = text.partition("\n")
if heading.startswith("# "):
label = heading[2:].strip()
content = _text_outline(body_text, links=output_source_links)
if source.suffix.lower() == ".xml":
label = ui["xml_result"]
if source in decks and workflow == "deck-correction":
label = ui["corrected_deck"]
if (self.product, workflow) == ("clara", "advisory-case-director"):
# Native workpaper history retains its original title and bytes.
# Identify its role in the reading view instead of presenting an
# earlier recommendation as another current document.
if source.suffix.lower() == ".md" and source.stem.startswith(
"advisory_workpaper."
):
label = f"{ui['previous_version']} — {label}"
result_navigation.append(
f"<a href='#result-{index + 1:02d}'>{_esc(label)}</a>"
)
sections.append(
f"<section id='result-{index + 1:02d}'><h2>{_esc(label)}</h2><p class='caption'>{_esc(source.name)}</p>{link}{content}</section>"
)
title = course["locales"][language]["title"]
input_index = (
f"<section><h2>{_esc(ui['input_preview'])}</h2><ul>"
+ "".join(
f"<li><a href='{_esc(view)}'>{_esc(source_labels[path])}</a></li>"
for path, view in source_links.items()
)
+ "</ul></section>"
if source_links
else ""
)
body = (
f"<main class='report'><p class='eyebrow'>{_esc(ui['live_results'])}</p>"
f"<h1>{_esc(title)}</h1><p>{_esc(ui['result_view_note'])}</p>"
+ (
f"<nav aria-label='{_esc(ui['result_navigation'])}'>"
+ "".join(result_navigation)
+ "</nav>"
if len(result_navigation) > 1
else ""
)
+ "".join(sections)
+ input_index
+ "</main>"
)
destination.mkdir(parents=True)
(destination / "outputs").mkdir()
for source, view_path, placeholder in source_pdf_previews:
pages = _pdf_pages(source, destination, Path(view_path).stem, ui)
document_views = [
(
(relative, page.replace(placeholder, pages))
if relative == view_path
else (relative, page)
)
for relative, page in document_views
]
for source, relative, expected_hash in downloads:
copied = destination / relative
copied.parent.mkdir(parents=True, exist_ok=True)
shutil.copyfile(source, copied)
if _digest(copied) != expected_hash:
raise CourseError("Result changed while preparing its download")
for relative, page in document_views:
(destination / relative).write_text(page, encoding="utf-8")
for name in (
"course.css",
"InstrumentSans-Regular.ttf",
"InstrumentSans-SemiBold.ttf",
"OFL.txt",
):
shutil.copyfile(_inside(assets, name), destination / name)
(destination / "course.html").write_text(
self._shell(title, language, body), encoding="utf-8"
)
provenance = {
"presentation_only": True,
"execution": evidence,
"outputs": execution["outputs"],
}
(destination / "result-view.json").write_text(
json.dumps(provenance, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return {"course": str(destination / "course.html"), **provenance}
def _page(
self,
course: dict[str, Any],
copy: dict[str, Any],
ui: dict[str, Any],
language: str,
files: list[dict[str, Any]],
) -> str:
labels = ui["stages"]
nav = "".join(
f"<a href='#step-{i}'><span>0{i+1}</span>{_esc(label)}</a>"
for i, label in enumerate(labels)
)
def file_links(role: str) -> str:
selected = [file for file in files if file["role"] == role]
names = [Path(file["path"]).name for file in selected]
links = []
for file in selected:
path = Path(file["path"])
label = path.name
if names.count(label) > 1:
# Keep the client/subfolder visible when basenames collide.
label = Path(*path.parts[2:]).as_posix()
links.append(
f"<li><a href='{_esc(file.get('preview_path', file['path']))}'>{_esc(label)}</a></li>"
)
return "".join(links)
downloads = file_links("source")
practice_files = file_links("practice")
parts = [
f"<p class='lead'>{_esc(copy['scenario'])}</p><p>{_esc(copy['scope'])}</p>",
f"<p>{_esc(copy['inputs'])}</p><ul>{downloads}</ul><blockquote>{_esc(copy['request'])}</blockquote>",
_list(copy["steps"])
+ f"<aside class='paired'>{_esc(ui['live_rule'])}</aside>",
_list(copy["deliverables"]) + f"<p>{_esc(copy['review'])}</p>",
_list(copy["checkpoints"]) + f"<p>{_esc(ui['checkpoint_rule'])}</p>",
f"<blockquote>{_esc(copy['practice'])}</blockquote><ul>{practice_files}</ul><p>{_esc(copy['success'])}</p><p>{_esc(copy['repeat'])}</p>",
]
sections = "".join(
f"<section id='step-{i}' class='lesson-step'><div class='section-label'><span>0{i+1}</span><h2>{_esc(labels[i])}</h2></div>"
f"<div class='section-body'>{part}</div></section>"
for i, part in enumerate(parts)
)
body = (
f"<header class='masthead'><b>{_esc(self.product.title())}</b><span>{_esc(ui['series'])}</span><span>5–8 min · {_esc(language.upper())}</span></header>"
f"<main><div class='hero'><p class='eyebrow'>{_esc(ui['prepared'])}</p><h1>{_esc(copy['title'])}</h1>"
f"<p class='lead'>{_esc(copy['goal'])}</p><p class='caption'>{_esc(ui['timing'])}</p></div>"
f"<aside class='paired'><b>{_esc(ui['two_threads'])}</b><p>{_esc(ui['pair_explanation'])}</p></aside>"
f"<nav aria-label='{_esc(ui['contents'])}'>{nav}</nav>{sections}"
f"<footer><p>{_esc(ui['kit_notice'])}</p><p>{_esc(ui['privacy'])}</p>"
f"<a href='course-provenance.json'>{_esc(ui['provenance'])}</a> · {_esc(course['revision'])}</footer></main>"
)
return self._shell(copy["title"], language, body)
def _teacher(
self, course: dict[str, Any], copy: dict[str, Any], ui: dict[str, Any]
) -> str:
blocks = [
copy["goal"] + "\n\n" + copy["scenario"] + "\n\n" + copy["scope"],
copy["inputs"] + "\n\n" + copy["request"],
"\n\n".join(copy["steps"]) + "\n\n" + ui["live_rule"],
"\n\n".join(copy["deliverables"]) + "\n\n" + copy["review"],
"\n\n".join(copy["checkpoints"]) + "\n\n" + ui["checkpoint_rule"],
copy["practice"] + "\n\n" + copy["success"] + "\n\n" + copy["repeat"],
]
text = f"# {copy['title']}\n\n{ui['timing']}\n\n{ui['pair_explanation']}\n\n{ui['teacher_rule']}\n\n"
for i, block in enumerate(blocks):
text += f"## {i+1}. {ui['stages'][i]} · {course['seconds'][i]} s\n\n{ui['cues'][i]}\n\n{block}\n\n"
return text + ui["kit_notice"] + "\n\n" + ui["privacy"] + "\n"
def main(plugin_root: Path, eligible: set[str], argv: list[str] | None = None) -> int:
"""Local CLI: catalog, inspect or render packaged content without user telemetry."""
from .execution import ExecutionError
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"action", choices=("list", "show", "render", "serve", "results")
)
parser.add_argument("--workflow")
parser.add_argument("--language")
parser.add_argument("--output-dir", type=Path)
parser.add_argument("--lesson-dir", type=Path)
parser.add_argument("--execution-record")
parser.add_argument("--columns", nargs="+")
args = parser.parse_args(argv)
try:
if args.action == "serve":
if not args.output_dir:
parser.error("serve requires --output-dir pointing to a rendered kit")
from .preview import serve
serve(args.output_dir)
return 0
library = CourseLibrary(plugin_root, eligible)
if args.action == "list":
result: Any = library.catalog()
else:
if not args.workflow or not args.language:
parser.error("--workflow and --language are required")
if args.action == "results":
if (
not args.output_dir
or not args.lesson_dir
or not args.execution_record
):
parser.error(
"results requires --lesson-dir, --execution-record and --output-dir"
)
result = library.results(
args.workflow,
args.language,
args.lesson_dir,
args.execution_record,
args.output_dir,
args.columns,
)
elif args.action == "render":
if not args.output_dir:
parser.error("render requires --output-dir")
result = library.render(args.workflow, args.language, args.output_dir)
else:
course = library.load(args.workflow, args.language)
result = {
"product": course["product"],
"workflow": course["workflow"],
"language": args.language,
"revision": course["revision"],
"seconds": course["seconds"],
"sources_current": True,
"source_count": len(course["sources"]),
"execution": course["execution"],
"group": course["group"],
"content": course["locales"][args.language],
"files": [asset["path"] for asset in course.get("files", [])],
}
# JSON escapes preserve all localized text on legacy Windows consoles.
print(json.dumps(result, ensure_ascii=True, indent=2))
except (
CourseError,
ExecutionError,
OSError,
KeyError,
json.JSONDecodeError,
) as exc:
parser.exit(2, f"Course unavailable: {exc}\n")
return 0
SHA-256: defe2c748452578308d1157ff0abc6ad69f4958429bb69fa2ac016dda5ae7e52