Bitácora 8–30 jun: contrato firmado, plan de actividades y skill proposal-pdf

- Comunicaciones (REGISTRO #15–#21): respuesta v1.1 (10-jun), luz verde (16-jun),
  contrato (envío 25-jun, ajustes + firma 26-jun), arranque con Erika y kickoff 1-jul
- Evidencia en fuentes/ (correos 8/10/16-jun, WhatsApp Paola y Erika)
- Contratos (sin firmar y firmado) en propuesta/
- Plan de actividades Etapa 0–3 + guion del kickoff (planeacion/)
- Skill proposal-pdf instalado (.claude/skills) + Propuesta-Balam.pdf regenerada
- README/PENDIENTES al día; limpieza de archivos sueltos; .gitignore (locks de Office)

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
JohannVelazquez
2026-06-30 12:42:45 -06:00
parent 633d05e330
commit 72b7602e8e
44 changed files with 2789 additions and 62 deletions
@@ -0,0 +1,459 @@
#!/usr/bin/env python3
"""
build_pdf.py — print-optimized PDF builder for editorial proposals/reports.
Pipeline (the reliable part — don't reinvent it):
1. Render the authored HTML to PDF with headless Chromium (Playwright),
letting CSS @page rules control the page geometry.
2. Resolve the table-of-contents page numbers in a SECOND pass: render once
with the {{PG_*}} tokens blanked, find the real start page of each section
by searching the rendered PDF for a unique body phrase, substitute the
numbers, then re-render.
3. Stamp a running footer (hairline + confidential line + page number) on
every page except the cover, using a registered TTF.
4. Write PDF metadata (title / author / subject).
Why two passes instead of computing pages from the DOM: with CSS paged media,
the mapping from DOM position to printed page is not linear (page margins and
break-before rules eat space). Searching the actually-rendered PDF sidesteps
all of it and is rock-solid.
Usage:
python3 build_pdf.py [config.json] # build
python3 build_pdf.py [config.json] --check # validate config + env only
If no config path is given it looks for ./pdf.config.json.
See pdf.config.example.json for the full schema.
"""
import sys, os, re, json, io, tempfile, pathlib
# Windows consoles default to cp1252 and choke on the ✓/⚠/✗ status glyphs this
# script prints. Force UTF-8 on stdout/stderr so progress output never crashes.
for _stream in (sys.stdout, sys.stderr):
try:
_stream.reconfigure(encoding="utf-8")
except Exception:
pass
# --- friendly dependency check -------------------------------------------- #
_MISSING = []
try:
from playwright.sync_api import sync_playwright
except Exception:
_MISSING.append("playwright")
try:
import pdfplumber
except Exception:
_MISSING.append("pdfplumber")
try:
import pypdf
except Exception:
_MISSING.append("pypdf")
try:
from reportlab.pdfgen import canvas
from reportlab.lib.units import mm
from reportlab.lib.colors import Color
from reportlab.pdfbase import pdfmetrics
from reportlab.pdfbase.ttfonts import TTFont as RLTTFont
except Exception:
_MISSING.append("reportlab")
if _MISSING:
sys.exit(
"Missing Python packages: " + ", ".join(_MISSING) + "\n"
"Install with:\n"
" pip install playwright pdfplumber pypdf reportlab\n"
" python -m playwright install chromium"
)
# --------------------------------------------------------------------------- #
# Page geometry
# --------------------------------------------------------------------------- #
# Named sizes in points (1pt = 1/72in). Used for the footer overlay canvas so
# it matches whatever Chromium printed.
PAGE_SIZES_PT = {
"letter": (612.0, 792.0),
"legal": (612.0, 1008.0),
"a4": (595.28, 841.89),
"a3": (841.89, 1190.55),
}
def resolve_page_size(page_size):
"""Return (playwright_format_or_None, width_pt, height_pt, width_mm, height_mm).
page_size may be a string ("Letter"/"A4"/...) or a dict
{"width_mm": .., "height_mm": ..} for a custom size.
"""
if isinstance(page_size, dict):
wmm = float(page_size["width_mm"]); hmm = float(page_size["height_mm"])
return None, wmm * 72 / 25.4, hmm * 72 / 25.4, wmm, hmm
key = str(page_size).strip().lower()
if key not in PAGE_SIZES_PT:
sys.exit(f"Unknown page_size {page_size!r}. Use one of "
f"{sorted(PAGE_SIZES_PT)} or a {{width_mm,height_mm}} object.")
wpt, hpt = PAGE_SIZES_PT[key]
return key.capitalize(), wpt, hpt, wpt * 25.4 / 72, hpt * 25.4 / 72
# --------------------------------------------------------------------------- #
# Config
# --------------------------------------------------------------------------- #
def load_config(path):
p = pathlib.Path(path)
if not p.exists():
sys.exit(f"Config not found: {path}\nCopy pdf.config.example.json and edit it.")
try:
cfg = json.loads(p.read_text(encoding="utf-8"))
except json.JSONDecodeError as e:
sys.exit(f"Config is not valid JSON ({path}): {e}")
cfg.setdefault("metadata", {})
cfg.setdefault("toc", {})
cfg.setdefault("footer", {})
f = cfg["footer"]
f.setdefault("skip_first_page", True)
f.setdefault("skip_pages", [])
f.setdefault("rule", True)
f.setdefault("font_size_pt", 7.5)
f.setdefault("margin_mm", 12) # distance from the bottom edge
f.setdefault("color", [0.46, 0.51, 0.57])
f.setdefault("page_number_format", "{page}")
cfg.setdefault("fonts", {})
cfg.setdefault("cover_full_bleed", True)
cfg.setdefault("page_size", "Letter")
cfg.setdefault("render_timeout_ms", 30000)
base = p.resolve().parent
for key in ("input_html", "output_pdf"):
if key in cfg and not os.path.isabs(cfg[key]):
cfg[key] = str(base / cfg[key])
ttf = cfg["fonts"].get("footer_ttf")
if ttf and not os.path.isabs(ttf):
cand = base / ttf
# only rewrite to a config-relative path if that file actually exists;
# otherwise leave the original (likely an absolute system font path)
if cand.exists():
cfg["fonts"]["footer_ttf"] = str(cand)
return cfg
# --------------------------------------------------------------------------- #
# Token / anchor helpers
# --------------------------------------------------------------------------- #
TOKEN_RE = re.compile(r"\{\{PG_([^}]+)\}\}")
def _norm(s):
return re.sub(r"\s+", " ", s or "")
def strip_comments(html):
# HTML comments never belong in the rendered PDF, and any {{PG_*}} examples
# inside them must not be treated as real tokens.
return re.sub(r"<!--.*?-->", "", html, flags=re.DOTALL)
def tokens_in_html(html):
return set(TOKEN_RE.findall(html))
def blank_tokens(html):
return TOKEN_RE.sub("", html)
def fill_tokens(html, pages):
def repl(m):
key = m.group(1)
if key not in pages:
sys.exit(f"Token {{{{PG_{key}}}}} has no matching entry in config 'toc'.")
return str(pages[key])
return TOKEN_RE.sub(repl, html)
# --------------------------------------------------------------------------- #
# Preflight validation
# --------------------------------------------------------------------------- #
def chromium_ok():
try:
with sync_playwright() as p:
b = p.chromium.launch()
b.close()
return True, ""
except Exception as e:
return False, str(e).splitlines()[0]
def preflight(cfg, html):
"""Validate config + environment. Returns list of warning strings; exits on
hard errors."""
problems, warnings = [], []
# input html
if not os.path.exists(cfg.get("input_html", "")):
problems.append(f"input_html not found: {cfg.get('input_html')!r}")
# token <-> toc consistency
toks = tokens_in_html(html)
toc_keys = {k for k in cfg["toc"] if not k.startswith("_")}
missing_cfg = toks - toc_keys # token in HTML, no anchor in config
unused_cfg = toc_keys - toks # anchor in config, no token in HTML
if missing_cfg:
problems.append("TOC tokens in the HTML with no anchor in config 'toc': "
+ ", ".join(sorted("{{PG_%s}}" % k for k in missing_cfg)))
if unused_cfg:
warnings.append("config 'toc' keys with no matching {{PG_*}} token in the HTML: "
+ ", ".join(sorted(unused_cfg)))
# footer font
ttf = cfg["fonts"].get("footer_ttf")
if ttf and not os.path.exists(ttf):
warnings.append(f"footer_ttf not found ({ttf}); falling back to Helvetica.")
# output dir
out = cfg.get("output_pdf")
if out:
os.makedirs(os.path.dirname(os.path.abspath(out)), exist_ok=True)
# chromium
ok, err = chromium_ok()
if not ok:
problems.append("Chromium failed to launch: " + err
+ "\n Run: python -m playwright install chromium")
if problems:
sys.exit("Preflight failed:\n" + "\n".join("" + p for p in problems))
return warnings
# --------------------------------------------------------------------------- #
# 1. Render
# --------------------------------------------------------------------------- #
def render(html_text, out_pdf, pw_format, width_pt, height_pt, timeout_ms, base_dir=None):
# Write the temp HTML in the SAME directory as the source document so that
# relative resources in the HTML — @font-face url("fonts/..."), <img src>,
# CSS background images — resolve against the document folder. (Rendering
# from the system temp dir would break every relative path.)
with tempfile.NamedTemporaryFile("w", prefix=".pdfbuild_", suffix=".html",
delete=False, encoding="utf-8", dir=base_dir) as fh:
fh.write(html_text)
tmp_path = fh.name
try:
with sync_playwright() as p:
browser = p.chromium.launch()
page = browser.new_page()
page.set_default_timeout(timeout_ms)
uri = pathlib.Path(tmp_path).resolve().as_uri()
# networkidle is ideal but can hang on a stuck resource; fall back to
# 'load' so a self-contained document always renders.
try:
page.goto(uri, wait_until="networkidle", timeout=timeout_ms)
except Exception:
page.goto(uri, wait_until="load", timeout=timeout_ms)
page.emulate_media(media="print")
pdf_kwargs = dict(
path=out_pdf,
print_background=True,
# margin 0 here so the CSS @page rules are the single source of
# truth — including @page:first{margin:0} for a full-bleed cover.
margin={"top": "0", "right": "0", "bottom": "0", "left": "0"},
prefer_css_page_size=True,
)
if pw_format:
pdf_kwargs["format"] = pw_format
else: # custom size
pdf_kwargs["width"] = f"{width_pt}pt"
pdf_kwargs["height"] = f"{height_pt}pt"
page.pdf(**pdf_kwargs)
browser.close()
finally:
os.unlink(tmp_path)
# --------------------------------------------------------------------------- #
# 2. TOC page-number resolution
# --------------------------------------------------------------------------- #
def detect_pages(pdf_path, anchors):
"""Map each TOC key to the 1-based page where its anchor phrase first appears."""
with pdfplumber.open(pdf_path) as pdf:
page_texts = [_norm(pg.extract_text() or "") for pg in pdf.pages]
pages, missing, ambiguous = {}, [], []
for key, phrase in anchors.items():
if key.startswith("_"):
continue
target = _norm(phrase)
hits = [i + 1 for i, txt in enumerate(page_texts) if target in txt]
if not hits:
missing.append((key, phrase))
else:
pages[key] = hits[0]
if len(hits) > 1:
ambiguous.append((key, hits))
if missing:
lines = "\n".join(f" - {k!r}: {p!r}" for k, p in missing)
sys.exit(
"Could not locate these TOC anchors in the rendered PDF.\n"
"Use a UNIQUE phrase from the section BODY (not its title), and avoid\n"
"the first letter of a drop-cap paragraph:\n" + lines
)
for k, hits in ambiguous:
print(f" ! anchor {k!r} appears on pages {hits}; using the first ({hits[0]}). "
"Use a more specific phrase if that's wrong.")
return pages
# --------------------------------------------------------------------------- #
# 3. Footer stamping + 4. Metadata
# --------------------------------------------------------------------------- #
def register_footer_font(ttf_path):
if ttf_path and os.path.exists(ttf_path):
try:
pdfmetrics.registerFont(RLTTFont("FooterFont", ttf_path))
return "FooterFont"
except Exception:
pass
return "Helvetica"
def make_overlay(page_number, total_pages, fcfg, font_name, width_pt, height_pt):
buf = io.BytesIO()
c = canvas.Canvas(buf, pagesize=(width_pt, height_pt))
left = 16 * mm
right_x = width_pt - 16 * mm
y_text = fcfg["margin_mm"] * mm
y_rule = y_text + 3.2 * mm
col = Color(*fcfg["color"])
if fcfg.get("rule", True):
c.setStrokeColor(Color(0.80, 0.83, 0.86))
c.setLineWidth(0.5)
c.line(left, y_rule, right_x, y_rule)
c.setFillColor(col)
c.setFont(font_name, float(fcfg["font_size_pt"]))
text = fcfg.get("text", "")
if text:
c.drawString(left, y_text, text)
num = fcfg.get("page_number_format", "{page}").format(page=page_number, pages=total_pages)
if num:
c.drawRightString(right_x, y_text, num)
c.showPage()
c.save()
buf.seek(0)
return pypdf.PdfReader(buf).pages[0]
def stamp_and_finalize(src_pdf, out_pdf, fcfg, font_name, metadata, width_pt, height_pt):
reader = pypdf.PdfReader(src_pdf)
writer = pypdf.PdfWriter()
total = len(reader.pages)
skip_first = fcfg.get("skip_first_page", True)
skip_pages = set(fcfg.get("skip_pages", []))
for idx, page in enumerate(reader.pages):
page_no = idx + 1
skip = (page_no in skip_pages) or (idx == 0 and skip_first)
if not skip:
page.merge_page(make_overlay(page_no, total, fcfg, font_name, width_pt, height_pt))
writer.add_page(page)
meta = {}
if metadata.get("title"): meta["/Title"] = metadata["title"]
if metadata.get("author"): meta["/Author"] = metadata["author"]
if metadata.get("subject"): meta["/Subject"] = metadata["subject"]
if metadata.get("author"): meta["/Creator"] = metadata["author"]
if metadata.get("keywords"): meta["/Keywords"] = metadata["keywords"]
if meta:
writer.add_metadata(meta)
with open(out_pdf, "wb") as fh:
writer.write(fh)
# --------------------------------------------------------------------------- #
# Post-build: warn if the design fonts didn't make it into the PDF
# --------------------------------------------------------------------------- #
# The design system is built on Caladea (serif display) + Carlito (sans body).
# If those aren't embedded, Chromium silently fell back to system fonts
# (Cambria/Calibri/Georgia/…) — it *looks close* but isn't the intended result,
# exactly the kind of regression that ships unnoticed. Flag it loudly.
DESIGN_FONTS = ("Caladea", "Carlito")
def report_embedded_fonts(pdf_path):
try:
reader = pypdf.PdfReader(pdf_path)
names = set()
for pg in reader.pages:
res = pg.get("/Resources")
fonts = res.get("/Font") if res else None
if not fonts:
continue
fobj = fonts.get_object()
for fk in fobj:
bf = fobj[fk].get_object().get("/BaseFont")
if bf:
names.add(str(bf).lstrip("/").split("+")[-1]) # drop subset prefix
except Exception:
return # a reporting step must never fail the build
present = sorted(names)
print(" fonts embedded:", ", ".join(present) if present else "(none)")
missing = [f for f in DESIGN_FONTS if not any(f in n for n in names)]
if missing:
print(" ⚠ design font(s) missing from the PDF: " + ", ".join(missing) + ".")
print(" Chromium fell back to system fonts, so the result looks")
print(" 'close but not identical' to the intended design. Ensure the")
print(" .ttf files sit next to the HTML (./fonts/) and the @font-face")
print(" url() paths resolve. See SKILL.md Fonts.")
# --------------------------------------------------------------------------- #
# Orchestration
# --------------------------------------------------------------------------- #
def main():
args = [a for a in sys.argv[1:]]
check_only = "--check" in args
args = [a for a in args if a != "--check"]
cfg_path = args[0] if args else "pdf.config.json"
cfg = load_config(cfg_path)
raw_html = pathlib.Path(cfg["input_html"]).read_text(encoding="utf-8") \
if os.path.exists(cfg["input_html"]) else ""
src_html = strip_comments(raw_html)
warnings = preflight(cfg, src_html)
for w in warnings:
print("" + w)
if check_only:
print("Preflight OK." + (" (with warnings)" if warnings else ""))
return
pw_format, width_pt, height_pt, _, _ = resolve_page_size(cfg["page_size"])
timeout_ms = int(cfg["render_timeout_ms"])
# render relative resources (fonts/images) against the document's folder
base_dir = os.path.dirname(os.path.abspath(cfg["input_html"])) or None
with tempfile.TemporaryDirectory() as td:
v1 = os.path.join(td, "v1.pdf")
v2 = os.path.join(td, "v2.pdf")
if any(not k.startswith("_") for k in cfg["toc"]):
print("Pass 1/2: resolving TOC page numbers…")
render(blank_tokens(src_html), v1, pw_format, width_pt, height_pt, timeout_ms, base_dir)
pages = detect_pages(v1, cfg["toc"])
print(" resolved:", pages)
html_final = fill_tokens(src_html, pages)
else:
html_final = blank_tokens(src_html)
print("Pass 2/2: rendering final document…")
render(html_final, v2, pw_format, width_pt, height_pt, timeout_ms, base_dir)
font_name = register_footer_font(cfg["fonts"].get("footer_ttf"))
stamp_and_finalize(v2, cfg["output_pdf"], cfg["footer"], font_name,
cfg["metadata"], width_pt, height_pt)
n = len(pypdf.PdfReader(cfg["output_pdf"]).pages)
print(f"✓ Wrote {cfg['output_pdf']} ({n} pages, {cfg['page_size']})")
report_embedded_fonts(cfg["output_pdf"])
if __name__ == "__main__":
main()
@@ -0,0 +1,55 @@
#!/usr/bin/env python3
"""render_check.py — rasterize PDF pages to PNG for visual verification.
Cross-platform: uses pypdfium2 (already pulled in with pdfplumber), so it needs
no poppler/pdftoppm. On Windows this is the practical way to do the "ALWAYS
verify visually" step from SKILL.md.
Usage:
python render_check.py Output.pdf # every page -> Output_p1.png, ...
python render_check.py Output.pdf 1 2 5 # only pages 1, 2, 5 (1-based)
python render_check.py Output.pdf 1 2 --scale 2 # higher resolution
"""
import sys, pathlib
try:
import pypdfium2 as pdfium
except Exception:
sys.exit("Missing pypdfium2 (it installs alongside pdfplumber).\n"
" pip install pypdfium2")
def main():
args = list(sys.argv[1:])
if not args:
sys.exit("Usage: python render_check.py <pdf> [pages...] [--scale N]")
scale = 1.5
if "--scale" in args:
i = args.index("--scale")
try:
scale = float(args[i + 1])
except (IndexError, ValueError):
sys.exit("--scale needs a number, e.g. --scale 2")
del args[i:i + 2]
pdf_path = args[0]
if not pathlib.Path(pdf_path).exists():
sys.exit(f"PDF not found: {pdf_path}")
pages = [int(a) for a in args[1:]]
pdf = pdfium.PdfDocument(pdf_path)
n = len(pdf)
idxs = [p - 1 for p in pages] if pages else range(n)
stem = pathlib.Path(pdf_path).with_suffix("")
for i in idxs:
if i < 0 or i >= n:
print(f" ! page {i + 1} out of range (1..{n})")
continue
out = f"{stem}_p{i + 1}.png"
pdf[i].render(scale=scale).to_pil().save(out)
print("wrote", out)
if __name__ == "__main__":
main()