Initial commit: PDF→static HTML/CSS site generator

main.py converts every PDF in examples/ into a browseable, mobile-responsive
HTML archive under output/ using poppler-utils. Includes the two NETgazet
sample PDFs, project metadata, OpenWolf scaffolding, and README covering
usage, a watch-folder script, and WordPress iframe embedding.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
2026-06-03 16:00:07 +02:00
co-authored by Claude Opus 4.7
commit 93174e1e09
33 changed files with 3652 additions and 0 deletions
+366
View File
@@ -0,0 +1,366 @@
"""Convert every PDF in examples/ into a static HTML/CSS site under output/.
Shells out to poppler-utils (pdftocairo, pdftotext, pdfinfo). No third-party deps.
Usage: python main.py (or: uv run python main.py)
"""
from __future__ import annotations
import html
import re
import shutil
import subprocess
import sys
from pathlib import Path
ROOT = Path(__file__).parent
EXAMPLES_DIR = ROOT / "examples"
OUTPUT_DIR = ROOT / "output"
DPI_FULL = 150
DPI_THUMB = 30
JPEG_QUALITY_FULL = 85
JPEG_QUALITY_THUMB = 80
REQUIRED_BINARIES = ("pdftocairo", "pdftotext", "pdfinfo")
def main() -> int:
missing = [b for b in REQUIRED_BINARIES if shutil.which(b) is None]
if missing:
print(f"error: missing poppler-utils binaries: {', '.join(missing)}", file=sys.stderr)
print("install with: sudo apt install poppler-utils", file=sys.stderr)
return 1
if not EXAMPLES_DIR.is_dir():
print(f"error: {EXAMPLES_DIR} does not exist", file=sys.stderr)
return 1
pdfs = sorted(EXAMPLES_DIR.glob("*.pdf"))
if not pdfs:
print(f"error: no PDFs found in {EXAMPLES_DIR}", file=sys.stderr)
return 1
OUTPUT_DIR.mkdir(exist_ok=True)
write_shared_css()
issues = []
for pdf in pdfs:
print(f"converting {pdf.name} ...")
issues.append(convert_pdf(pdf))
write_root_index(issues)
print(f"\ndone. open {OUTPUT_DIR / 'index.html'} in a browser.")
return 0
def convert_pdf(pdf_path: Path) -> dict:
stem = pdf_path.stem
out = OUTPUT_DIR / stem
reset_dir(out)
page_count = pdf_page_count(pdf_path)
page_texts = extract_text_per_page(pdf_path, page_count)
for n in range(1, page_count + 1):
render_jpeg(pdf_path, out / f"page-{n:02d}", n, DPI_FULL, JPEG_QUALITY_FULL)
render_jpeg(pdf_path, out / f"thumb-{n:02d}", n, DPI_THUMB, JPEG_QUALITY_THUMB)
write_page_html(out, stem, n, page_count, page_texts[n - 1])
write_issue_index(out, stem, page_count)
return {"stem": stem, "title": stem, "pages": page_count}
def reset_dir(path: Path) -> None:
if path.exists():
shutil.rmtree(path)
path.mkdir(parents=True)
def pdf_page_count(pdf_path: Path) -> int:
out = subprocess.run(
["pdfinfo", str(pdf_path)], check=True, capture_output=True, text=True
).stdout
m = re.search(r"^Pages:\s+(\d+)", out, re.MULTILINE)
if not m:
raise RuntimeError(f"could not parse page count from pdfinfo for {pdf_path}")
return int(m.group(1))
def extract_text_per_page(pdf_path: Path, page_count: int) -> list[str]:
"""Return a list of page texts. pdftotext separates pages with form-feed (\\f)."""
out = subprocess.run(
["pdftotext", "-layout", str(pdf_path), "-"],
check=True,
capture_output=True,
text=True,
).stdout
pages = out.split("\f")
# pdftotext appends a trailing form-feed → one extra empty element; trim it.
if pages and pages[-1] == "":
pages.pop()
# Pad/truncate so the list matches page_count exactly (defensive).
while len(pages) < page_count:
pages.append("")
return pages[:page_count]
def render_jpeg(pdf_path: Path, out_stem: Path, page: int, dpi: int, quality: int) -> None:
"""pdftocairo always appends -<page>.jpg unless -singlefile is used.
We render one page at a time with -singlefile so the output filename is exact.
"""
subprocess.run(
[
"pdftocairo",
"-jpeg",
"-jpegopt",
f"quality={quality}",
"-r",
str(dpi),
"-f",
str(page),
"-l",
str(page),
"-singlefile",
str(pdf_path),
str(out_stem), # pdftocairo appends .jpg automatically
],
check=True,
capture_output=True,
)
# ---------- HTML emission ----------
def write_shared_css() -> None:
(OUTPUT_DIR / "styles.css").write_text(SHARED_CSS, encoding="utf-8")
def write_root_index(issues: list[dict]) -> None:
cards = "\n".join(
f""" <a class="card" href="{html.escape(i['stem'])}/index.html">
<img src="{html.escape(i['stem'])}/thumb-01.jpg" alt="Cover of {html.escape(i['title'])}">
<h2>{html.escape(i['title'])}</h2>
<p>{i['pages']} pages</p>
</a>"""
for i in issues
)
page = PAGE_SHELL.format(
title="NETgazet archive",
css_href="styles.css",
body=f""" <header>
<h1>NETgazet archive</h1>
<p>Static HTML rendering of the available editions.</p>
</header>
<main class="card-grid">
{cards}
</main>""",
)
(OUTPUT_DIR / "index.html").write_text(page, encoding="utf-8")
def write_issue_index(out: Path, stem: str, page_count: int) -> None:
thumbs = "\n".join(
f""" <a class="thumb" href="page-{n:02d}.html">
<img src="thumb-{n:02d}.jpg" alt="Page {n}" loading="lazy">
<span>Page {n}</span>
</a>"""
for n in range(1, page_count + 1)
)
page = PAGE_SHELL.format(
title=html.escape(stem),
css_href="../styles.css",
body=f""" <header>
<p><a href="../index.html">← all editions</a></p>
<h1>{html.escape(stem)}</h1>
<p>{page_count} pages — tap a thumbnail to read.</p>
</header>
<main class="thumb-grid">
{thumbs}
</main>""",
)
(out / "index.html").write_text(page, encoding="utf-8")
def write_page_html(out: Path, stem: str, n: int, total: int, text: str) -> None:
prev_link = (
f'<a href="page-{n - 1:02d}.html" rel="prev">← Page {n - 1}</a>'
if n > 1
else '<span class="nav-disabled">← Page</span>'
)
next_link = (
f'<a href="page-{n + 1:02d}.html" rel="next">Page {n + 1} →</a>'
if n < total
else '<span class="nav-disabled">Page →</span>'
)
text_block = (
f""" <details class="page-text">
<summary>Show page text</summary>
<pre>{html.escape(text)}</pre>
</details>"""
if text.strip()
else ""
)
body = f""" <header>
<p><a href="index.html">← {html.escape(stem)}</a> · <a href="../index.html">all editions</a></p>
<h1>Page {n} <span class="of">of {total}</span></h1>
</header>
<nav class="pager">
{prev_link}
<a href="index.html" class="nav-up">Index</a>
{next_link}
</nav>
<main>
<figure class="page">
<img src="page-{n:02d}.jpg" alt="Page {n} of {html.escape(stem)}">
</figure>
{text_block}
</main>
<nav class="pager pager-bottom">
{prev_link}
<a href="index.html" class="nav-up">Index</a>
{next_link}
</nav>"""
page = PAGE_SHELL.format(
title=f"{html.escape(stem)} — page {n}",
css_href="../styles.css",
body=body,
)
(out / f"page-{n:02d}.html").write_text(page, encoding="utf-8")
# ---------- Templates ----------
PAGE_SHELL = """<!doctype html>
<html lang="nl">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>{title}</title>
<link rel="stylesheet" href="{css_href}">
</head>
<body>
{body}
</body>
</html>
"""
SHARED_CSS = """:root {
--fg: #1a1a1a;
--muted: #666;
--bg: #fafafa;
--card: #fff;
--border: #e2e2e2;
--accent: #1a4480;
}
* { box-sizing: border-box; }
html, body { margin: 0; padding: 0; }
body {
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif;
color: var(--fg);
background: var(--bg);
line-height: 1.45;
max-width: 1000px;
margin: 0 auto;
padding: 1rem;
}
img { max-width: 100%; height: auto; display: block; }
a { color: var(--accent); text-decoration: none; }
a:hover { text-decoration: underline; }
header { margin-bottom: 1.5rem; }
header h1 { margin: 0.25rem 0; font-size: 1.5rem; }
header p { margin: 0.25rem 0; color: var(--muted); }
.of { color: var(--muted); font-weight: normal; font-size: 0.85em; }
.card-grid {
display: grid;
grid-template-columns: repeat(auto-fill, minmax(220px, 1fr));
gap: 1.25rem;
}
.card {
display: block;
background: var(--card);
border: 1px solid var(--border);
border-radius: 6px;
padding: 0.75rem;
color: inherit;
}
.card:hover { border-color: var(--accent); text-decoration: none; }
.card img { border: 1px solid var(--border); margin-bottom: 0.5rem; }
.card h2 { margin: 0.25rem 0; font-size: 1.05rem; }
.card p { margin: 0; color: var(--muted); font-size: 0.9rem; }
.thumb-grid {
display: grid;
grid-template-columns: repeat(auto-fill, minmax(140px, 1fr));
gap: 1rem;
}
.thumb {
display: block;
text-align: center;
color: inherit;
}
.thumb img {
border: 1px solid var(--border);
border-radius: 3px;
margin: 0 auto 0.25rem;
}
.thumb:hover img { border-color: var(--accent); }
.thumb span { font-size: 0.85rem; color: var(--muted); }
.pager {
display: flex;
flex-wrap: wrap;
gap: 0.5rem;
justify-content: space-between;
align-items: center;
margin: 1rem 0;
padding: 0.5rem 0;
border-top: 1px solid var(--border);
border-bottom: 1px solid var(--border);
font-size: 0.95rem;
}
.pager .nav-up { font-weight: 600; }
.pager .nav-disabled { color: #bbb; }
.pager-bottom { margin-top: 1rem; }
.page { margin: 0; }
.page img {
border: 1px solid var(--border);
box-shadow: 0 2px 8px rgba(0,0,0,0.06);
margin: 0 auto;
}
.page-text {
margin-top: 1rem;
background: var(--card);
border: 1px solid var(--border);
border-radius: 4px;
padding: 0.5rem 0.75rem;
}
.page-text summary {
cursor: pointer;
color: var(--accent);
font-size: 0.95rem;
}
.page-text pre {
white-space: pre-wrap;
word-wrap: break-word;
font-size: 0.85rem;
line-height: 1.4;
margin: 0.75rem 0 0;
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
}
@media (max-width: 480px) {
body { padding: 0.5rem; }
header h1 { font-size: 1.25rem; }
.pager { font-size: 0.85rem; }
}
"""
if __name__ == "__main__":
sys.exit(main())