Initial commit: PDF→static HTML/CSS site generator
main.py converts every PDF in examples/ into a browseable, mobile-responsive HTML archive under output/ using poppler-utils. Includes the two NETgazet sample PDFs, project metadata, OpenWolf scaffolding, and README covering usage, a watch-folder script, and WordPress iframe embedding. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,366 @@
|
||||
"""Convert every PDF in examples/ into a static HTML/CSS site under output/.
|
||||
|
||||
Shells out to poppler-utils (pdftocairo, pdftotext, pdfinfo). No third-party deps.
|
||||
Usage: python main.py (or: uv run python main.py)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).parent
|
||||
EXAMPLES_DIR = ROOT / "examples"
|
||||
OUTPUT_DIR = ROOT / "output"
|
||||
|
||||
DPI_FULL = 150
|
||||
DPI_THUMB = 30
|
||||
JPEG_QUALITY_FULL = 85
|
||||
JPEG_QUALITY_THUMB = 80
|
||||
|
||||
REQUIRED_BINARIES = ("pdftocairo", "pdftotext", "pdfinfo")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
missing = [b for b in REQUIRED_BINARIES if shutil.which(b) is None]
|
||||
if missing:
|
||||
print(f"error: missing poppler-utils binaries: {', '.join(missing)}", file=sys.stderr)
|
||||
print("install with: sudo apt install poppler-utils", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
if not EXAMPLES_DIR.is_dir():
|
||||
print(f"error: {EXAMPLES_DIR} does not exist", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
pdfs = sorted(EXAMPLES_DIR.glob("*.pdf"))
|
||||
if not pdfs:
|
||||
print(f"error: no PDFs found in {EXAMPLES_DIR}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
OUTPUT_DIR.mkdir(exist_ok=True)
|
||||
write_shared_css()
|
||||
|
||||
issues = []
|
||||
for pdf in pdfs:
|
||||
print(f"converting {pdf.name} ...")
|
||||
issues.append(convert_pdf(pdf))
|
||||
|
||||
write_root_index(issues)
|
||||
print(f"\ndone. open {OUTPUT_DIR / 'index.html'} in a browser.")
|
||||
return 0
|
||||
|
||||
|
||||
def convert_pdf(pdf_path: Path) -> dict:
|
||||
stem = pdf_path.stem
|
||||
out = OUTPUT_DIR / stem
|
||||
reset_dir(out)
|
||||
|
||||
page_count = pdf_page_count(pdf_path)
|
||||
page_texts = extract_text_per_page(pdf_path, page_count)
|
||||
|
||||
for n in range(1, page_count + 1):
|
||||
render_jpeg(pdf_path, out / f"page-{n:02d}", n, DPI_FULL, JPEG_QUALITY_FULL)
|
||||
render_jpeg(pdf_path, out / f"thumb-{n:02d}", n, DPI_THUMB, JPEG_QUALITY_THUMB)
|
||||
write_page_html(out, stem, n, page_count, page_texts[n - 1])
|
||||
|
||||
write_issue_index(out, stem, page_count)
|
||||
return {"stem": stem, "title": stem, "pages": page_count}
|
||||
|
||||
|
||||
def reset_dir(path: Path) -> None:
|
||||
if path.exists():
|
||||
shutil.rmtree(path)
|
||||
path.mkdir(parents=True)
|
||||
|
||||
|
||||
def pdf_page_count(pdf_path: Path) -> int:
|
||||
out = subprocess.run(
|
||||
["pdfinfo", str(pdf_path)], check=True, capture_output=True, text=True
|
||||
).stdout
|
||||
m = re.search(r"^Pages:\s+(\d+)", out, re.MULTILINE)
|
||||
if not m:
|
||||
raise RuntimeError(f"could not parse page count from pdfinfo for {pdf_path}")
|
||||
return int(m.group(1))
|
||||
|
||||
|
||||
def extract_text_per_page(pdf_path: Path, page_count: int) -> list[str]:
|
||||
"""Return a list of page texts. pdftotext separates pages with form-feed (\\f)."""
|
||||
out = subprocess.run(
|
||||
["pdftotext", "-layout", str(pdf_path), "-"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
).stdout
|
||||
pages = out.split("\f")
|
||||
# pdftotext appends a trailing form-feed → one extra empty element; trim it.
|
||||
if pages and pages[-1] == "":
|
||||
pages.pop()
|
||||
# Pad/truncate so the list matches page_count exactly (defensive).
|
||||
while len(pages) < page_count:
|
||||
pages.append("")
|
||||
return pages[:page_count]
|
||||
|
||||
|
||||
def render_jpeg(pdf_path: Path, out_stem: Path, page: int, dpi: int, quality: int) -> None:
|
||||
"""pdftocairo always appends -<page>.jpg unless -singlefile is used.
|
||||
|
||||
We render one page at a time with -singlefile so the output filename is exact.
|
||||
"""
|
||||
subprocess.run(
|
||||
[
|
||||
"pdftocairo",
|
||||
"-jpeg",
|
||||
"-jpegopt",
|
||||
f"quality={quality}",
|
||||
"-r",
|
||||
str(dpi),
|
||||
"-f",
|
||||
str(page),
|
||||
"-l",
|
||||
str(page),
|
||||
"-singlefile",
|
||||
str(pdf_path),
|
||||
str(out_stem), # pdftocairo appends .jpg automatically
|
||||
],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
|
||||
|
||||
# ---------- HTML emission ----------
|
||||
|
||||
|
||||
def write_shared_css() -> None:
|
||||
(OUTPUT_DIR / "styles.css").write_text(SHARED_CSS, encoding="utf-8")
|
||||
|
||||
|
||||
def write_root_index(issues: list[dict]) -> None:
|
||||
cards = "\n".join(
|
||||
f""" <a class="card" href="{html.escape(i['stem'])}/index.html">
|
||||
<img src="{html.escape(i['stem'])}/thumb-01.jpg" alt="Cover of {html.escape(i['title'])}">
|
||||
<h2>{html.escape(i['title'])}</h2>
|
||||
<p>{i['pages']} pages</p>
|
||||
</a>"""
|
||||
for i in issues
|
||||
)
|
||||
page = PAGE_SHELL.format(
|
||||
title="NETgazet archive",
|
||||
css_href="styles.css",
|
||||
body=f""" <header>
|
||||
<h1>NETgazet archive</h1>
|
||||
<p>Static HTML rendering of the available editions.</p>
|
||||
</header>
|
||||
<main class="card-grid">
|
||||
{cards}
|
||||
</main>""",
|
||||
)
|
||||
(OUTPUT_DIR / "index.html").write_text(page, encoding="utf-8")
|
||||
|
||||
|
||||
def write_issue_index(out: Path, stem: str, page_count: int) -> None:
|
||||
thumbs = "\n".join(
|
||||
f""" <a class="thumb" href="page-{n:02d}.html">
|
||||
<img src="thumb-{n:02d}.jpg" alt="Page {n}" loading="lazy">
|
||||
<span>Page {n}</span>
|
||||
</a>"""
|
||||
for n in range(1, page_count + 1)
|
||||
)
|
||||
page = PAGE_SHELL.format(
|
||||
title=html.escape(stem),
|
||||
css_href="../styles.css",
|
||||
body=f""" <header>
|
||||
<p><a href="../index.html">← all editions</a></p>
|
||||
<h1>{html.escape(stem)}</h1>
|
||||
<p>{page_count} pages — tap a thumbnail to read.</p>
|
||||
</header>
|
||||
<main class="thumb-grid">
|
||||
{thumbs}
|
||||
</main>""",
|
||||
)
|
||||
(out / "index.html").write_text(page, encoding="utf-8")
|
||||
|
||||
|
||||
def write_page_html(out: Path, stem: str, n: int, total: int, text: str) -> None:
|
||||
prev_link = (
|
||||
f'<a href="page-{n - 1:02d}.html" rel="prev">← Page {n - 1}</a>'
|
||||
if n > 1
|
||||
else '<span class="nav-disabled">← Page</span>'
|
||||
)
|
||||
next_link = (
|
||||
f'<a href="page-{n + 1:02d}.html" rel="next">Page {n + 1} →</a>'
|
||||
if n < total
|
||||
else '<span class="nav-disabled">Page →</span>'
|
||||
)
|
||||
text_block = (
|
||||
f""" <details class="page-text">
|
||||
<summary>Show page text</summary>
|
||||
<pre>{html.escape(text)}</pre>
|
||||
</details>"""
|
||||
if text.strip()
|
||||
else ""
|
||||
)
|
||||
body = f""" <header>
|
||||
<p><a href="index.html">← {html.escape(stem)}</a> · <a href="../index.html">all editions</a></p>
|
||||
<h1>Page {n} <span class="of">of {total}</span></h1>
|
||||
</header>
|
||||
<nav class="pager">
|
||||
{prev_link}
|
||||
<a href="index.html" class="nav-up">Index</a>
|
||||
{next_link}
|
||||
</nav>
|
||||
<main>
|
||||
<figure class="page">
|
||||
<img src="page-{n:02d}.jpg" alt="Page {n} of {html.escape(stem)}">
|
||||
</figure>
|
||||
{text_block}
|
||||
</main>
|
||||
<nav class="pager pager-bottom">
|
||||
{prev_link}
|
||||
<a href="index.html" class="nav-up">Index</a>
|
||||
{next_link}
|
||||
</nav>"""
|
||||
page = PAGE_SHELL.format(
|
||||
title=f"{html.escape(stem)} — page {n}",
|
||||
css_href="../styles.css",
|
||||
body=body,
|
||||
)
|
||||
(out / f"page-{n:02d}.html").write_text(page, encoding="utf-8")
|
||||
|
||||
|
||||
# ---------- Templates ----------
|
||||
|
||||
PAGE_SHELL = """<!doctype html>
|
||||
<html lang="nl">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>{title}</title>
|
||||
<link rel="stylesheet" href="{css_href}">
|
||||
</head>
|
||||
<body>
|
||||
{body}
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
SHARED_CSS = """:root {
|
||||
--fg: #1a1a1a;
|
||||
--muted: #666;
|
||||
--bg: #fafafa;
|
||||
--card: #fff;
|
||||
--border: #e2e2e2;
|
||||
--accent: #1a4480;
|
||||
}
|
||||
* { box-sizing: border-box; }
|
||||
html, body { margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif;
|
||||
color: var(--fg);
|
||||
background: var(--bg);
|
||||
line-height: 1.45;
|
||||
max-width: 1000px;
|
||||
margin: 0 auto;
|
||||
padding: 1rem;
|
||||
}
|
||||
img { max-width: 100%; height: auto; display: block; }
|
||||
a { color: var(--accent); text-decoration: none; }
|
||||
a:hover { text-decoration: underline; }
|
||||
header { margin-bottom: 1.5rem; }
|
||||
header h1 { margin: 0.25rem 0; font-size: 1.5rem; }
|
||||
header p { margin: 0.25rem 0; color: var(--muted); }
|
||||
.of { color: var(--muted); font-weight: normal; font-size: 0.85em; }
|
||||
|
||||
.card-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fill, minmax(220px, 1fr));
|
||||
gap: 1.25rem;
|
||||
}
|
||||
.card {
|
||||
display: block;
|
||||
background: var(--card);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 6px;
|
||||
padding: 0.75rem;
|
||||
color: inherit;
|
||||
}
|
||||
.card:hover { border-color: var(--accent); text-decoration: none; }
|
||||
.card img { border: 1px solid var(--border); margin-bottom: 0.5rem; }
|
||||
.card h2 { margin: 0.25rem 0; font-size: 1.05rem; }
|
||||
.card p { margin: 0; color: var(--muted); font-size: 0.9rem; }
|
||||
|
||||
.thumb-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fill, minmax(140px, 1fr));
|
||||
gap: 1rem;
|
||||
}
|
||||
.thumb {
|
||||
display: block;
|
||||
text-align: center;
|
||||
color: inherit;
|
||||
}
|
||||
.thumb img {
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 3px;
|
||||
margin: 0 auto 0.25rem;
|
||||
}
|
||||
.thumb:hover img { border-color: var(--accent); }
|
||||
.thumb span { font-size: 0.85rem; color: var(--muted); }
|
||||
|
||||
.pager {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 0.5rem;
|
||||
justify-content: space-between;
|
||||
align-items: center;
|
||||
margin: 1rem 0;
|
||||
padding: 0.5rem 0;
|
||||
border-top: 1px solid var(--border);
|
||||
border-bottom: 1px solid var(--border);
|
||||
font-size: 0.95rem;
|
||||
}
|
||||
.pager .nav-up { font-weight: 600; }
|
||||
.pager .nav-disabled { color: #bbb; }
|
||||
.pager-bottom { margin-top: 1rem; }
|
||||
|
||||
.page { margin: 0; }
|
||||
.page img {
|
||||
border: 1px solid var(--border);
|
||||
box-shadow: 0 2px 8px rgba(0,0,0,0.06);
|
||||
margin: 0 auto;
|
||||
}
|
||||
|
||||
.page-text {
|
||||
margin-top: 1rem;
|
||||
background: var(--card);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 4px;
|
||||
padding: 0.5rem 0.75rem;
|
||||
}
|
||||
.page-text summary {
|
||||
cursor: pointer;
|
||||
color: var(--accent);
|
||||
font-size: 0.95rem;
|
||||
}
|
||||
.page-text pre {
|
||||
white-space: pre-wrap;
|
||||
word-wrap: break-word;
|
||||
font-size: 0.85rem;
|
||||
line-height: 1.4;
|
||||
margin: 0.75rem 0 0;
|
||||
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
|
||||
}
|
||||
|
||||
@media (max-width: 480px) {
|
||||
body { padding: 0.5rem; }
|
||||
header h1 { font-size: 1.25rem; }
|
||||
.pager { font-size: 0.85rem; }
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user