from pathlib import Path
import zipfile, textwrap, os, json
base = Path("/mnt/data/sitemap-pdf-generator")
base.mkdir(exist_ok=True)
files = {
"app.py": r'''
import os, re, uuid, threading, zipfile
from pathlib import Path
from urllib.parse import urlparse
import requests
from bs4 import BeautifulSoup
from flask import Flask, render_template, request, jsonify, send_file
BASE = Path(__file__).parent
JOBS = BASE / "jobs"
JOBS.mkdir(exist_ok=True)
app = Flask(__name__)
jobs = {}
def safe_filename(url, index):
p = urlparse(url)
path = p.path.strip("/")
name = path.split("/")[-1] if path else "startseite"
name = re.sub(r"[^a-zA-Z0-9äöüÄÖÜß._-]+", "-", name).strip("-")
if not name:
name = f"seite-{index}"
return f"{index:04d}_{name[:100]}.pdf"
def read_sitemap(url):
r = requests.get(url, timeout=30, headers={"User-Agent":"Sitemap-PDF-Generator/1.0"})
r.raise_for_status()
soup = BeautifulSoup(r.text, "xml")
# Normal sitemap
urls = [x.get_text(strip=True) for x in soup.find_all("loc")]
return list(dict.fromkeys(urls))
def render_pdf(url, output):
from playwright.sync_api import sync_playwright
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
page.goto(url, wait_until="networkidle", timeout=60000)
page.pdf(
path=str(output),
format="A4",
print_background=True,
margin={"top":"15mm","right":"15mm","bottom":"15mm","left":"15mm"}
)
browser.close()
def worker(job_id, sitemap_url):
job = jobs[job_id]
outdir = JOBS / job_id / "PDFs"
outdir.mkdir(parents=True, exist_ok=True)
try:
urls = read_sitemap(sitemap_url)
job["total"] = len(urls)
if not urls:
raise RuntimeError("Die Sitemap enthält keine -Einträge.")
for i, url in enumerate(urls, 1):
job["current"] = url
try:
target = outdir / safe_filename(url, i)
render_pdf(url, target)
job["done"] += 1
except Exception as e:
job["errors"].append({"url": url, "error": str(e)[:500]})
zip_path = JOBS / job_id / "sitemap-pdfs.zip"
with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as z:
for f in outdir.glob("*.pdf"):
z.write(f, f"PDFs/{f.name}")
job["zip"] = str(zip_path)
job["status"] = "finished"
except Exception as e:
job["status"] = "error"
job["message"] = str(e)
@app.route("/")
def index():
return render_template("index.html")
@app.post("/api/start")
def start():
data = request.get_json(silent=True) or {}
sitemap = (data.get("sitemap") or "").strip()
if not sitemap.startswith(("http://","https://")):
return jsonify({"error":"Bitte eine vollständige Sitemap-URL eingeben."}), 400
job_id = uuid.uuid4().hex
jobs[job_id] = {
"status":"running", "total":0, "done":0, "current":"",
"errors":[], "zip":None, "message":""
}
threading.Thread(target=worker, args=(job_id, sitemap), daemon=True).start()
return jsonify({"job_id":job_id})
@app.get("/api/status/")
def status(job_id):
if job_id not in jobs:
return jsonify({"error":"Job nicht gefunden."}), 404
j = jobs[job_id].copy()
if j.get("zip"):
j["download"] = f"/download/{job_id}"
j.pop("zip", None)
return jsonify(j)
@app.get("/download/")
def download(job_id):
path = JOBS / job_id / "sitemap-pdfs.zip"
if not path.exists():
return "Noch nicht fertig.", 404
return send_file(path, as_attachment=True, download_name="sitemap-pdfs.zip")
if __name__ == "__main__":
app.run(host="0.0.0.0", port=5000, debug=False)
''',
"templates/index.html": r'''
Sitemap → PDF Generator
''',
"requirements.txt": r'''
Flask>=3.0
requests>=2.31
beautifulsoup4>=4.12
lxml>=5.0
playwright>=1.45
''',
"README.md": r'''
# Sitemap → PDF Generator
Das Tool nimmt eine XML-Sitemap entgegen, liest alle ``-URLs aus und erzeugt für jede URL eine eigene PDF-Datei.
## Installation
```bash
python -m venv .venv
Sitemap → PDF Generator
Eine Sitemap eintragen. Das Tool liest alle <loc>-URLs aus und erzeugt für jede Unterseite eine eigene PDF.
Bereit.
Kommentare
Kommentar veröffentlichen