ccc

from pathlib import Path import zipfile, textwrap, os, json base = Path("/mnt/data/sitemap-pdf-generator") base.mkdir(exist_ok=True) files = { "app.py": r''' import os, re, uuid, threading, zipfile from pathlib import Path from urllib.parse import urlparse import requests from bs4 import BeautifulSoup from flask import Flask, render_template, request, jsonify, send_file BASE = Path(__file__).parent JOBS = BASE / "jobs" JOBS.mkdir(exist_ok=True) app = Flask(__name__) jobs = {} def safe_filename(url, index): p = urlparse(url) path = p.path.strip("/") name = path.split("/")[-1] if path else "startseite" name = re.sub(r"[^a-zA-Z0-9äöüÄÖÜß._-]+", "-", name).strip("-") if not name: name = f"seite-{index}" return f"{index:04d}_{name[:100]}.pdf" def read_sitemap(url): r = requests.get(url, timeout=30, headers={"User-Agent":"Sitemap-PDF-Generator/1.0"}) r.raise_for_status() soup = BeautifulSoup(r.text, "xml") # Normal sitemap urls = [x.get_text(strip=True) for x in soup.find_all("loc")] return list(dict.fromkeys(urls)) def render_pdf(url, output): from playwright.sync_api import sync_playwright with sync_playwright() as p: browser = p.chromium.launch(headless=True) page = browser.new_page() page.goto(url, wait_until="networkidle", timeout=60000) page.pdf( path=str(output), format="A4", print_background=True, margin={"top":"15mm","right":"15mm","bottom":"15mm","left":"15mm"} ) browser.close() def worker(job_id, sitemap_url): job = jobs[job_id] outdir = JOBS / job_id / "PDFs" outdir.mkdir(parents=True, exist_ok=True) try: urls = read_sitemap(sitemap_url) job["total"] = len(urls) if not urls: raise RuntimeError("Die Sitemap enthält keine -Einträge.") for i, url in enumerate(urls, 1): job["current"] = url try: target = outdir / safe_filename(url, i) render_pdf(url, target) job["done"] += 1 except Exception as e: job["errors"].append({"url": url, "error": str(e)[:500]}) zip_path = JOBS / job_id / "sitemap-pdfs.zip" with zipfile.ZipFile(zip_path, "w", zipfile.ZIP_DEFLATED) as z: for f in outdir.glob("*.pdf"): z.write(f, f"PDFs/{f.name}") job["zip"] = str(zip_path) job["status"] = "finished" except Exception as e: job["status"] = "error" job["message"] = str(e) @app.route("/") def index(): return render_template("index.html") @app.post("/api/start") def start(): data = request.get_json(silent=True) or {} sitemap = (data.get("sitemap") or "").strip() if not sitemap.startswith(("http://","https://")): return jsonify({"error":"Bitte eine vollständige Sitemap-URL eingeben."}), 400 job_id = uuid.uuid4().hex jobs[job_id] = { "status":"running", "total":0, "done":0, "current":"", "errors":[], "zip":None, "message":"" } threading.Thread(target=worker, args=(job_id, sitemap), daemon=True).start() return jsonify({"job_id":job_id}) @app.get("/api/status/") def status(job_id): if job_id not in jobs: return jsonify({"error":"Job nicht gefunden."}), 404 j = jobs[job_id].copy() if j.get("zip"): j["download"] = f"/download/{job_id}" j.pop("zip", None) return jsonify(j) @app.get("/download/") def download(job_id): path = JOBS / job_id / "sitemap-pdfs.zip" if not path.exists(): return "Noch nicht fertig.", 404 return send_file(path, as_attachment=True, download_name="sitemap-pdfs.zip") if __name__ == "__main__": app.run(host="0.0.0.0", port=5000, debug=False) ''', "templates/index.html": r''' Sitemap → PDF Generator

Sitemap → PDF Generator

Eine Sitemap eintragen. Das Tool liest alle <loc>-URLs aus und erzeugt für jede Unterseite eine eigene PDF.

Bereit.
''', "requirements.txt": r''' Flask>=3.0 requests>=2.31 beautifulsoup4>=4.12 lxml>=5.0 playwright>=1.45 ''', "README.md": r''' # Sitemap → PDF Generator Das Tool nimmt eine XML-Sitemap entgegen, liest alle ``-URLs aus und erzeugt für jede URL eine eigene PDF-Datei. ## Installation ```bash python -m venv .venv

Kommentare