Summary layout for kids: Wikipedia lead image, airy text, facts box, generated illustration fallback

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LzKpYaEsuWDevuNnvSjcFW
This commit is contained in:
idriz
2026-09-02 18:47:23 +02:00
parent 78967e15f9
commit 881ad46c11
6 changed files with 172 additions and 43 deletions

View File

@@ -5,6 +5,7 @@ import datetime as dt
import json
import logging
import re
from pathlib import Path
from openai import AsyncOpenAI
@@ -85,12 +86,35 @@ class Assistant:
article = await asyncio.to_thread(wiki.find_article, topic)
if not article:
return f"Nisam našao članak o temi '{topic_bs}' na Wikipediji."
title, body = await textgen.summarize(self.client, self.cfg.text_model, article, topic_bs, age)
stamp = dt.datetime.now().strftime("%Y%m%d-%H%M%S")
out = self.cfg.out_dir / f"{stamp}-sazetak-{_slug(article.title)}.pdf"
await asyncio.to_thread(render.render_summary, title, body, article.title, out, self.cfg.max_pages)
base = self.cfg.out_dir / f"{stamp}-sazetak-{_slug(article.title)}"
# sažetak i slika sa Wikipedije idu paralelno
summary_task = asyncio.create_task(textgen.summarize(self.client, self.cfg.text_model, article, topic_bs, age))
image_task = asyncio.create_task(asyncio.to_thread(self._wiki_image, article.title, base))
summary = await summary_task
image = await image_task
if image is None:
try:
raw = await coloring.generate_illustration(self.client, self.cfg.image_model, summary.image_prompt,
base.with_suffix(".gen.png"))
image = render.prepare_image(raw, base.with_suffix(".img.png"))
except Exception: # noqa: BLE001
log.exception("ilustracija nije uspjela, štampam bez slike")
out = base.with_suffix(".pdf")
await asyncio.to_thread(render.render_summary, summary.title, summary.html, summary.facts, summary.caption,
image, article.title, out, self.cfg.max_pages)
await printer.print_file(out, self.cfg.printer, self.cfg.dry_run, fit=False)
return f"Sažetak '{title}' (po članku '{article.title}') je odštampan."
return f"Sažetak '{summary.title}' (po članku '{article.title}') je odštampan."
@staticmethod
def _wiki_image(title: str, base):
url = wiki.page_image_url(title)
if not url:
return None
raw = base.with_suffix(".wiki" + Path(url).suffix.lower())
if not wiki.download(url, raw):
return None
return render.prepare_image(raw, base.with_suffix(".img.png"))
async def tool_bojanka(self, args: dict) -> str:
subject = args.get("opis") or args.get("opis_bosanski") or "a cute animal"

View File

@@ -35,3 +35,22 @@ async def generate(client: AsyncOpenAI, model: str, subject: str, out: Path) ->
page.save(out.with_suffix(".png"))
log.info("bojanka spremljena: %s", out)
return out
ILLUSTRATION_PROMPT = (
"A simple, friendly black and white line drawing for a children's encyclopedia page: {subject}. "
"Clean black outlines on white, light hatching allowed, no text, no watermark, landscape orientation."
)
async def generate_illustration(client: AsyncOpenAI, model: str, subject: str, out: Path) -> Path:
"""Crtež za sažetak kad Wikipedija nema sliku (PNG, vodoravno)."""
log.info("generišem ilustraciju: %r", subject)
resp = await client.images.generate(model=model, prompt=ILLUSTRATION_PROMPT.format(subject=subject),
size="1536x1024", quality="medium", n=1)
b64 = resp.data[0].b64_json
if not b64:
raise RuntimeError("model nije vratio sliku")
img = Image.open(io.BytesIO(base64.b64decode(b64))).convert("L")
ImageOps.autocontrast(img).save(out, "PNG")
return out

View File

@@ -1,55 +1,85 @@
"""Pravljenje PDF-a od HTML sažetka (WeasyPrint) uz ograničenje broja stranica."""
"""Pravljenje PDF-a od sažetka (WeasyPrint): slika, prozračan tekst, okvir sa zanimljivostima, najviše 2 stranice."""
import datetime as dt
import html
import logging
from pathlib import Path
from PIL import Image, ImageOps
from weasyprint import HTML
log = logging.getLogger("idriz.render")
CSS = """
@page {{ size: A4; margin: 16mm 18mm 18mm 18mm;
@page {{ size: A4; margin: 15mm 17mm 17mm 17mm;
@bottom-center {{ content: "Idriz • Izvor: engleska Wikipedija ({source}) • {date} • strana " counter(page) " / " counter(pages);
font-family: "DejaVu Sans", sans-serif; font-size: 8pt; color: #666; }} }}
body {{ font-family: "DejaVu Sans", "Noto Sans", sans-serif; font-size: {size}pt; line-height: 1.38; color: #111; }}
h1 {{ font-size: {h1}pt; margin: 0 0 6pt 0; border-bottom: 2pt solid #333; padding-bottom: 4pt; }}
h2 {{ font-size: {h2}pt; margin: 10pt 0 3pt 0; }}
p {{ margin: 0 0 6pt 0; text-align: justify; }}
ul {{ margin: 0 0 6pt 0; padding-left: 16pt; }}
li {{ margin-bottom: 2pt; }}
.meta {{ font-size: {meta}pt; color: #555; margin-bottom: 8pt; }}
font-family: "DejaVu Sans", sans-serif; font-size: 8pt; color: #777; }} }}
body {{ font-family: "DejaVu Sans", "Noto Sans", sans-serif; font-size: {size}pt; line-height: 1.5; color: #111; }}
h1 {{ font-size: {h1}pt; line-height: 1.15; margin: 0 0 4pt 0; }}
.meta {{ font-size: {meta}pt; color: #666; margin: 0 0 8pt 0; }}
h2 {{ font-size: {h2}pt; margin: 12pt 0 3pt 0; color: #222; }}
p {{ margin: 0 0 7pt 0; }}
figure {{ margin: 4pt 0 10pt 0; text-align: center; page-break-inside: avoid; }}
figure img {{ max-width: 100%; max-height: {imgh}mm; border: 1.5pt solid #333; }}
figcaption {{ font-size: {meta}pt; color: #555; margin-top: 3pt; font-style: italic; }}
.facts {{ border: 2pt dashed #444; border-radius: 8pt; padding: 8pt 12pt; margin: 12pt 0 0 0; page-break-inside: avoid; }}
.facts h2 {{ margin: 0 0 4pt 0; }}
.facts ul {{ margin: 0; padding-left: 16pt; }}
.facts li {{ margin-bottom: 3pt; }}
"""
def _document(title: str, body_html: str, source: str, size: float) -> str:
def prepare_image(src: Path, dst: Path, max_w: int = 1400) -> Path | None:
"""Pretvori sliku u sivu, pojačaj kontrast (mono laser) i smanji."""
try:
img = Image.open(src)
img.seek(0)
img = ImageOps.exif_transpose(img).convert("L")
img = ImageOps.autocontrast(img, cutoff=1)
if img.width > max_w:
img = img.resize((max_w, int(img.height * max_w / img.width)))
img.save(dst, "PNG")
return dst
except Exception: # noqa: BLE001
log.exception("slika %s se ne može obraditi", src)
return None
def _document(title: str, body_html: str, facts: list[str], caption: str, image: Path | None,
source: str, size: float, imgh: int) -> str:
date = dt.date.today().strftime("%d.%m.%Y.")
css = CSS.format(source=html.escape(source), date=date, size=size, h1=size * 1.8, h2=size * 1.25, meta=size * 0.85)
css = CSS.format(source=html.escape(source), date=date, size=size, h1=size * 2.1, h2=size * 1.3,
meta=size * 0.85, imgh=imgh)
fig = ""
if image:
fig = f"<figure><img src='{image.resolve().as_uri()}'><figcaption>{html.escape(caption)}</figcaption></figure>"
facts_html = ""
if facts:
items = "".join(f"<li>{html.escape(f)}</li>" for f in facts)
facts_html = f"<div class='facts'><h2>Da li si znao?</h2><ul>{items}</ul></div>"
return (
f"<html><head><meta charset='utf-8'><style>{css}</style></head><body>"
f"<h1>{html.escape(title)}</h1>"
f"<div class='meta'>Sažetak za tebe pripremio Idriz</div>"
f"{body_html}</body></html>"
f"{fig}{body_html}{facts_html}</body></html>"
)
def render_summary(title: str, body_html: str, source: str, out: Path, max_pages: int = 2) -> Path:
"""Renderuje PDF; ako ne stane na max_pages, smanjuje font, a na kraju skraćuje tekst."""
def render_summary(title: str, body_html: str, facts: list[str], caption: str, image: Path | None,
source: str, out: Path, max_pages: int = 2) -> Path:
"""Renderuje PDF; ako ne stane, prvo smanjuje sliku i font, na kraju skraćuje tekst."""
body = body_html
for attempt in range(8):
for size in (12, 11.5, 11, 10.5, 10):
doc = HTML(string=_document(title, body, source, size)).render()
for size, imgh in ((13, 110), (12.5, 100), (12, 90), (11.5, 80), (11, 70)):
doc = HTML(string=_document(title, body, facts, caption, image, source, size, imgh)).render()
pages = len(doc.pages)
if pages <= max_pages:
doc.write_pdf(out)
log.info("PDF %s: %d stranica, font %spt", out.name, pages, size)
return out
# skrati: izbaci zadnji paragraf/stavku
cut = max(body.rfind("<p>"), body.rfind("<li>"))
cut = body.rfind("<p>")
if cut <= 0:
break
body = body[:cut]
log.info("skraćujem tekst (pokušaj %d)", attempt + 1)
doc = HTML(string=_document(title, body, source, 10)).render()
doc.write_pdf(out)
HTML(string=_document(title, body, facts, caption, image, source, 11, 70)).write_pdf(out)
return out

View File

@@ -1,6 +1,7 @@
"""Prevod i sažimanje članka na bosanski, prilagođeno djetetu."""
import json
import logging
from dataclasses import dataclass
from openai import AsyncOpenAI
@@ -9,19 +10,32 @@ from .wiki import Article
log = logging.getLogger("idriz.textgen")
SYSTEM = """Ti si Idriz, pomoćnik za djecu u Bosni i Hercegovini. Dobit ćeš članak sa engleske Wikipedije.
Napiši sažetak na BOSANSKOM jeziku (ijekavica, latinica) koji dijete može razumjeti.
Napiši kratak, zanimljiv sažetak na BOSANSKOM jeziku (ijekavica, latinica) koji dijete rado čita.
Pravila:
- Prilagodi jezik uzrastu djeteta koji ti je dat. Za mlađe: kratke rečenice, jednostavne riječi. Za starije: više činjenica.
- Dužina: između 450 i 650 riječi. Nikako više, jer mora stati na dvije stranice A4.
- Struktura: naslov, uvod (2-3 rečenice), zatim 3 do 5 kratkih poglavlja sa podnaslovima, i na kraju "Zanimljivosti" sa 3 crtice.
- Prilagodi jezik uzrastu djeteta. Za mlađe: vrlo kratke rečenice, jednostavne riječi, obraćaj se djetetu sa "ti". Za starije: više činjenica, ali i dalje živo.
- Dužina: između 280 i 420 riječi. Nikako više. Ovo ide na papir sa slikom, mora biti prozračno.
- Struktura: naslov (kratak, može i sa uzvikom), uvod od 2-3 rečenice koji odmah privuče pažnju, zatim 3 kratka poglavlja sa podnaslovima u obliku pitanja ("Kako nastaje?", "Gdje ih ima?"), po 2-4 rečenice, i na kraju "Da li si znao?" sa tačno 3 kratke crtice.
- Piši tačno, prema članku. Ne izmišljaj podatke. Brojeve i godine zadrži.
- Bez engleskih riječi osim imena koja se ne prevode.
Odgovori ISKLJUČIVO JSON objektom oblika:
{"naslov": "...", "html": "<p>uvod</p><h2>Podnaslov</h2><p>...</p>...<h2>Zanimljivosti</h2><ul><li>...</li></ul>"}
U "html" koristi samo oznake p, h2, ul, li, strong."""
{"naslov": "...",
"html": "<p>uvod</p><h2>Pitanje?</h2><p>...</p>...",
"zanimljivosti": ["...", "...", "..."],
"slika_opis": "kratak opis slike na bosanskom, jedna rečenica (potpis ispod slike)",
"slika_prompt": "simple English description of one illustration for this topic, for a children's drawing"}
U "html" koristi samo oznake p, h2, strong. Ne stavljaj zanimljivosti u html, one idu posebno."""
async def summarize(client: AsyncOpenAI, model: str, article: Article, topic: str, age: str) -> tuple[str, str]:
@dataclass
class Summary:
title: str
html: str
facts: list[str]
caption: str
image_prompt: str
async def summarize(client: AsyncOpenAI, model: str, article: Article, topic: str, age: str) -> Summary:
user = (
f"Tema koju je dijete tražilo: {topic}\n"
f"Uzrast djeteta: {age or 'nepoznat, pretpostavi 10 godina'}\n\n"
@@ -32,9 +46,13 @@ async def summarize(client: AsyncOpenAI, model: str, article: Article, topic: st
messages=[{"role": "system", "content": SYSTEM}, {"role": "user", "content": user}],
response_format={"type": "json_object"},
)
content = resp.choices[0].message.content or "{}"
data = json.loads(content)
title = data.get("naslov") or article.title
html = data.get("html") or "<p>Nažalost, nisam uspio napisati sažetak.</p>"
log.info("sažetak %r: %d znakova", title, len(html))
return title, html
data = json.loads(resp.choices[0].message.content or "{}")
s = Summary(
title=data.get("naslov") or article.title,
html=data.get("html") or "<p>Nažalost, nisam uspio napisati sažetak.</p>",
facts=[str(f) for f in data.get("zanimljivosti", [])][:3],
caption=data.get("slika_opis") or "",
image_prompt=data.get("slika_prompt") or article.title,
)
log.info("sažetak %r: %d znakova, %d zanimljivosti", s.title, len(s.html), len(s.facts))
return s

View File

@@ -67,3 +67,31 @@ def find_article(topic: str) -> Article | None:
if art and len(art.text) > 400:
return art
return None
def page_image_url(title: str, width: int = 1400) -> str | None:
"""Glavna slika članka (umanjena na `width` px), ako postoji."""
r = requests.get(
API,
params={"action": "query", "prop": "pageimages", "piprop": "thumbnail|original", "pithumbsize": width,
"pilicense": "any", "redirects": 1, "titles": title, "format": "json"},
headers=HEADERS,
timeout=20,
)
r.raise_for_status()
for page in r.json().get("query", {}).get("pages", {}).values():
for key in ("thumbnail", "original"):
src = (page.get(key) or {}).get("source", "")
clean = src.split("?", 1)[0]
if clean.lower().endswith((".jpg", ".jpeg", ".png", ".gif", ".webp")):
return clean
return None
def download(url: str, path) -> bool:
r = requests.get(url, headers=HEADERS, timeout=60)
if r.status_code != 200 or len(r.content) < 2000:
log.warning("slika nije preuzeta (%s): %s", r.status_code, url)
return False
path.write_bytes(r.content)
return True

View File

@@ -1,4 +1,4 @@
"""Ručni test: Wikipedia -> (opciono) sažetak -> PDF. Bez ključa radi samo sa sirovim tekstom članka."""
"""Ručni test: Wikipedia -> (opciono) sažetak -> PDF. Bez ključa radi sa sirovim tekstom članka i slikom sa Wikipedije."""
import asyncio
import html
import sys
@@ -16,13 +16,23 @@ async def main(topic: str, age: str) -> None:
print("nema članka"); return
print("članak:", art.title, art.url, len(art.text), "znakova")
out = Path("out"); out.mkdir(exist_ok=True)
url = wiki.page_image_url(art.title)
image = None
if url:
raw = out / ("test-wiki" + Path(url).suffix.lower())
if wiki.download(url, raw):
image = render.prepare_image(raw, out / "test-img.png")
print("slika:", url, "->", image)
if cfg.openai_api_key:
from openai import AsyncOpenAI
title, body = await textgen.summarize(AsyncOpenAI(api_key=cfg.openai_api_key), cfg.text_model, art, topic, age)
s = await textgen.summarize(AsyncOpenAI(api_key=cfg.openai_api_key), cfg.text_model, art, topic, age)
title, body, facts, caption = s.title, s.html, s.facts, s.caption
else:
title = art.title
body = "".join(f"<p>{html.escape(p)}</p>" for p in art.text.split("\n\n")[:12])
pdf = render.render_summary(title, body, art.title, out / "test-sazetak.pdf", cfg.max_pages)
paras = [p for p in art.text.split("\n\n") if p and not p.startswith("==")][:5]
title, body = art.title, "".join(f"<p>{html.escape(p[:600])}</p>" for p in paras)
facts = ["Ovo je probna zanimljivost broj jedan.", "Druga zanimljivost, kratka.", "Treća, još kraća."]
caption = "Probni potpis ispod slike."
pdf = render.render_summary(title, body, facts, caption, image, art.title, out / "test-sazetak.pdf", cfg.max_pages)
print("PDF:", pdf)