Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01LzKpYaEsuWDevuNnvSjcFW
98 lines
3.0 KiB
Python
98 lines
3.0 KiB
Python
"""Pretraga i preuzimanje članaka sa engleske Wikipedije."""
|
|
import logging
|
|
from dataclasses import dataclass
|
|
|
|
import requests
|
|
|
|
log = logging.getLogger("idriz.wiki")
|
|
|
|
API = "https://en.wikipedia.org/w/api.php"
|
|
HEADERS = {"User-Agent": "Idriz/0.1 (kucni asistent za djecu; kontakt: idriz@uka.life)"}
|
|
MAX_CHARS = 16000
|
|
|
|
|
|
@dataclass
|
|
class Article:
|
|
title: str
|
|
url: str
|
|
text: str
|
|
|
|
|
|
def search(topic: str, limit: int = 5) -> list[str]:
|
|
r = requests.get(
|
|
API,
|
|
params={"action": "query", "list": "search", "srsearch": topic, "srlimit": limit, "format": "json"},
|
|
headers=HEADERS,
|
|
timeout=20,
|
|
)
|
|
r.raise_for_status()
|
|
return [hit["title"] for hit in r.json().get("query", {}).get("search", [])]
|
|
|
|
|
|
def fetch(title: str) -> Article | None:
|
|
r = requests.get(
|
|
API,
|
|
params={
|
|
"action": "query",
|
|
"prop": "extracts|info",
|
|
"explaintext": 1,
|
|
"redirects": 1,
|
|
"inprop": "url",
|
|
"titles": title,
|
|
"format": "json",
|
|
},
|
|
headers=HEADERS,
|
|
timeout=30,
|
|
)
|
|
r.raise_for_status()
|
|
pages = r.json().get("query", {}).get("pages", {})
|
|
for page in pages.values():
|
|
if "missing" in page or not page.get("extract"):
|
|
continue
|
|
text = page["extract"]
|
|
# skrati na najzanimljiviji dio i izbaci "See also"/"References" repove
|
|
for marker in ("\n\n== See also ==", "\n\n== References ==", "\n\n== Notes ==", "\n\n== External links =="):
|
|
idx = text.find(marker)
|
|
if idx > 0:
|
|
text = text[:idx]
|
|
return Article(title=page["title"], url=page.get("fullurl", ""), text=text[:MAX_CHARS])
|
|
return None
|
|
|
|
|
|
def find_article(topic: str) -> Article | None:
|
|
titles = search(topic)
|
|
log.info("wikipedia pretraga %r -> %s", topic, titles)
|
|
for t in titles:
|
|
art = fetch(t)
|
|
if art and len(art.text) > 400:
|
|
return art
|
|
return None
|
|
|
|
|
|
def page_image_url(title: str, width: int = 1400) -> str | None:
|
|
"""Glavna slika članka (umanjena na `width` px), ako postoji."""
|
|
r = requests.get(
|
|
API,
|
|
params={"action": "query", "prop": "pageimages", "piprop": "thumbnail|original", "pithumbsize": width,
|
|
"pilicense": "any", "redirects": 1, "titles": title, "format": "json"},
|
|
headers=HEADERS,
|
|
timeout=20,
|
|
)
|
|
r.raise_for_status()
|
|
for page in r.json().get("query", {}).get("pages", {}).values():
|
|
for key in ("thumbnail", "original"):
|
|
src = (page.get(key) or {}).get("source", "")
|
|
clean = src.split("?", 1)[0]
|
|
if clean.lower().endswith((".jpg", ".jpeg", ".png", ".gif", ".webp")):
|
|
return clean
|
|
return None
|
|
|
|
|
|
def download(url: str, path) -> bool:
|
|
r = requests.get(url, headers=HEADERS, timeout=60)
|
|
if r.status_code != 200 or len(r.content) < 2000:
|
|
log.warning("slika nije preuzeta (%s): %s", r.status_code, url)
|
|
return False
|
|
path.write_bytes(r.content)
|
|
return True
|