There is the limitation that you still need to manually import all modules that register a new comic format, but this does give us a lot of flexibility
142 lines
5.4 KiB
Python
142 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
|
|
# Copyright (c) 2025 Markil 3
|
|
|
|
import logging
|
|
import datetime
|
|
import threading
|
|
import textwrap
|
|
from zipfile import ZipFile
|
|
from urllib.parse import urlparse
|
|
|
|
from pathlib import Path
|
|
from tempfile import NamedTemporaryFile
|
|
|
|
import requests
|
|
from PIL import Image, ImageDraw, ImageFont
|
|
from bs4 import BeautifulSoup
|
|
|
|
from comic_download.comic_strip import Comic, ComicStrip, register_comic_class
|
|
|
|
class XKCDComic(Comic):
|
|
"""
|
|
The XKCD Comic.
|
|
"""
|
|
|
|
def __init__(self):
|
|
super().__init__("xkcd")
|
|
|
|
@classmethod
|
|
def create_from_url(cls, url):
|
|
return cls()
|
|
|
|
def _load_data(self):
|
|
self.title = "XKCD"
|
|
self.author = "Randal Munroe"
|
|
self.image_urls = {
|
|
"banner": "https://xkcd.com/s/0b7742.png"
|
|
}
|
|
self.description = "A webcomic of romance, sarcasm, math, and language."
|
|
result = requests.get("https://xkcd.com/info.0.json")
|
|
result.raise_for_status()
|
|
self.data = result.json()
|
|
self.latest_identifier = self.data["num"]
|
|
|
|
def get_first_strip(self):
|
|
return 1
|
|
|
|
def get_latest_strip(self):
|
|
return self.latest_identifier
|
|
|
|
def get_all_strips(self) -> list:
|
|
return list(range(self.get_first_strip(), self.get_latest_strip() + 1))
|
|
|
|
def __new__(cls):
|
|
if not hasattr(cls, 'instance'):
|
|
cls.instance = super(XKCDComic, cls).__new__(cls)
|
|
return cls.instance
|
|
|
|
class XKCDComicStrip(ComicStrip):
|
|
"""
|
|
A comic strip object represents a single XKCD strip.
|
|
"""
|
|
def __init__(self, comic, index: int):
|
|
"""
|
|
Creates a comic strip object.
|
|
|
|
:param comic: The XKCD comic.
|
|
:param index: The index of the strip to load.
|
|
"""
|
|
super().__init__(comic, index)
|
|
self.url = f"https://xkcd.com/{index}"
|
|
self.data_url = f"{self.url}/info.0.json"
|
|
self.transcript = None
|
|
|
|
@property
|
|
def index(self):
|
|
# Alias
|
|
return self.identifier
|
|
|
|
def _load_data(self):
|
|
logging.info("Loading data for comic %s from %s", self.index, self.url)
|
|
result = requests.get(self.data_url)
|
|
result.raise_for_status()
|
|
self.data = result.json()
|
|
self.date = datetime.date(int(self.data["year"]), int(self.data["month"]), int(self.data["day"]))
|
|
self.title = self.data["title"]
|
|
self.image_urls.append(self.data["img"])
|
|
self.captions.append(self.data["alt"])
|
|
self.transcript = self.data["transcript"]
|
|
|
|
def _transform_images(self):
|
|
"""
|
|
Takes the raw image data from #download_data and transforms it to add a title, captions, etc. This function is not idempotent.
|
|
"""
|
|
f_width = 780
|
|
title_size = 21
|
|
caption_size = 12
|
|
title_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans Bold.ttf"), title_size, layout_engine=ImageFont.Layout.RAQM)
|
|
caption_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans.ttf"), caption_size)
|
|
|
|
title_box = title_font.getbbox(self.title)
|
|
title_box = (title_box[2] - title_box[0], title_box[3] - title_box[1])
|
|
|
|
url_box = caption_font.getbbox(self.url)
|
|
url_box = (url_box[2] - url_box[0], url_box[3] - url_box[1])
|
|
date_box = caption_font.getbbox(self.date.isoformat())
|
|
date_box = (date_box[2] - date_box[0], date_box[3] - date_box[1])
|
|
|
|
self.transformed_images = []
|
|
for i in range(len(self.images)):
|
|
image = self.images[i]
|
|
image_path = Path(image.name)
|
|
f_image = NamedTemporaryFile(mode='wb', suffix=f"f-{i}.png", prefix=self.get_identifier_string(), delete = False)
|
|
if image_path.exists():
|
|
caption = self.captions[i]
|
|
try:
|
|
caption = textwrap.wrap(caption, width=(f_width - 20) / caption_size)
|
|
cap_max = max([cap_box[2] - cap_box[0] for cap_box in [caption_font.getbbox(cap_line) for cap_line in caption]])
|
|
caption_box = (cap_max, caption_size * len(caption))
|
|
caption = "\n".join(caption)
|
|
except Exception as e:
|
|
logging.exception("Unable to get caption bounding box for panel %d", i, exc_info=e)
|
|
caption_box = caption_font.getbbox(caption)
|
|
caption_box = (caption_box[2] - caption_box[0], caption_box[3] - caption_box[1])
|
|
img = Image.open(image_path)
|
|
f_img = Image.new("RGBA", (f_width, img.size[1] + caption_box[1] + title_box[1] + 40), (255, 255, 255, 255))
|
|
f_img.paste(img, ((f_width - img.size[0]) // 2, title_box[1] + 20))
|
|
|
|
draw = ImageDraw.Draw(f_img)
|
|
|
|
draw.text(((f_width - title_box[0]) / 2, 2), self.title.upper(), (0, 0, 0), font=title_font, features=["c2sc", "smcp"])
|
|
draw.multiline_text(((f_width - caption_box[0]) / 2, img.size[1] + title_box[1] + 22), caption, fill=(0, 0, 0), align="center", font=caption_font)
|
|
draw.text((10, f_img.size[1] - url_box[1] - 10), self.url, (0, 0, 0), align="left", font=caption_font)
|
|
draw.text((f_width - date_box[0] - 10, f_img.size[1] - date_box[1] - 10), self.date.isoformat(), (0, 0, 0), align="right", font=caption_font)
|
|
f_img.save(f_image)
|
|
f_image.close()
|
|
self.transformed_images.append(f_image)
|
|
|
|
|
|
XKCDComic.strip_cls = XKCDComicStrip
|
|
register_comic_class("xkcd.com", XKCDComic)
|