commit 01c8970e3a059c60f49725a1f63415ce8b1b5729 Author: Markil 3 Date: Mon Jul 28 16:18:00 2025 -0600 Initial commit for XKCD and Existential Comics diff --git a/Lucida Sans Bold.ttf b/Lucida Sans Bold.ttf new file mode 100644 index 0000000..f1209b7 Binary files /dev/null and b/Lucida Sans Bold.ttf differ diff --git a/Lucida Sans.ttf b/Lucida Sans.ttf new file mode 100644 index 0000000..607f839 Binary files /dev/null and b/Lucida Sans.ttf differ diff --git a/download.sh b/download.sh new file mode 100755 index 0000000..7377fee --- /dev/null +++ b/download.sh @@ -0,0 +1,9 @@ +#!/bin/bash + +# Copyright (c) 2025 Markil 3 + +path=$(dirname $(realpath $0)) +source "$path/venv/bin/activate" +"$path/xkcd.py" --output /mnt/MEDIA/Comics/xkcd -l +"$path/existential_comics.py" --output /mnt/MEDIA/Comics/existential_comics -l + diff --git a/existential_comics.py b/existential_comics.py new file mode 100755 index 0000000..b91ab8d --- /dev/null +++ b/existential_comics.py @@ -0,0 +1,262 @@ +#!/usr/bin/env python3 + +# Copyright (c) 2025 Markil 3 + +import logging +import logging.handlers +import argparse +import datetime +import threading +import time +import textwrap +import json +from zipfile import ZipFile +from urllib.parse import urlparse + +from pathlib import Path + +import requests +from PIL import Image, ImageDraw, ImageFont +from bs4 import BeautifulSoup + +class ComicStrip: + def __init__(self, index): + self.index = index or 0 + if index: + self.url = f"https://existentialcomics.com/comic/{index}" + else: + self.url = f"https://existentialcomics.com" + self.title = None + self.image_url = [] + self.caption = [] + + def get_metadata_path(self, base_path: Path): + return Path(base_path, f"{self.index}.txt") + + def get_image_path(self, base_path: Path, image_num: int = 1): + return Path(base_path, f"{self.index}-{image_num}.png") + + def get_package_path(self, base_path: Path): + return Path(base_path, f"{self.index}.cbz") + + def get_header_path(self, base_path: Path): + return Path(base_path, "header.jpg") + + def get_safety_path(self, base_path: Path): + return Path(base_path, "safety.png") + + def load_data(self): + logging.info("Loading data for comic %s from %s", self.index, self.url) + result = requests.get(self.url) + result.raise_for_status() + + content = BeautifulSoup(result.content) + #print(content.prettify()) + if not self.index: + index = content.find("meta", property="og:url") + if index and index["content"]: + self.url = index["content"] + self.index = int(self.url.split('/')[-1]) + logging.info("Url of %s and index of %d", self.url, self.index) + title = content.find("meta", property="og:title") + if title and title["content"]: + self.title = title["content"] + for img in content.find_all("img", class_="comicImg"): + self.image_url.append(f'http:{img["src"]}') + if img.get("title"): + self.caption.append(img["title"]) + explanation = content.find(id="explanation") + if explanation: + self.caption.append(explanation.get_text()) + philosophers = content.find(id="philosophers-comic") + if philosophers: + self.caption.append(philosophers.get_text()) + self.date = datetime.date(2013, 11, 11) + datetime.timedelta(weeks=self.index - 2) + + + def download_data(self, base_path: Path): + if self.image_url: + meta_path = self.get_metadata_path(base_path) + if not meta_path.exists(): + with open(meta_path, 'w') as meta_stream: + json.dump({ + "title": self.title, + "image_url": self.image_url, + "caption": self.caption, + }, meta_stream) + header_path = self.get_header_path(base_path) + if not header_path.exists(): + result = requests.get("https://static.existentialcomics.com/title.jpg") + result.raise_for_status() + with open(header_path, 'wb') as image_stream: + image_stream.write(result.content) + safety_path = self.get_safety_path(base_path) + if not safety_path.exists(): + result = requests.get("https://static.existentialcomics.com/safety.png") + result.raise_for_status() + with open(safety_path, 'wb') as image_stream: + image_stream.write(result.content) + for i in range(len(self.image_url)): + logging.info(" Downloading %d-%d", self.index, i + 1) + image_path = self.get_image_path(base_path, i + 1) + if not image_path.exists(): + result = requests.get(self.image_url[i]) + result.raise_for_status() + with open(image_path, 'wb') as image_stream: + image_stream.write(result.content) + + def transform_data(self, base_path: Path): + """ + Takes the raw image data from #download_data and transforms it to add a title, captions, etc. This function is not idempotent. + """ + title_size = 24 + caption_size = 15 + f_size = (1000, 1500) + title_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans Bold.ttf"), title_size) + caption_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans.ttf"), caption_size) + title_box = title_font.getbbox(self.title) + title_box = (title_box[2] - title_box[0], title_box[3] - title_box[1]) + + if self.date >= datetime.date(2023, 11, 13): + safe = (self.date - datetime.date(2023, 11, 12)).days + else: + safe = (self.date - datetime.date(2013, 11, 11)).days + safe = str(safe) + safe_box = title_font.getbbox(safe) + safe_box = (safe_box[2] - safe_box[0], safe_box[3] - safe_box[1]) + url_box = caption_font.getbbox(self.url) + url_box = (url_box[2] - url_box[0], url_box[3] - url_box[1]) + date_box = caption_font.getbbox(self.date.isoformat()) + date_box = (date_box[2] - date_box[0], date_box[3] - date_box[1]) + + caption = "\n".join(self.caption).split("\n") + try: + wrapped_caption = [] + for line in caption: + if line: + wrapped_caption.extend(textwrap.wrap(line.strip(), width=(f_size[0] - 20) / caption_size)) + else: + wrapped_caption.append(line.strip()) + caption = wrapped_caption + cap_max = max([cap_box[2] - cap_box[0] for cap_box in [caption_font.getbbox(cap_line) for cap_line in caption]]) + caption_box = (cap_max, caption_size * len(caption)) + caption = "\n".join(caption) + except Exception as e: + logging.exception("Unable to get caption bounding box", exc_info=e) + caption_box = caption_font.getbbox(caption) + caption_box = (caption_box[2] - caption_box[0], caption_box[3] - caption_box[1]) + + header_path = self.get_header_path(base_path) + header = Image.open(header_path) + f_img = Image.new("RGBA", f_size, (255, 255, 255, 255)) + f_img.paste(header, (0, 20)) + + safety_path = self.get_safety_path(base_path) + safety = Image.open(safety_path) + f_img.paste(safety, (20, 230)) + + draw = ImageDraw.Draw(f_img) + draw.text(((f_size[0] - title_box[0]) // 2, 250), self.title, (0, 0, 0), font=title_font) + draw.text((70 + safe_box[0] // 2, 240 + safe_box[1] // 2), safe, (0, 0, 0), font=title_font) + draw.text(((f_size[0] - url_box[0]) // 2, 300), self.url, (0, 0, 0), align="center", font=caption_font) + draw.text((f_size[0] - date_box[0] - 10, 250), self.date.isoformat(), (0, 0, 0), align="right", font=caption_font) + f_img.save(self.get_image_path(base_path, 0)) + + l_img = Image.new("RGBA", f_size, (255, 255, 255, 255)) + draw = ImageDraw.Draw(l_img) + draw.rounded_rectangle(((f_size[0] - caption_box[0]) // 2 - 10, 10, (f_size[0] + caption_box[0]) // 2 + 10, 100 + caption_box[1]), 20, (200, 200, 200), (0, 0, 0), 2) + draw.multiline_text(((f_size[0] - caption_box[0]) / 2, 20), caption, fill=(0, 0, 0), align="center", font=caption_font) + l_img.save(self.get_image_path(base_path, len(self.image_url) + 1)) + + def package_data(self, base_path: Path): + """ + Compresses all the data into a cbz file and removes the source files + """ + with ZipFile(self.get_package_path(base_path), 'w') as comic_zip: + img_url = urlparse(self.image_url[0]) + img_name = Path(img_url.path.split('/')[-1]) + image_path = self.get_image_path(base_path, 0) + for i in range(len(self.image_url) + 2): + image_path = self.get_image_path(base_path, i) + comic_zip.write(image_path, img_name.stem + "-" + str(i) + img_name.suffix) + if self.get_metadata_path(base_path).exists(): + comic_zip.write(self.get_metadata_path(base_path), "info.0.json") + for i in range(len(self.image_url) + 2): + self.get_image_path(base_path, i).unlink() + self.get_metadata_path(base_path).unlink(missing_ok=True) + + +def setup_args(): + parser = argparse.ArgumentParser( + prog='existential_comics', + description='Downloads the entire Existential Comics collection') + + parser.add_argument('-o', '--output', type=Path, default=Path(), help="The directory to dump the files to.") + parser.add_argument('-w', '--wait', type=int, default=0, help="How many seconds to wait between each request") + parser.add_argument('-t', '--threads', type=int, default=10, help="How many download threads will run at once") + parser.add_argument('-s', '--start', type=int, default=0, help="The comic index to start at. A zero will be interpreted as using up to the first comic.") + parser.add_argument('-e', '--end', type=int, default=0, help="The comic index to end at. A zero will be interpreted as using up to the last comic.") + parser.add_argument('-l', '--latest', action='store_true', help="If set, only the latest comic will be downloaded") + parser.add_argument('-p', '--plain', action='store_true', help="If set, only the raw image will be downloaded, and titles, caption, etc. will not be added.") + + return parser + +def setup_logging(): + logger = logging.getLogger() + logger.setLevel(logging.DEBUG) + + ch = logging.StreamHandler() + ch.setLevel(logging.INFO) + formatter = logging.Formatter('%(levelname)s - %(message)s') + ch.setFormatter(formatter) + logger.addHandler(ch) + + ch = logging.handlers.RotatingFileHandler("gutenberg_download.log", encoding='utf-8') + ch.setLevel(logging.DEBUG) + formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') + ch.setFormatter(formatter) + ch.doRollover() + logger.addHandler(ch) + +def load_strip(base_path, index, plain, thread_limit): + + with thread_limit: + strip = ComicStrip(index) + if not strip.get_image_path(base_path).exists() and not strip.get_package_path(base_path).exists(): + strip.load_data() + strip.download_data(base_path) + if not plain: + strip.transform_data(base_path) + strip.package_data(base_path) + +if __name__ == "__main__": + setup_logging() + parser = setup_args() + + args = parser.parse_args() + base_path = args.output + + if not base_path.is_dir(): + base_path.mkdir(parents=True) + + logging.info("Beginning parsing") + latest = ComicStrip("") + latest.load_data() + logging.info("There are %d comics", latest.index) + if args.latest: + r = range(latest.index, latest.index + 1) + else: + start = args.start or 1 + end = args.end or latest.index + r = range(end, start - 1, -1) + + threads = [] + thread_limit = threading.BoundedSemaphore(value=args.threads) + + for index in r: + t = threading.Thread(name=str(index), target=load_strip, args=(base_path, index, args.plain, thread_limit)) + threads.append(t) + t.start() + + for t in threads: + t.join() diff --git a/requirements.txt b/requirements.txt new file mode 100755 index 0000000..3f00347 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +requests +beautifulsoup4 +lxml +pillow +libraqm diff --git a/xkcd.py b/xkcd.py new file mode 100755 index 0000000..fe6e227 --- /dev/null +++ b/xkcd.py @@ -0,0 +1,207 @@ +#!/usr/bin/env python3 + +# Copyright (c) 2025 Markil 3 + +import logging +import logging.handlers +import argparse +import datetime +import threading +import time +import textwrap +from zipfile import ZipFile +from urllib.parse import urlparse + +from pathlib import Path + +import requests +from PIL import Image, ImageDraw, ImageFont +from bs4 import BeautifulSoup + +class ComicStrip: + """ + A comic strip object represents a single XKCD strip. + """ + def __init__(self, index): + """ + Creates a comic strop object. + """ + self.index = index + self.url = f"https://xkcd.com/{index}" + self.data_url = f"{self.url}/info.0.json" + self.title = None + self.image_url = None + self.caption = None + self.transcript = None + + def get_metadata_path(self, base_path: Path): + return Path(base_path, f"{self.index}.json") + + def get_image_path(self, base_path: Path): + return Path(base_path, f"{self.index}.png") + + def get_package_path(self, base_path: Path): + return Path(base_path, f"{self.index}.cbz") + + def load_data(self): + logging.info("Loading data for comic %s from %s", self.index, self.url) + result = requests.get(self.data_url) + result.raise_for_status() + self.data = result.json() + self.date = datetime.date(int(self.data["year"]), int(self.data["month"]), int(self.data["day"])) + self.image_url = self.data["img"] + self.title = self.data["title"] + self.caption = self.data["alt"] + self.transcript = self.data["transcript"] + + + def download_data(self, base_path: Path): + """ + Downloads the raw image data. + """ + if self.image_url: + image_path = self.get_image_path(base_path) + if not image_path.exists(): + with open(image_path, 'wb') as image_stream: + result = requests.get(self.image_url) + result.raise_for_status() + image_stream.write(result.content) + metadata_path = self.get_metadata_path(base_path) + if not metadata_path.exists(): + with open(metadata_path, 'wb') as meta_stream: + result = requests.get(self.data_url) + result.raise_for_status() + meta_stream.write(result.content) + + def transform_data(self, base_path: Path): + """ + Takes the raw image data from #download_data and transforms it to add a title, captions, etc. This function is not idempotent. + """ + image_path = self.get_image_path(base_path) + if image_path.exists(): + f_width = 780 + title_size = 21 + caption_size = 12 + title_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans Bold.ttf"), title_size, layout_engine=ImageFont.Layout.RAQM) + title_box = title_font.getbbox(self.title) + title_box = (title_box[2] - title_box[0], title_box[3] - title_box[1]) + + caption_font = ImageFont.truetype(Path(Path(__file__).parent, "Lucida Sans.ttf"), caption_size) + caption = self.caption + try: + caption = textwrap.wrap(caption, width=(f_width - 20) / caption_size) + cap_max = max([cap_box[2] - cap_box[0] for cap_box in [caption_font.getbbox(cap_line) for cap_line in caption]]) + caption_box = (cap_max, caption_size * len(caption)) + caption = "\n".join(caption) + except Exception as e: + logging.exception("Unable to get caption bounding box", exc_info=e) + caption_box = caption_font.getbbox(caption) + caption_box = (caption_box[2] - caption_box[0], caption_box[3] - caption_box[1]) + url_box = caption_font.getbbox(self.url) + url_box = (url_box[2] - url_box[0], url_box[3] - url_box[1]) + date_box = caption_font.getbbox(self.date.isoformat()) + date_box = (date_box[2] - date_box[0], date_box[3] - date_box[1]) + img = Image.open(image_path) + f_img = Image.new("RGBA", (f_width, img.size[1] + caption_box[1] + title_box[1] + 40), (255, 255, 255, 255)) + f_img.paste(img, ((f_width - img.size[0]) // 2, title_box[1] + 20)) + + draw = ImageDraw.Draw(f_img) + + draw.text(((f_width - title_box[0]) / 2, 2), self.title.upper(), (0, 0, 0), font=title_font, features=["c2sc", "smcp"]) + draw.multiline_text(((f_width - caption_box[0]) / 2, img.size[1] + title_box[1] + 22), caption, fill=(0, 0, 0), align="center", font=caption_font) + draw.text((10, f_img.size[1] - url_box[1] - 10), self.url, (0, 0, 0), align="left", font=caption_font) + draw.text((f_width - date_box[0] - 10, f_img.size[1] - date_box[1] - 10), self.date.isoformat(), (0, 0, 0), align="right", font=caption_font) + f_img.save(image_path) + + def package_data(self, base_path: Path): + """ + Compresses all the data into a cbz file and removes the source files + """ + with ZipFile(self.get_package_path(base_path), 'w') as comic_zip: + img_url = urlparse(self.image_url) + comic_zip.write(self.get_image_path(base_path), img_url.path.split('/')[-1]) + if self.get_metadata_path(base_path).exists(): + comic_zip.write(self.get_metadata_path(base_path), "info.0.json") + self.get_image_path(base_path).unlink() + self.get_metadata_path(base_path).unlink(missing_ok=True) + + +def setup_args(): + parser = argparse.ArgumentParser( + prog='xkcd', + description='Downloads the entire XKCD collection') + + parser.add_argument('-o', '--output', type=Path, default=Path(), help="The directory to dump the files to.") + parser.add_argument('-w', '--wait', type=int, default=0, help="How many seconds to wait between each request") + parser.add_argument('-t', '--threads', type=int, default=10, help="How many download threads will run at once") + parser.add_argument('-s', '--start', type=int, default=0, help="The comic index to start at. A zero will be interpreted as using up to the first comic.") + parser.add_argument('-e', '--end', type=int, default=0, help="The comic index to end at. A zero will be interpreted as using up to the last comic.") + parser.add_argument('-l', '--latest', action='store_true', help="If set, only the latest comic will be downloaded") + parser.add_argument('-p', '--plain', action='store_true', help="If set, only the raw image will be downloaded, and titles, caption, etc. will not be added.") + + return parser + +def setup_logging(): + logger = logging.getLogger() + logger.setLevel(logging.DEBUG) + + ch = logging.StreamHandler() + ch.setLevel(logging.INFO) + formatter = logging.Formatter('%(levelname)s - %(message)s') + ch.setFormatter(formatter) + logger.addHandler(ch) + + ch = logging.handlers.RotatingFileHandler("gutenberg_download.log", encoding='utf-8') + ch.setLevel(logging.DEBUG) + formatter = logging.Formatter('%(asctime)s - %(name)s - %(levelname)s - %(message)s') + ch.setFormatter(formatter) + ch.doRollover() + logger.addHandler(ch) + +def load_strip(base_path, index, plain, thread_limit): + """ + Threading function for downloading XKCD strips. + + Arguments: + + """ + with thread_limit: + strip = ComicStrip(index) + if not strip.get_image_path(base_path).exists() and not strip.get_package_path(base_path).exists(): + strip.load_data() + strip.download_data(base_path) + if not plain: + strip.transform_data(base_path) + strip.package_data(base_path) + +if __name__ == "__main__": + setup_logging() + parser = setup_args() + + args = parser.parse_args() + base_path = args.output + + if not base_path.is_dir(): + base_path.mkdir(parents=True) + + logging.info("Beginning parsing") + latest = ComicStrip("") + latest.load_data() + logging.info("There are %d comics", latest.data["num"]) + if args.latest: + r = range(latest.data["num"], latest.data["num"] + 1) + else: + start = args.start or 1 + end = args.end or latest.data["num"] + r = range(end, start - 1, -1) + + threads = [] + thread_limit = threading.BoundedSemaphore(value=args.threads) + + for index in r: + t = threading.Thread(name=str(index), target=load_strip, args=(base_path, index, args.plain, thread_limit)) + threads.append(t) + t.start() + + for t in threads: + t.join()