Files
ComicDownload/comic_download/comic_strip.py
Markil 3 10cd0a9547 Adds a strip index file
This should reduce the amount of time we spend on querying for comics that are already present
2025-08-02 14:43:00 -06:00

678 lines
25 KiB
Python

#!/usr/bin/env python3
# Copyright (c) 2025 Markil 3
import logging
import logging.handlers
import argparse
from abc import ABC, abstractmethod
from datetime import datetime
from threading import Lock, Condition
import bisect
import time
import re
import textwrap
import json
from zipfile import ZipFile
from urllib.parse import urlparse
from pathlib import Path
from tempfile import NamedTemporaryFile
import requests
from PIL import Image, ImageDraw, ImageFont
def file_safe_string(string):
"""
Makes a string safe for saving to a file by replacing characters.
The characters "/", "&", "\\", "|", and "=" are replaced by " - ".
The ":" character is replaced by " -".
The "{" and "<" characters are replaced by "(", along with the corresponding closed brackets.
All quote-like characters are replaced by a single quote.
The "?", "*", "%", and "$" characters are removed entirely.
Note that file lengths are not considered here.
:param string: The string to convert.
:returns: The file-safe form of the string.
"""
new_string = re.sub(r'[/&\\|=]', " - ", string)
new_string = re.sub(r':', " -", new_string)
new_string = re.sub(r'[{<]', '(', new_string)
new_string = re.sub(r'[}>]', ')', new_string)
new_string = re.sub(r'["“‘`]', '\'', new_string)
new_string = re.sub(r'[?*%$]', '', new_string)
new_string = re.sub(r'\s+', ' ', new_string)
return new_string
def get_identifier_string(identifier):
"""
Obtains a default file-safe string representation of a comic or strip identifer. If the identifier is a list or tuple, the
elements will be joined by a dash character.
:param identifier: The identifier of the comic or strip.
:returns: The identifier in string form.
"""
if type(identifier) == str:
return identifier
elif type(identifier) == tuple or type(identifier) == list:
# Check for spaces. Their presence influenced the delimiter string.
if any(re.search(r'\s', id) for id in identifier):
delimiter = ' - '
else:
delimiter = '-'
return file_safe_string(delimiter.join([str(id) for id in identifier]))
else:
return str(identifier)
class SequenceException(Exception):
"""
This exception is thrown when a repository object is processed out of order (i.e. downloading images before we have loaded the images).
"""
pass
class ImageRepo(ABC):
"""
A generic representation of an online repository of data. This class allows for the
setup of downloading the main information in one pass, and secondary information (images)
seperately.
"""
def __init__(self, identifier, image_type):
"""
Creates an information repository
:param identifier: The globally-unique identifier used to find this repository.
:param image_type: The type of collection used for images (i.e. "dict" or "list"). This needs to be an indexed collection.
"""
self.identifier = identifier
self.image_urls = image_type()
"""A named collection of image URLs. The value is overridden by implementing classes. This can be any sort of indexed collection (named dictionary, ordered list, etc.)
Note that despite the name, this does not have to necessarily link to images. Any data that is accessible with GET HTTP requests are valid."""
self.images = image_type()
"""A collection of temporary files for the relevent images in #image_urls. This should be the same collection type as #image_urls."""
self._data_loaded = False
"""A flag for when the data has been successfully loaded."""
self._data_downloaded = False
"""A flag for when the data has been successfully downloaded."""
self._load_lock = Condition(Lock())
"""A lock for enforcing thread safety when loading this repository."""
self._download_lock = Condition(Lock())
"""A lock for enforcing thread safety when downloading secondary resources."""
def get_identifier_string(self):
"""
Obtains a file-safe string representation of the identifer. If the identifier is a list or tuple, the
elements will be joined by a dash character.
:returns: The identifier in string form.
"""
return get_identifier_string(self.identifier)
def is_loaded(self):
"""
Checks to see if the repostiory data is fully loaded.
:returns: True if the data has been fully loaded, false otherwise.
"""
return self._data_loaded
def is_downloaded(self):
"""
Checks to see if all images (if any are present) are downloaded.
:returns: True if the images have been fully downloaded, false otherwise.
"""
return self._data_downloaded
def load_data(self):
"""
Queries information on the comic in a thread-safe way.
This load is only performed once, and subsequent calls
will be ignored. Calls made while an existing load is
being performed are also ignored. If you need to obtain
the data directly after triggering a load, #await_load
should be called.
Implementing classes should overload the #_load_data method.
:see: #await_load
:see: #_load_data
"""
with self._load_lock:
if not self._data_loaded:
self._load_data()
logging.info("Completed loading of %s", self.get_identifier_string())
self._data_loaded = True
self._load_lock.notify_all()
def await_load(self):
"""
Locks the current thread until all data has been loaded, starting
a load as needed.
:see: #load_data
"""
# Loads this ourselves
self.load_data()
@abstractmethod
def _load_data(self):
"""
Queries information on the repository. To be implemented by successor classes.
Successor classes should not attempt any thread safety or be concerned about
checking for previous/concurrent calls to this method, as this logic is handled
by #load_data.
"""
pass
def download_data(self):
"""
Downloads all images for the comic.
This download is only performed once, and subsequent calls
will be ignored. Calls made while an existing download is
being performed are also ignored. If you need to obtain
the images directly after triggering a download, #await_download
should be called.
Implementing classes should overload the #_download_data method.
:see: #await_download
:see: #_download_data
"""
if not self.is_loaded():
raise SequenceException("Cannot download images before first loading performed.")
with self._download_lock:
# Checks to see if we need to download and that one is not already in progress
if self.image_urls and not self.images:
logging.info("Downloading %s", self.get_identifier_string())
self._download_data()
logging.info("Completed downloading of %s", self.get_identifier_string())
self._data_downloaded = True
self._download_lock.notify_all()
def await_download(self):
"""
Locks the current thread until all images have been downloaded, starting
the download as needed.
:see: #download_data
"""
self.download_data()
def _download_data(self):
"""
Downloads all images and saves them to temporary files in the #images field.
The default implementation should be satisfactory for most use cases, but
can be overridden if necessary. Overriders should not attempt any thread
safety or be concerned about checking for previous/concurrent calls to
this method, as this logic is handled by #download_data.
"""
if self.image_urls:
if type(self.image_urls) == dict:
iterator = self.image_urls.keys()
elif type(self.image_urls) == list:
iterator = range(len(self.image_urls))
for name in iterator:
logging.info(" Downloading repository %s resource %s", self.get_identifier_string(), name)
image_path = NamedTemporaryFile(mode='wb', suffix=str(name), prefix=self.get_identifier_string(), delete = False)
result = requests.get(self.image_urls[name])
result.raise_for_status()
image_path.write(result.content)
image_path.close()
if type(self.images) == list:
self.images.append(image_path)
else:
self.images[name] = image_path
class Comic(ImageRepo, ABC):
"""
A generic representation of a comic collection. Comic-wide resources are stored here.
"""
strip_cls = None
"""A reference to the class used for this comic's strips."""
def __init__(self, identifier):
"""
Creates a comic collection.
:param identifier: The globally-unique identifier used to find the comic.
"""
super().__init__(identifier, dict)
self.title = None
"""The title of the comic."""
self.author = None
"""The author(s) of the comic. Can be a string for a name, a list of names, or a list of objects containing the names under "name" and role under "role"."""
self.strip_index = {}
"""
The strip index maps comic identifiers to their filenames (sans stems). If we
cache this data, we can find out where each strip lives without having to
load data from the website, thus using less requests for already-downloaded
strips.
"""
self._index_loaded = False
"""A flag for when the index has been successfully loaded."""
self._index_lock = Condition(Lock())
"""A lock for enforcing thread safety when loading the comic index file."""
def read_index(self, base_path: Path):
"""
Saves the strip index to the same base path we download comic strips to.
It is saved to "index.json."
:param base_path: The folder to save the index to.
"""
with self._index_lock:
if not self._index_loaded:
index_path = Path(base_path, "index.json")
if index_path.exists():
try:
with open(Path(base_path, "index.json"), 'r') as index_fp:
self.strip_index = json.load(index_fp)
except json.JSONDecodeError as e:
logging.exception(e)
self.strip_index = {}
self._index_loaded = True
self._index_lock.notify_all()
def save_index(self, base_path: Path):
"""
Saves the strip index to the same base path we download comic strips to.
It is saved to "index.json."
:param base_path: The folder to save the index to.
"""
with self._index_lock:
with open(Path(base_path, "index.json"), 'w') as index_fp:
json.dump(self.strip_index, index_fp)
def index_strip(self, strip):
"""
Registers a comic to the index. This does not save the comic index.
:param strip: The comic strip to index
"""
with self._index_lock:
self.strip_index[strip.identifier] = strip.get_filename()
def is_strip_present(self, base_path: Path, suffix: str, identifier) -> bool:
"""
Checks to see if a given strip has been saved to the file.
:param base_path: The folder we expect the strip to show up in.
:param suffix: The type suffix we want to check (i.e. ".cbz")
:param identifier: The identifier to check.
:returns: True if the strip file has been saved.
"""
if not identifier in self.strip_index:
return False
path = Path(base_path, self.strip_index[identifier] + suffix)
return path.exists()
#@abstractmethod
@classmethod
def create_from_url(cls, url):
"""
Creates a Comic instance from a given URL
"""
pass
@abstractmethod
def _load_data(self):
"""
Queries the source to obtain information on the comic.
Implementing methods need to find the following information:
* The title of the comic, loaded to `title`.
* The author of the comic, loaded to `author`.
* A collection of banners, author avatars, and other
relevant images, loaded into `image_urls`.
"""
pass
@abstractmethod
def get_first_strip(self):
"""
Obtains the identifier of the first strip.
"""
pass
@abstractmethod
def get_latest_strip(self):
"""
Obtains the identifier of the latest strip.
"""
pass
@abstractmethod
def get_all_strips(self) -> list:
"""
Obtains a list of all strips.
"""
pass
def create_strip(self, identifier):
"""
Creates a new ComicStrip object.
The class variable #strip_cls must be set for this to work.
"""
return self.strip_cls(self, identifier)
class ComicStrip(ImageRepo, ABC):
"""
A generic representation of a single comic strip or issue.
"""
def __init__(self, comic, identifier):
"""
Creates a comic strip
:param comic: The `Comic` object that this strip is part of.
:param indentifier: The globally unique identifier used to find the strip on the website. Can be a tuple.
"""
super().__init__(identifier, list)
self.comic = comic
"""The comic that this strip is part of."""
self.title = None
"""The title of the comic. Empty until the comic data is loaded."""
self.image_urls = []
"""A list of URLs we can download the comic panels from, in order. Empty until the comic data is loaded."""
self._image_paths = []
"""A list of temporary files for the raw downloaded panels. Used internally."""
self.captions = []
"""A list of captions for each panel."""
self.date = None
"""The date that the comic was published."""
self._transform_lock = Condition(Lock())
"""A lock for enforcing thread safety when transforming resources."""
self.transformed_images = []
"""A list of NamedTemporaryFiles linking to the transformed versions of the image downloads."""
self._transformed = False
"""A flag keeping track of whether the strip transformation took place."""
def get_filename(self):
"""
Obtains the name of the final export (not including the stem). Calling this method requires the data be loaded first.
:returns: A human-friendly file, like "05 - Riddles in the Dark" or "2025-06-01 - Chapter 5."
"""
prefix_num_form = "{:02}"
if type(self.identifier) == int:
prefix = prefix_num_form.format(self.identifier)
elif (type(self.identifier) == list or type(self.identifier) == tuple) and type(self.identifier[-1]) == int:
prefix = prefix_num_form.format(self.identifier[-1])
elif type(self.date) == datetime.date:
prefix = self.date.isoformat()
elif type(self.date) == datetime.datetime:
prefix = self.date.date().isoformat()
else:
prefix = self.get_identifier_string()
if self.title:
return (prefix + " - " + file_safe_string(self.title))[0:250]
else:
return prefix
def get_package_path(self, base_path: Path):
"""
Obtains the path to save the comic archive to.
:param base_path: The folder to save comics to.
:returns: The path to save to.
"""
return Path(base_path, f"{self.get_filename()}.cbz")
def load_data(self):
super().load_data()
self.comic.load_data()
self.comic.index_strip(self)
def download_data(self):
super().download_data()
self.comic.download_data()
def await_load(self):
super().await_load()
self.comic.await_load()
def await_download(self):
super().await_download()
self.comic.await_download()
@abstractmethod
def _load_data(self):
"""
Queries the source to obtain information on the comic.
Implementing methods need to find the following information:
The title of the comic, loaded to `title`.
The publish date (or datetime) of the comic, loaded to `date`.
A list of panel images, loaded into `image_urls`.
A list of captions, loaded into `captions`. Any captions beyond the list of images can be added to new pages without comics.
"""
pass
@abstractmethod
def _transform_images(self):
"""
Called by the packaging method to create a series of locally-transformed
images and returning the ordered list of them.
This method is meant to do things like add alt text, descriptions, title pages, etc.
to the final comic. Simpler implementations can simply return a list of the
unmodified images with the comic headers from #comic at the beginning.
All implementations should assign the ordered list to #transformed_images. If this
field is not set, then packaging will default to just package the comic strip
images.
"""
self.transformed_images = self.images
def package_data(self, base_path: Path):
"""
Compresses all the data into a cbz file.
"""
if not self.is_loaded():
raise SequenceException("Cannot package comic strip before strip data has been fully loaded.")
if not self.is_downloaded():
raise SequenceException("Cannot package comic strip before strip data has been fully downloaded.")
if not self.comic.is_loaded():
raise SequenceException("Cannot package comic strip before comic data has been fully loaded.")
if not self.comic.is_downloaded():
raise SequenceException("Cannot package comic strip before comic data has been fully downloaded.")
if not self.transformed_images:
if not self._transformed:
if self._transform_lock._lock.locked():
# Await for the existing transform to complete
self._transform_lock.wait()
else:
# Run the transformation outselves
with self._transform_lock:
self._transform_images()
self._transformed = True
self._transform_lock.notify_all()
with ZipFile(self.get_package_path(base_path), 'w') as comic_zip:
for i in range(len(self.transformed_images)):
image = self.transformed_images[i]
image_path = Path(image.name)
with open(image.name, 'rb') as image_stream, comic_zip.open(str(i + 1) + image_path.suffix, 'w') as zip_stream:
zip_stream.write(image_stream.read())
# Delete the base images, as they are no longer needed.
# We cannot touch the transformed images, since some
# of them may come directly from the base comic, and
# we have no way of telling.
for image in self.images:
image_path = Path(image.name)
if image_path.exists():
image_path.unlink()
class URLFormat:
"""An internal class used to help sort URLs based on subdomains and paths"""
def __init__(self, url: str):
self.raw_url = url
if not re.match(r'^\w+://', url):
url = 'http://' + url
self.url = urlparse(url)
self.domains = self.url.netloc.split('.')
self.paths = self.url.path.split('/')
def __lt__(self, other):
if type(other) == str:
other = URLFormat(other)
if len(self.domains) != len(other.domains):
return len(self.domains) < len(other.domains)
if self.domains != other.domains:
return self.domains < other.domains
if len(self.paths) != len(other.paths):
return len(self.paths) < len(other.paths)
if self.paths != other.paths:
return self.paths < other.paths
return False
def __le__(self, other):
if type(other) == str:
other = URLFormat(other)
if self == other:
return True
return self < other
def __gt__(self, other):
if type(other) == str:
other = URLFormat(other)
if len(self.domains) != len(other.domains):
return len(self.domains) > len(other.domains)
if self.domains != other.domains:
return self.domains > other.domains
if len(self.paths) != len(other.paths):
return len(self.paths) > len(other.paths)
if self.paths != other.paths:
return self.paths > other.paths
return False
def __ge__(self, other):
if type(other) == str:
other = URLFormat(other)
if self == other:
return True
return self > other
def __eq__(self, other):
if type(other) == str:
other = URLFormat(other)
return self.domains == other.domains and self.paths == other.paths
def __ne__(self, other):
if type(other) == str:
other = URLFormat(other)
return self.domains != other.domains or self.paths != other.paths
def __str__(self):
return f"{self.url.netloc}{self.url.path}"
def __repr__(self):
return f"\"{self.__str__()}\""
def __hash__(self):
return hash(tuple(self.domains + self.paths))
comic_registry = {}
"""A registry that maps URL Domains to Comic classes."""
_registry_subdomains = []
"""An internal list keeping track of the order we should use to find the right parser (priority is last)"""
def register_comic_class(url: str, comic_cls: type[Comic]):
"""
Registers a comic type.
The URL format provided here is used to help determine
which class should be used for a given strip. There are
a few rules for the URL parameter provided here:
1. The url should not include the schema (i.e. "http://").
For example, "xkcd.com" is okay, but "https://xkcd.com" is
not.
2. The URL provided here should usually just express the
domain name. Paths should only be expressed if absolutely
necessary, such as in cases where different comic formats
are published to the same site (i.e. Tumblr). For example,
"bsky.app" works, but "bsky.app/profile" is discourages
(although "bsky.app/profile/strange_user.bsky.social"
is acceptable if strange_user does something unusual
with the way he publishes his comics and needs a custom
parser).
3. If the subdomain is "www", it should be ommited.
Use "tumblr.com" instead of "www.tumblr.com" (although
"john_doe.tumblr.com" is valid).
:param url: The URL format to register.
:param comic_cls: The class to use for parsing this comic.
"""
url_f = URLFormat(url)
if url_f in comic_registry:
raise ValueError("URL %s is already registered" % url)
comic_registry[url_f] = comic_cls
bisect.insort(_registry_subdomains, url_f)
def find_comic_class(url: str) -> type[Comic]:
"""
Finds the appropriate comic for a given URL.
:param url: The URL of the comic to parse. This must be a proper URL (including schema, etc.)
:returns: A parser
"""
url_f = URLFormat(url)
logging.info(_registry_subdomains)
while len(url_f.domains) > 0:
i = bisect.bisect(_registry_subdomains, url_f)
if i == 0:
raise ValueError("Could not find a parser for url \"%s\"" % url)
if url_f.domains == _registry_subdomains[i - 1].domains:
# We have found a series of parsers with a matching subdomain.
# Now we check the paths.
while len(url_f.paths) > 0:
i = bisect.bisect(_registry_subdomains, url_f)
if i == 0:
raise ValueError("Could not find a parser for url \"%s\"" % url)
if url_f.paths == _registry_subdomains[i - 1].paths:
# We have found a registered URL pattern that matches the URL
# provided.
return comic_registry[url_f]
else:
# Drill down to find an exact matching path
url_f = URLFormat('.'.join(url_f.domains) + '/'.join(url_f.paths[:-1]))
else:
# The result we found has a different subdomain. Look for the root subdomain
url_f = URLFormat('.'.join(url_f.domains[1:]) + '/'.join(url_f.paths))
raise ValueError("Could not find a parse for url \"%s\"" % url)