diff --git a/.gitignore b/.gitignore index f1d8d08..b273408 100644 --- a/.gitignore +++ b/.gitignore @@ -2,3 +2,4 @@ *.log **/__pycache__ *~ +index.json diff --git a/__main__.py b/__main__.py index 68d18c7..d286bac 100644 --- a/__main__.py +++ b/__main__.py @@ -71,11 +71,14 @@ if __name__ == "__main__": logging.info("Beginning parsing") threads = [] + comics = [] thread_limit = threading.BoundedSemaphore(value=args.threads) for url in args.url: comic_type = find_comic_class(url.netloc + url.path) comic = comic_type.create_from_url(url) + comics.append(comic) + comic.read_index(base_path) comic.await_load() logging.info("There are %d comics", comic.latest_identifier) if args.latest: @@ -85,11 +88,19 @@ if __name__ == "__main__": start = args.start or comic.get_first_strip() end = args.end or comic.get_latest_strip() r = r[start - 1:end] - - for index in reversed(r): - t = threading.Thread(name=str(comic.get_identifier_string()) + "-" + str(index), target=load_strip, args=(base_path, comic, index, args.plain, thread_limit)) - threads.append(t) - t.start() - for t in threads: - t.join() + try: + for index in reversed(r): + if not comic.is_strip_present(base_path, ".cbz", str(index)): + t = threading.Thread(name=str(comic.get_identifier_string()) + "-" + str(index), target=load_strip, args=(base_path, comic, index, args.plain, thread_limit)) + threads.append(t) + t.start() + finally: + comic.save_index(base_path) + + try: + for t in threads: + t.join() + finally: + for comic in comics: + comic.save_index(base_path) diff --git a/comic_download/comic_strip.py b/comic_download/comic_strip.py index dd7c42f..65d6c11 100644 --- a/comic_download/comic_strip.py +++ b/comic_download/comic_strip.py @@ -263,6 +263,73 @@ class Comic(ImageRepo, ABC): """The title of the comic.""" self.author = None """The author(s) of the comic. Can be a string for a name, a list of names, or a list of objects containing the names under "name" and role under "role".""" + + self.strip_index = {} + """ + The strip index maps comic identifiers to their filenames (sans stems). If we + cache this data, we can find out where each strip lives without having to + load data from the website, thus using less requests for already-downloaded + strips. + """ + self._index_loaded = False + """A flag for when the index has been successfully loaded.""" + self._index_lock = Condition(Lock()) + """A lock for enforcing thread safety when loading the comic index file.""" + + def read_index(self, base_path: Path): + """ + Saves the strip index to the same base path we download comic strips to. + It is saved to "index.json." + + :param base_path: The folder to save the index to. + """ + with self._index_lock: + if not self._index_loaded: + index_path = Path(base_path, "index.json") + if index_path.exists(): + try: + with open(Path(base_path, "index.json"), 'r') as index_fp: + self.strip_index = json.load(index_fp) + except json.JSONDecodeError as e: + logging.exception(e) + self.strip_index = {} + self._index_loaded = True + self._index_lock.notify_all() + + def save_index(self, base_path: Path): + """ + Saves the strip index to the same base path we download comic strips to. + It is saved to "index.json." + + :param base_path: The folder to save the index to. + """ + with self._index_lock: + with open(Path(base_path, "index.json"), 'w') as index_fp: + json.dump(self.strip_index, index_fp) + + def index_strip(self, strip): + """ + Registers a comic to the index. This does not save the comic index. + + :param strip: The comic strip to index + """ + with self._index_lock: + self.strip_index[strip.identifier] = strip.get_filename() + + def is_strip_present(self, base_path: Path, suffix: str, identifier) -> bool: + """ + Checks to see if a given strip has been saved to the file. + + :param base_path: The folder we expect the strip to show up in. + :param suffix: The type suffix we want to check (i.e. ".cbz") + :param identifier: The identifier to check. + :returns: True if the strip file has been saved. + """ + if not identifier in self.strip_index: + return False + path = Path(base_path, self.strip_index[identifier] + suffix) + return path.exists() + #@abstractmethod @classmethod @@ -382,6 +449,7 @@ class ComicStrip(ImageRepo, ABC): def load_data(self): super().load_data() self.comic.load_data() + self.comic.index_strip(self) def download_data(self): super().download_data()