From d7e6e812739919e47ee93d27b94b2f3bc3c3a4c7 Mon Sep 17 00:00:00 2001 From: Dragonatorul <109987846+Dragonatorul@users.noreply.github.com> Date: Sat, 7 Jun 2025 21:04:01 +0300 Subject: [PATCH 01/11] wip: adding wpmadara --- gallery_dl/extractor/__init__.py | 2 +- gallery_dl/extractor/common.py | 56 +++++++++++++++++++ .../extractor/{mangaread.py => wpmadara.py} | 37 ++++++++---- test/results/mangaread.py | 16 +++--- 4 files changed, 92 insertions(+), 19 deletions(-) rename gallery_dl/extractor/{mangaread.py => wpmadara.py} (80%) diff --git a/gallery_dl/extractor/__init__.py b/gallery_dl/extractor/__init__.py index fcf27c8a10..15483a48e7 100644 --- a/gallery_dl/extractor/__init__.py +++ b/gallery_dl/extractor/__init__.py @@ -106,7 +106,7 @@ "mangahere", "manganelo", "mangapark", - "mangaread", + "wpmadara", "mangoxo", "misskey", "motherless", diff --git a/gallery_dl/extractor/common.py b/gallery_dl/extractor/common.py index b7562a1ba3..b8b16db5d9 100644 --- a/gallery_dl/extractor/common.py +++ b/gallery_dl/extractor/common.py @@ -28,6 +28,7 @@ class Extractor(): + """Base class for all extractors""" category = "" subcategory = "" @@ -50,6 +51,18 @@ class Extractor(): request_timestamp = 0.0 def __init__(self, match): + """ + Initialize the CommonExtractor object. + + Args: + match (re.Match): The regular expression match object. + + Attributes: + log (logging.Logger): The logger object for logging messages. + url (str): The URL string. + _cfgpath (tuple): The configuration path. + _parentdir (str): The parent directory. + """ self.log = logging.getLogger(self.category) self.url = match.string self.match = match @@ -62,16 +75,31 @@ def __init__(self, match): @classmethod def from_url(cls, url): + """ + Create an instance of the class from a given URL. + + Args: + url (str): The URL to match against the class's pattern. + + Returns: + cls: An instance of the class if the URL matches the pattern, otherwise None. + """ if isinstance(cls.pattern, str): cls.pattern = util.re_compile(cls.pattern) match = cls.pattern.match(url) return cls(match) if match else None def __iter__(self): + """ + Returns an iterator object that iterates over the items in the extractor. + """ self.initialize() return self.items() def initialize(self): + """ + Initializes the extractor. + """ self._init_options() self._init_session() self._init_cookies() @@ -79,15 +107,43 @@ def initialize(self): self.initialize = util.noop def finalize(self): + """ + This method is called to perform any necessary finalization steps. + """ pass def items(self): + """ + Generator that yields a tuple of Message.Version and 1. + + Returns: + A tuple containing Message.Version and 1. + """ yield Message.Version, 1 def skip(self, num): + """ + Skips a specified number of items. + + Args: + num (int): The number of items to skip. + + Returns: + int: The number of items skipped. + """ return 0 def config(self, key, default=None): + """ + Retrieves the value of a configuration key from the specified configuration file. + + Args: + key (str): The configuration key to retrieve. + default (Any, optional): The default value to return if the key is not found. Defaults to None. + + Returns: + Any: The value of the configuration key, or the default value if the key is not found. + """ return config.interpolate(self._cfgpath, key, default) def config2(self, key, key2, default=None, sentinel=util.SENTINEL): diff --git a/gallery_dl/extractor/mangaread.py b/gallery_dl/extractor/wpmadara.py similarity index 80% rename from gallery_dl/extractor/mangaread.py rename to gallery_dl/extractor/wpmadara.py index 6444282e78..c54f9ed9c9 100644 --- a/gallery_dl/extractor/mangaread.py +++ b/gallery_dl/extractor/wpmadara.py @@ -6,13 +6,14 @@ """Extractors for https://mangaread.org/""" -from .common import ChapterExtractor, MangaExtractor -from .. import text, util, exception +from .common import BaseExtractor, ChapterExtractor, MangaExtractor +from .. import text, exception +import re -class MangareadBase(): +class WPMadaraBase(BaseExtractor): """Base class for Mangaread extractors""" - category = "mangaread" + basecategory = "wpmadara" root = "https://www.mangaread.org" @staticmethod @@ -29,13 +30,24 @@ def parse_chapter_string(chapter_string, data): data["lang"] = "en" data["language"] = "English" +BASE_PATTERN = WPMadaraBase.update({ + "mangaread": { + "root": "https://www.mangaread.org", + "pattern": r"(?:https?://)?(?:www\.)?mangaread\.org", + }, +}) -class MangareadChapterExtractor(MangareadBase, ChapterExtractor): + +class WPMadaraChapterExtractor(WPMadaraBase): """Extractor for manga-chapters from mangaread.org""" - pattern = (r"(?:https?://)?(?:www\.)?mangaread\.org" - r"(/manga/[^/?#]+/[^/?#]+)") + subcategory = "chapter" + pattern = BASE_PATTERN + r"(/(manga)/[^/?#]+/[^/?#]+)" example = "https://www.mangaread.org/manga/MANGA/chapter-01/" + def __init__(self, match): + WPMadaraBase.__init__(self, match) + self.chapter = match.group(match.lastindex) + def metadata(self, page): tags = text.extr(page, 'class="wp-manga-tags-list">', '') data = {"tags": list(text.split_html(tags)[::2])} @@ -54,12 +66,17 @@ def images(self, page): ] -class MangareadMangaExtractor(MangareadBase, MangaExtractor): +class WPMadaraMangaExtractor(WPMadaraBase): """Extractor for manga from mangaread.org""" - chapterclass = MangareadChapterExtractor - pattern = r"(?:https?://)?(?:www\.)?mangaread\.org(/manga/[^/?#]+)/?$" + chapterclass = WPMadaraChapterExtractor + subcategory = "manga" + pattern = BASE_PATTERN + r"(/(manga)/[^/?#]+)/?$" example = "https://www.mangaread.org/manga/MANGA" + def __init__(self, match): + WPMadaraBase.__init__(self, match) + self.manga = match.group(match.lastindex) + def chapters(self, page): if 'class="error404' in page: raise exception.NotFoundError("manga") diff --git a/test/results/mangaread.py b/test/results/mangaread.py index 1c5e9a356a..c31786f66a 100644 --- a/test/results/mangaread.py +++ b/test/results/mangaread.py @@ -4,7 +4,7 @@ # it under the terms of the GNU General Public License version 2 as # published by the Free Software Foundation. -from gallery_dl.extractor import mangaread +from gallery_dl.extractor import wpmadara from gallery_dl import exception @@ -12,7 +12,7 @@ { "#url" : "https://www.mangaread.org/manga/one-piece/chapter-1053-3/", "#category": ("", "mangaread", "chapter"), - "#class" : mangaread.MangareadChapterExtractor, + "#class" : wpmadara.MangareadChapterExtractor, "#pattern" : r"https://www\.mangaread\.org/wp-content/uploads/WP-manga/data/manga_[^/]+/[^/]+/[^.]+\.\w+", "#count" : 11, @@ -28,14 +28,14 @@ { "#url" : "https://www.mangaread.org/manga/one-piece/chapter-1000000/", "#category": ("", "mangaread", "chapter"), - "#class" : mangaread.MangareadChapterExtractor, + "#class" : wpmadara.MangareadChapterExtractor, "#exception": exception.NotFoundError, }, { "#url" : "https://www.mangaread.org/manga/kanan-sama-wa-akumade-choroi/chapter-10/", "#category": ("", "mangaread", "chapter"), - "#class" : mangaread.MangareadChapterExtractor, + "#class" : wpmadara.MangareadChapterExtractor, "#pattern" : r"https://www\.mangaread\.org/wp-content/uploads/WP-manga/data/manga_[^/]+/[^/]+/[^.]+\.\w+", "#count" : 9, @@ -52,7 +52,7 @@ "#url" : "https://www.mangaread.org/manga/above-all-gods/chapter146-5/", "#comment" : "^^ no whitespace", "#category": ("", "mangaread", "chapter"), - "#class" : mangaread.MangareadChapterExtractor, + "#class" : wpmadara.MangareadChapterExtractor, "#pattern" : r"https://www\.mangaread\.org/wp-content/uploads/WP-manga/data/manga_[^/]+/[^/]+/[^.]+\.\w+", "#count" : 6, @@ -68,7 +68,7 @@ { "#url" : "https://www.mangaread.org/manga/kanan-sama-wa-akumade-choroi", "#category": ("", "mangaread", "manga"), - "#class" : mangaread.MangareadMangaExtractor, + "#class" : wordpressmadara.MangareadMangaExtractor, "#pattern" : r"https://www\.mangaread\.org/manga/kanan-sama-wa-akumade-choroi/chapter-\d+([_-].+)?/", "#count" : ">= 13", @@ -94,7 +94,7 @@ { "#url" : "https://www.mangaread.org/manga/one-piece", "#category": ("", "mangaread", "manga"), - "#class" : mangaread.MangareadMangaExtractor, + "#class" : wordpressmadara.MangareadMangaExtractor, "#pattern" : r"https://www\.mangaread\.org/manga/one-piece/chapter-\d+(-.+)?/", "#count" : ">= 1066", @@ -115,7 +115,7 @@ { "#url" : "https://www.mangaread.org/manga/doesnotexist", "#category": ("", "mangaread", "manga"), - "#class" : mangaread.MangareadMangaExtractor, + "#class" : wordpressmadara.MangareadMangaExtractor, "#exception": exception.NotFoundError, }, From 161c4a24626e00a83e50161b190cf48898ce4541 Mon Sep 17 00:00:00 2001 From: Dragonatorul <109987846+Dragonatorul@users.noreply.github.com> Date: Sat, 7 Jun 2025 21:04:35 +0300 Subject: [PATCH 02/11] undo comment changes --- gallery_dl/extractor/common.py | 48 +++------------------------------- 1 file changed, 4 insertions(+), 44 deletions(-) diff --git a/gallery_dl/extractor/common.py b/gallery_dl/extractor/common.py index b8b16db5d9..3bd7c77e1d 100644 --- a/gallery_dl/extractor/common.py +++ b/gallery_dl/extractor/common.py @@ -28,7 +28,6 @@ class Extractor(): - """Base class for all extractors""" category = "" subcategory = "" @@ -51,6 +50,10 @@ class Extractor(): request_timestamp = 0.0 def __init__(self, match): + self.log = logging.getLogger(self.category) + self.url = match.string + self._cfgpath = ("extractor", self.category, self.subcategory) + self._parentdir = "" """ Initialize the CommonExtractor object. @@ -75,31 +78,16 @@ def __init__(self, match): @classmethod def from_url(cls, url): - """ - Create an instance of the class from a given URL. - - Args: - url (str): The URL to match against the class's pattern. - - Returns: - cls: An instance of the class if the URL matches the pattern, otherwise None. - """ if isinstance(cls.pattern, str): cls.pattern = util.re_compile(cls.pattern) match = cls.pattern.match(url) return cls(match) if match else None def __iter__(self): - """ - Returns an iterator object that iterates over the items in the extractor. - """ self.initialize() return self.items() def initialize(self): - """ - Initializes the extractor. - """ self._init_options() self._init_session() self._init_cookies() @@ -107,43 +95,15 @@ def initialize(self): self.initialize = util.noop def finalize(self): - """ - This method is called to perform any necessary finalization steps. - """ pass def items(self): - """ - Generator that yields a tuple of Message.Version and 1. - - Returns: - A tuple containing Message.Version and 1. - """ yield Message.Version, 1 def skip(self, num): - """ - Skips a specified number of items. - - Args: - num (int): The number of items to skip. - - Returns: - int: The number of items skipped. - """ return 0 def config(self, key, default=None): - """ - Retrieves the value of a configuration key from the specified configuration file. - - Args: - key (str): The configuration key to retrieve. - default (Any, optional): The default value to return if the key is not found. Defaults to None. - - Returns: - Any: The value of the configuration key, or the default value if the key is not found. - """ return config.interpolate(self._cfgpath, key, default) def config2(self, key, key2, default=None, sentinel=util.SENTINEL): From d1f26e06c47a766fbebe737c9fb7414cd917bf64 Mon Sep 17 00:00:00 2001 From: Dragonatorul <109987846+Dragonatorul@users.noreply.github.com> Date: Sat, 7 Jun 2025 21:04:35 +0300 Subject: [PATCH 03/11] add support for toonily and webtoonxyz --- gallery_dl/extractor/wpmadara.py | 28 +++++++++++--- test/results/toonily.py | 64 ++++++++++++++++++++++++++++++++ test/results/webtoonxyz.py | 62 +++++++++++++++++++++++++++++++ 3 files changed, 148 insertions(+), 6 deletions(-) create mode 100644 test/results/toonily.py create mode 100644 test/results/webtoonxyz.py diff --git a/gallery_dl/extractor/wpmadara.py b/gallery_dl/extractor/wpmadara.py index c54f9ed9c9..ea0087d9e2 100644 --- a/gallery_dl/extractor/wpmadara.py +++ b/gallery_dl/extractor/wpmadara.py @@ -35,18 +35,31 @@ def parse_chapter_string(chapter_string, data): "root": "https://www.mangaread.org", "pattern": r"(?:https?://)?(?:www\.)?mangaread\.org", }, + "toonily": { + "root": "https://www.toonily.com", + "pattern": r"(?:https?://)?(?:www\.)?toonily\.com", + }, + "webtoonxyz": { + "root": "https://www.webtoon.xyz", + "pattern": r"(?:https?://)?(?:www\.)?webtoon\.xyz", + }, }) -class WPMadaraChapterExtractor(WPMadaraBase): +class WPMadaraChapterExtractor(WPMadaraBase, ChapterExtractor): """Extractor for manga-chapters from mangaread.org""" subcategory = "chapter" - pattern = BASE_PATTERN + r"(/(manga)/[^/?#]+/[^/?#]+)" + pattern = BASE_PATTERN + r"(/(manga|webtoon|read)/[^/?#]+/[^/?#]+)" example = "https://www.mangaread.org/manga/MANGA/chapter-01/" - def __init__(self, match): + def __init__(self, match, url=None): WPMadaraBase.__init__(self, match) self.chapter = match.group(match.lastindex) + self.log.debug("chapter: %s", self.chapter) + self.gallery_url = self.root + match.group(match.lastindex) + + if self.config("chapter-reverse", False): + self.reverse = not self.reverse def metadata(self, page): tags = text.extr(page, 'class="wp-manga-tags-list">', '') @@ -55,27 +68,30 @@ def metadata(self, page): if not info: raise exception.NotFoundError("chapter") self.parse_chapter_string(info, data) + self.log.debug("data: %s", data) return data def images(self, page): page = text.extr( page, '