diff --git a/Comics/exporters.py b/Comics/exporters.py index 8ce1550..7543c42 100644 --- a/Comics/exporters.py +++ b/Comics/exporters.py @@ -52,9 +52,9 @@ class JsonExport(JsonItemExporter): class ComicInfoXmlItemExporter(XmlItemExporter): custom_root_element = "ComicInfo" - def __init__(self, comic, chapter): - file_path = os.path.join(COMIC_INFO_XML_STORE, comic, - chapter, f"{self.custom_root_element}.xml") + def __init__(self, dir): + file_path = os.path.join(COMIC_INFO_XML_STORE, dir, + f"{self.custom_root_element}.xml") dir_path = os.path.dirname(file_path) if not os.path.exists(dir_path): os.makedirs(dir_path) self.xml_file = open(file_path, "wb") diff --git a/Comics/items.py b/Comics/items.py index cf2eeda..173f28b 100644 --- a/Comics/items.py +++ b/Comics/items.py @@ -6,8 +6,12 @@ import os,Comics.settings as settings,logging from scrapy.item import Item, Field from Comics.utils.Constant import ComicPath from Comics.utils.FileUtils import imageUtils -from scrapy.loader.processors import TakeFirst, MapCompose, Join +from itemloaders.processors import TakeFirst, MapCompose, Join +from scrapy.spiders import Spider +def current_project(): + return Spider.name + def serialize_to_chinese(value): return ComicPath.chinese_convert(value) @@ -35,7 +39,7 @@ def _serialize_to_images(value, result_type=None): images_item.append(image_src) image_urls.append(image_src) count += 1 - logging.info(f"images_len: {len(images_item)}") + logging.debug(f"images_len: {len(images_item)}") if result_type == "image_urls": return image_urls else: return images_item @@ -51,6 +55,8 @@ class ListComicItem(Item): class ComicItem(Item): + # 工程 + current_project = Field() # 编号 index = Field(output_processor=TakeFirst()) # 漫画名 @@ -113,7 +119,8 @@ def _serializer_info_imagesa(value, result_type=None): def _serialize_info_images(value, result_type=None): images = [] for image in value: - images.append(ComicPath().getFileScrambleImageSave(image,True,False)) + if os.sep not in image: + images.append(ComicPath().getFileScrambleImageSave(image,True,False)) if result_type == "count": return len(images) else: diff --git a/Comics/pipelines.py b/Comics/pipelines.py index 225a3b3..d06453e 100644 --- a/Comics/pipelines.py +++ b/Comics/pipelines.py @@ -5,8 +5,9 @@ # useful for handling different item types with a single interface -import os, scrapy,logging,time,random +import os, scrapy,logging,time,random,shutil from Comics import settings +from Comics.settings import CBZ_EXPORT_PATH,OUTPUT_DIR,PROJECT_KEY from Comics.utils.Constant import ComicPath from Comics.items import ComicItem from scrapy.pipelines.images import ImagesPipeline @@ -19,43 +20,81 @@ class ComicsPipeline: # item就是yield后面的对象 def process_item(self, item, spider): if isinstance(item, ComicItem): - file = os.path.join(settings.OUTPUT_DIR,"json", item['name'], item['chapter']) + file = os.path.join(OUTPUT_DIR, spider.name, "json", item['name'], item['chapter']) item['count'] = len(item['images']) - item['images'].append({'src':'icon.jpg', 'scramble': False}) - item['image_urls'].append({'src': item['icon'] , 'scramble': False}) data = JsonExport(file=file).export_json(item, if_return=True) - #item['images'] = data['images'] + data[PROJECT_KEY] = spider.name return data # image解析 def close_spider(self,spider): pass - class ImgDownloadPipeline(ImagesPipeline): - def file_exits(self, image_path): + def get_file_path(self, item, file=None, result_type="image"): + if result_type == "image": + if os.path.sep not in file: + file = os.path.join(item[settings.PROJECT_KEY], "images", item['name'], item['chapter'], file) + elif result_type == "comic_info": + file = os.path.join(item[settings.PROJECT_KEY], "images", item['name'], item['chapter']) + elif result_type == "cbz_icon": + file = os.path.join("CBZ", item[settings.PROJECT_KEY], item['name'], item['chapter']+".jpg") + elif result_type == "down_icon": + file = self.download_icon(item,result_type='fullpath') + elif result_type == "icon": + file = os.path.join(item[settings.PROJECT_KEY], "icons", item['name'], item['chapter']+".jpg") + elif result_type == "cbz": + file = os.path.join("CBZ", item[settings.PROJECT_KEY], item['name'], item['chapter']+".CBZ") + elif result_type == "images_dir": + file = os.path.join(settings.IMAGES_STORE, item[settings.PROJECT_KEY], "images", item['name'], item['chapter']) + return file + + def image_scramble_exits(self, item,image_path): en_image_path = ComicPath().getFileScrambleImageSave(image_path, relative="fullpath") - return os.path.exists(os.path.join(settings.IMAGES_STORE, en_image_path)) - - def file_full_path(self, item, image): return os.path.join(item['name'], item['chapter'], image) + return os.path.exists(os.path.join(settings.IMAGES_STORE, self.get_file_path(item, en_image_path))) + ## Icon Path : CBZ/NAME/CHAPTER.jpg + def download_icon(self, item, result_type="download"): + icon_path = self.get_file_path(item, result_type="icon") + if result_type == "fullpath": + return os.path.join(settings.IMAGES_STORE, icon_path) + if os.path.exists(icon_path): + return False + else: + self.image_urls.append(item['icon']) + self.images.append(icon_path) + return True + def file_path(self, request, response=None, info=None, *, item=None): return request.meta['path'] def get_media_requests(self, item, info): - for image_url,image_path in zip(item['image_urls'],item['images']): - image_path = self.file_full_path(item, image_path) - if self.file_exits(image_path): - logging.info(f"file exists: {image_path}") + self.image_urls = item['image_urls'] + self.images = item['images'] + # 下载封面 + self.download_icon(item) + for image_url,image in zip(self.image_urls,self.images): + image_path = self.get_file_path(item, image) + if self.image_scramble_exits(item, image_path): + logging.info(f"file exists: IMAGE_STORE {image_path}") else: - logging.info(f"downloading {image_url} --> {image_path}") + logging.info(f"downloading {image_url} --> IMAGE_STORE {image_path}") yield scrapy.Request(url=image_url, meta={'path': image_path}) + def pack_icon(self, item): + cbz_icon = self.get_file_path(item=item, result_type="cbz_icon") + dwn_icon = self.get_file_path(item=item, result_type="down_icon") + logging.info(f"icon packing {dwn_icon} => {cbz_icon}") + cbz_icon_dir = os.path.dirname(cbz_icon) + if not os.path.exists(cbz_icon_dir): os.makedirs(cbz_icon_dir) + shutil.copyfile(dwn_icon, cbz_icon) + def item_completed(self, results, item, info): - item['images_name'] = results # return item # ComicInfoXml 生成 - comic_info = ComicInfoXmlItemExporter(comic=item['name'], chapter=item['chapter']).export_xml(item) + comic_info = ComicInfoXmlItemExporter(dir=self.get_file_path(item=item, result_type="comic_info")).export_xml(item) # 打包 - CBZUtils.packComicChapterCBZ(comic=item['name'], chapter=item['chapter'], - comic_info_images= comic_info["Pages"], remove=False) + if CBZUtils.packComicChapterCBZ(src_dir= self.get_file_path(item, result_type="images_dir"), + dts_path= self.get_file_path(item, result_type="cbz"), + comic_info_images= comic_info['Pages'], remove=False): + self.pack_icon(item) time.sleep(random.randint(5,10)) \ No newline at end of file diff --git a/Comics/settings.py b/Comics/settings.py index 029d18f..4062709 100644 --- a/Comics/settings.py +++ b/Comics/settings.py @@ -9,11 +9,11 @@ from fake_useragent import UserAgent import os +PROJECT_KEY = "current_project" BOT_NAME = 'Comics' SPIDER_MODULES = ['Comics.spiders'] NEWSPIDER_MODULE = 'Comics.spiders' - OUTPUT_DIR = "output" # Crawl responsibly by identifying yourself (and your website) on the user-agent #USER_AGENT = 'Comics (+http://www.yourdomain.com)' @@ -23,12 +23,12 @@ ROBOTSTXT_OBEY = False HTTPERROR_ALLOWED_CODES = [ 200 , 403] # Configure maximum concurrent requests performed by Scrapy (default: 16) -CONCURRENT_REQUESTS = 16 +CONCURRENT_REQUESTS = 16 # Configure a delay for requests for the same website (default: 0) # See https://docs.scrapy.org/en/latest/topics/settings.html#download-delay # See also autothrottle settings and docs -IMAGES_STORE = os.path.join(OUTPUT_DIR, 'images') +IMAGES_STORE = "output" IMAGES_NAME_FORMAT = "{:0>3d}" COMIC_INFO_XML_STORE = IMAGES_STORE DOWNLOAD_DELAY = 0 @@ -36,7 +36,7 @@ DOWNLOAD_DELAY = 0 RETRY_ENABLED = True RETRY_TIMES = 10 # 想重试几次就写几 # 下面这行可要可不要 -RETRY_HTTP_CODES = [500, 502, 503, 504, 408, 401] +# RETRY_HTTP_CODES = [500, 502, 503, 504, 408, 401] # The download delay setting will honor only one of: CONCURRENT_REQUESTS_PER_DOMAIN = 16 CONCURRENT_REQUESTS_PER_IP = 16 @@ -105,7 +105,6 @@ AUTOTHROTTLE_DEBUG = False HTTPCACHE_ENABLED = True HTTPCACHE_EXPIRATION_SECS = 0 HTTPCACHE_DIR = 'httpcache' -HTTPCACHE_IGNORE_HTTP_CODES = [500, 502, 404] HTTPCACHE_ALLOW_PREFIXS = [ 'jpg', 'png', 'gif'] HTTPCACHE_STORAGE = 'Comics.middlewares.MyFilesystemCacheStorage' diff --git a/Comics/spiders/rm_comic.py b/Comics/spiders/rm_comic.py index 4cf4e9c..2f10ad7 100644 --- a/Comics/spiders/rm_comic.py +++ b/Comics/spiders/rm_comic.py @@ -19,7 +19,6 @@ class RmComicSpider(scrapy.Spider): books = books_comic.get_exec(data, str_exec=str_exec) for book in books: books_comic.add_value('link', book['id']) - logging.info(f"downloading books %s" % book['name']) time.sleep(3) yield scrapy.Request(url=self.start_urls+"/"+book['id'], callback=self.parse_comic) diff --git a/Comics/utils/FileUtils.py b/Comics/utils/FileUtils.py index 036fa4a..9f00a74 100644 --- a/Comics/utils/FileUtils.py +++ b/Comics/utils/FileUtils.py @@ -275,30 +275,27 @@ class CBZUtils: y = 0 for filename in filenames: y = y + 1 - logging.info(f"打包中:" + str(y) + "/" + str(len(filenames)), os.path.join(source_dir, filename)) + print("打包中:" + str(y) + "/" + str(len(filenames)), os.path.join(source_dir, filename)) zf.write(path.joinpath(filename), arc_dir.joinpath(filename)) zf.close() logging.info(f"打包完成:{target_file}") @classmethod - def packComicChapterCBZ(cls, comic, chapter, comic_info_images, remove=True): - images_chapter_path = os.path.join(IMAGES_STORE, comic, chapter) - cbz_chapter_path = os.path.join(CBZ_EXPORT_PATH, comic, chapter) + ".CBZ" - if os.path.exists(images_chapter_path): - dirs = os.listdir(images_chapter_path) + def packComicChapterCBZ(cls, src_dir, dts_path, comic_info_images, remove=True): + if os.path.exists(src_dir): + dirs = os.listdir(src_dir) for file in dirs: if file.startswith(ComicPath.PREFIX_SCRAMBLE): try: - imageUtils.deScrambleImagesByPath(os.path.join(images_chapter_path,file)) + imageUtils.deScrambleImagesByPath(os.path.join(src_dir,file)) except Exception as e: print(f"删除 {file} 发生错误 {e},已跳过") return False - cls.zip_compression(images_chapter_path, cbz_chapter_path) + cls.zip_compression(src_dir, dts_path) time.sleep(0.1) - if remove: shutil.rmtree(images_chapter_path) + if remove: shutil.rmtree(src_dir) # validation - cls.cbz_validate(cbz_chapter_path, comic_info_images) - return True + return cls.cbz_validate(dts_path, comic_info_images) @classmethod def replaceZip(cls, filepath, unpack_dir=None): @@ -356,6 +353,8 @@ class CBZUtils: def cbz_validate(cls, zip_path, comic_info_images): if len(cls.zip_info(zip_path)) == len(comic_info_images): logging.info(f"validating successfully === {zip_path}") + return True else: os.remove(zip_path) - logging.error(f"validating fail === {zip_path}") \ No newline at end of file + logging.error(f"validating fail === {zip_path}") + return False \ No newline at end of file