245 lines
10 KiB
Python
245 lines
10 KiB
Python
import json,os,time,random,shutil
|
||
from utils.NetUtils import netUtils
|
||
from utils.HtmlUtils import htmlUtils
|
||
from utils.ImageUtils import imageUtils
|
||
from utils.comic.ComicInfo import comicInfo
|
||
from utils.CBZUtils import CBZUtils
|
||
from utils.downloader import download_images
|
||
from utils.Ntfy import ntfy
|
||
from utils.VerUtils import verUtils
|
||
|
||
class comicEntity:
|
||
count_chapter = 0
|
||
|
||
|
||
@classmethod
|
||
def baseComicData(cls,url,update=False):
|
||
data = htmlUtils.xpathData('//script[@id="__NEXT_DATA__"]/text()',url=url,update=update)
|
||
data = json.loads(data[0])
|
||
data = data.get("props")
|
||
x = data.get("pageProps")
|
||
return x
|
||
|
||
@classmethod
|
||
def downladsComcis(cls,url):
|
||
#漫画名
|
||
x = cls.baseComicData(url,update=True)
|
||
books = x.get("books")
|
||
len_books = len(books)
|
||
base_url = comicInfo.getBaseUrl(url)
|
||
for x in range(0, len_books):
|
||
book = books[x]
|
||
book_id = book.get("id")
|
||
book_name = book.get("name")
|
||
updated = book.get("updatedAt")
|
||
comicInfo.setComicName(book_name)
|
||
comicInfo.setUpdateAt(updated)
|
||
comic_href = base_url+"/books/"+book_id
|
||
random_int = random.randint(5,20)
|
||
comicInfo.setComicName(book_name)
|
||
dir_conf_comic = comicInfo.getDirConfComic()
|
||
if not os.path.exists(dir_conf_comic):
|
||
ntfy.sendMsg(f"{random_int}秒后开始下载 漫画:{book_name}")
|
||
time.sleep(random_int)
|
||
else:
|
||
ntfy.sendMsg(f"已存在 漫画:{book_name}")
|
||
if comicInfo.isUpdateComic():
|
||
cls.oneComic(comic_href, random.uniform(0,3))
|
||
comicInfo.updateComicDate()
|
||
else:
|
||
ntfy.sendMsg(f"{book_name} 已是最新")
|
||
|
||
#print(books)
|
||
#for comicHref in comicsHref:
|
||
# cls.oneComic(comicHref,random.uniform(10,20))
|
||
|
||
@classmethod
|
||
def oneComic(cls,c_url,sleep=None):
|
||
#漫画名
|
||
title = htmlUtils.xpathData('//div[@class="col"]/h5/text()',url=c_url,num=0,update=True)
|
||
#别名
|
||
alias = htmlUtils.xpathData('//span[contains(@class,"bookid_alias")]/text()',num=1)
|
||
icon = htmlUtils.xpathData('//img[@class="img-thumbnail"]/@src',num=0)
|
||
|
||
author = htmlUtils.xpathData('//div[contains(@class,"bookid_bookInfo")]/p[1]/text()',num=1)
|
||
tags = htmlUtils.xpathData('//div[contains(@class,"bookid_bookInfo")]/p[3]/b/text()',num=0)
|
||
action = htmlUtils.xpathData('//div[contains(@class,"bookid_bookInfo")]/p[2]/text()',num=1)
|
||
dep = htmlUtils.xpathData('//div[contains(@class,"bookid_bookInfo")]/p[4]/text()',num=1)
|
||
update_date = htmlUtils.xpathData('//div[contains(@class,"bookid_bookInfo")]/p[5]/small/text()',num=1)
|
||
chapters = htmlUtils.xpathData('//div[contains(@class,"bookid_chapterBox")]//div[contains(@class,"bookid_chapter")]/a/text()')
|
||
chapter_href = htmlUtils.xpathData('//div[contains(@class,"bookid_chapterBox")]//div[contains(@class,"bookid_chapter")]/a/@href')
|
||
|
||
author = str(author).replace("&",",").replace(" ",",")
|
||
comicInfo.setHomePage(c_url)
|
||
comicInfo.setComicName(str(title))
|
||
comicInfo.setComicNames(title+","+alias)
|
||
comicInfo.setAuthor(author)
|
||
comicInfo.setIcon(icon)
|
||
comicInfo.setTag(tags)
|
||
comicInfo.setTags(tags)
|
||
comicInfo.setDep(dep)
|
||
comicInfo.setCBS("韩漫")
|
||
comicInfo.setLang("zh")
|
||
comicInfo.setComicNames(title+","+alias)
|
||
comicInfo.setListChapter(chapters)
|
||
|
||
#comicUtils.setComic(title,alias,icon,author,tags,action,dep,update_date,chapters,chapter_href)
|
||
cls.count_chapter = 0
|
||
for href in chapter_href:
|
||
chapter = chapters[cls.count_chapter]
|
||
comicInfo.setChapterName(chapter)
|
||
if not comicInfo.nextExistsGetPath("done_"):
|
||
comicEntity.comicChapter(href,scramble=True,sleep=random.randint(5,15))
|
||
#存在就校验CBZ包是否完整
|
||
if comicInfo.nextExistsGetPath("done_"):
|
||
verUtils.verCBZ()
|
||
cls.count_chapter += 1
|
||
#一本漫画下载后等待
|
||
#清空文件夹
|
||
path_dir_comic = comicInfo.getDirComic()
|
||
if os.path.exists(path_dir_comic):
|
||
shutil.rmtree(path_dir_comic)
|
||
if sleep != None:
|
||
time.sleep(sleep)
|
||
|
||
'''
|
||
|
||
读取某章节下所有图片
|
||
'''
|
||
@classmethod
|
||
def comicChapter(cls,chapter_url,scramble=None,sleep=None):
|
||
is_next = True
|
||
try:
|
||
is_next = cls.Onechapter(chapter_url,scramble)
|
||
#进入下个阶段
|
||
if comicInfo.nextExistsGetPath("down_"):
|
||
#章节图片全部下载后,调用下载封面
|
||
netUtils.downloadComicIcon()
|
||
#下个阶段
|
||
if comicInfo.nextExistsGetPath("cbz_"):
|
||
time.sleep(0.1)
|
||
#下载后自动打包
|
||
is_next = CBZUtils.packAutoComicChapterCBZ()
|
||
#完成删除原文件
|
||
remove_path = comicInfo.getDirComicChapter()
|
||
shutil.rmtree(remove_path)
|
||
print(f"文件已删除: {remove_path}")
|
||
except:
|
||
ntfy.sendMsg(f"{comicInfo.getComicName()} 下载出错了")
|
||
is_next = False
|
||
ntfy.sendMsg(f"预计总章节大小:{cls.count_chapter + 1} / "+ str(comicInfo.getLenChapters()))
|
||
if sleep != None and is_next == True:
|
||
ntfy.sendMsg(f"{sleep} 秒后开始下载下一个章节")
|
||
time.sleep(sleep)
|
||
|
||
|
||
@classmethod
|
||
def Onechapter(cls,chapter_url,scramble=None):
|
||
if not str(chapter_url).startswith("http"):
|
||
chapter_url = comicInfo.getBaseUrl() + chapter_url
|
||
try:
|
||
is_next = cls.comicChapterDownload(chapter_url)
|
||
except:
|
||
htmlUtils.remove_HtmlCache(chapter_url)
|
||
is_next = cls.comicChapterDownload(chapter_url)
|
||
comicInfo.nextInfoToImgChapter()
|
||
#下载完成后, 开始解密图片
|
||
if scramble:
|
||
#获取章节图片路径
|
||
chapter_dir = comicInfo.getDirComicChapter()
|
||
dirs = os.listdir(chapter_dir)
|
||
for img in dirs:
|
||
isScramble = str(img).startswith("scramble=")
|
||
if isScramble:
|
||
c_path = os.path.join(chapter_dir, img)
|
||
#imageUtils.getScrambleImage(c_path)
|
||
imageUtils.encode_scramble_image(c_path)
|
||
#进入下一阶段
|
||
comicInfo.nextImgToDownloadChapter()
|
||
return is_next
|
||
|
||
@classmethod
|
||
def comicChapterDownload(cls,chapter_url):
|
||
x = cls.baseComicData(chapter_url,update=True)
|
||
bookName = x.get("bookName")
|
||
chapterName = x.get("chapterName")
|
||
#fileUtils.saveConfComicChapterInfo(chapterName,x,bookName)
|
||
#if comicInfo.nextExistsGetPath("info_"):
|
||
# print(f"{bookName} {chapterName} info文件已存在跳过")
|
||
alias = x.get("alias")
|
||
description = x.get("description")
|
||
images = x.get("images")
|
||
chapterAPIPath = x.get("chapterAPIPath")
|
||
comicInfo.setComicName(bookName)
|
||
comicInfo.setChapterName(chapterName)
|
||
comicInfo.setDep(description)
|
||
|
||
if not chapterAPIPath == None:
|
||
chapterAPIPath = str(chapterAPIPath).encode('utf-8').decode('unicode_escape')
|
||
base_url = comicInfo.getBaseUrl(chapter_url)
|
||
chapterAPIUrl = base_url+chapterAPIPath
|
||
ntfy.sendMsg(f"chapterApiUrl= {chapterAPIUrl}",alert=False)
|
||
data = htmlUtils.getJSON(chapterAPIUrl)
|
||
if data != None:
|
||
data = data.get("chapter")
|
||
chapterName = data.get("name")
|
||
images = data.get("images")
|
||
if images == None:
|
||
ntfy.sendMsg(f"未获取到章节图像 comic_name={bookName} chapter={chapterName}")
|
||
|
||
tags = x.get("tags")
|
||
x = tags
|
||
count = 1
|
||
list_img = []
|
||
list_scramble = []
|
||
list_fileName = []
|
||
for image in images:
|
||
image_src = image.get("src")
|
||
scramble = image.get("scramble")
|
||
count_image = "{:0>3d}".format(count)
|
||
list_img.append(image_src)
|
||
image_src_prefix = "."+str(image_src).split(".")[-1]
|
||
if scramble:
|
||
su = "."+str(image_src).split(".")[-1]
|
||
de_str = str(image_src).split("/")[-1].replace(su,"==")
|
||
blocks = imageUtils.encodeImage(de_str)
|
||
count_image = "scramble="+str(blocks)+"_"+count_image
|
||
list_fileName.append(count_image+image_src_prefix)
|
||
count+=1
|
||
#print("count_all_img=", count)
|
||
#netUtils.downloadComicChapterImages(list_img,scrambles=list_scramble)
|
||
comicInfo.setChapterImgs(list_img)
|
||
#保存图像
|
||
comicInfo.nextSaveInfoChapter(chapterName, list_img)
|
||
#验证数据是已存在且是否完整
|
||
cbz_path = comicInfo.getDirCBZComicChapter()+".CBZ"
|
||
is_next = True
|
||
if os.path.exists(cbz_path):
|
||
try:
|
||
cbz_size = len(CBZUtils.zip_info(cbz_path)) - 1
|
||
except:
|
||
cbz_size = 0
|
||
if len(list_img) == cbz_size:
|
||
ntfy.sendMsg(f"{bookName} {chapterName} 数据完整,已跳过")
|
||
comicInfo.nextDoneSave(list_img)
|
||
is_next = False
|
||
else:
|
||
ntfy.sendMsg(f"{bookName} {chapterName} 数据不完整,尝试删除配置CBZ文件后重试")
|
||
htmlUtils.remove_HtmlCache(chapter_url)
|
||
try:
|
||
if cbz_size < len(list_img) or os.path.getsize(cbz_path) < 300000:
|
||
ntfy.sendMsg(f"删除 {cbz_path}")
|
||
os.remove(cbz_path)
|
||
else:
|
||
is_next = False
|
||
except:
|
||
ntfy(f"删除失败 {cbz_path}")
|
||
if is_next:
|
||
pathComicInfo = comicInfo.getPathComicInfoXML()
|
||
if not os.path.exists(pathComicInfo):
|
||
#print("不存在ComicInfo.xml 生成中...")
|
||
comicInfo.setPages(list_fileName)
|
||
comicInfo.writeComicInfoXML(chapterName)
|
||
ntfy.sendMsg(f"{bookName} {chapterName} 下载中")
|
||
download_images(list_img,comicInfo.getDirComicChapter(), filesName=list_fileName)
|
||
return is_next |