diff --git a/utils/HtmlUtils.py b/utils/HtmlUtils.py
index 8bac3b3..a542b81 100644
--- a/utils/HtmlUtils.py
+++ b/utils/HtmlUtils.py
@@ -1,22 +1,45 @@
from fake_useragent import UserAgent
-import requests
+import requests,os
from lxml import html
import traceback
import time
from urllib3.util.retry import Retry
from requests.adapters import HTTPAdapter
from utils.Ntfy import ntfy
+import re
+from utils.comic.PathStr import pathStr
+import json
class htmlUtils:
headers = {'User-Agent': UserAgent().random}
url_data = {}
+
+ @classmethod
+ def getPathSaveHtml(cls,url,type=None):
+ rstr = r"[\/\\\:\*\?\"\<\>\|\.]" # '/ \ : * ? " < > |'
+ file_url = re.sub(rstr, "_", url)
+ file_path = os.path.join(pathStr.base_html_data,file_url)
+ if type == "new":
+ return file_path
+ if os.path.exists(file_path):
+ if type == "read":
+ with open(file_path,"r",encoding="utf-8") as fs:
+ return fs.read()
+ return file_path
+ else:
+ return None
+
+ @classmethod
+ def saveHtml(cls,url,data):
+ file_path = cls.getPathSaveHtml(url,type="new")
+ dir_name = os.path.dirname(file_path)
+ if not os.path.exists(dir_name):
+ os.makedirs(dir_name)
+ with open(file_path,"w",encoding="utf-8") as fs:
+ fs.write(str(data))
@classmethod
def getHTML(cls, curl,type=None):
- rstr = r"[\/\\\:\*\?\"\<\>\|\.]" # '/ \ : * ? " < > |'
- #file_url = re.sub(rstr, "_", curl)
- keys = cls.url_data.keys()
- url_text = cls.url_data.get(curl)
retries = Retry(total=3,
backoff_factor=0.1,
status_forcelist=[ 500, 502, 503, 504 ])
@@ -25,6 +48,9 @@ class htmlUtils:
s.mount('https://', HTTPAdapter(max_retries=retries))
count = 1
#数据为空则获取数据
+ url_text = cls.getPathSaveHtml(curl,"read")
+ if url_text != None:
+ return html.fromstring(url_text)
while url_text == None:
if count <= 10:
try:
@@ -33,8 +59,12 @@ class htmlUtils:
time.sleep(3)
if type == "bytes":
url_text = res
+ if type == "json":
+ url_text = res.json()
else:
url_text = html.fromstring(res.text)
+
+ cls.saveHtml(curl,res.text)
except:
print(f'Retry! 第{count}次')
time.sleep(3)
@@ -45,9 +75,6 @@ class htmlUtils:
ntfy.sendMsg(f"fail 请求失败:{curl}")
exit()
break
- if type == "None":
- data = { curl : url_text}
- cls.url_data.update(data)
return url_text
@classmethod
@@ -55,22 +82,8 @@ class htmlUtils:
return cls.getHTML(url,type="bytes")
@classmethod
- def getJSON(cls,curl):
- count = 0
- if count < 5:
- try:
- res = requests.get(curl, headers=cls.headers, timeout=180)
- time.sleep(1.5)
- data_json = res.json()
- return data_json
- except:
- print(f"请求失败,重试中... {curl}")
- count += 1
- time.sleep(1)
- cls.getJSON(curl)
- else:
- ntfy.sendMsg(f"重试请求失败... {curl}, 已退出")
- exit()
+ def getJSON(cls,url):
+ return cls.getHTML(url,type="json")
@classmethod
def xpathData(cls,c_xpath,url=None,num=None,not_eq=None):
@@ -85,10 +98,10 @@ class htmlUtils:
count = 1
xpaths = et.xpath(c_xpath)
for x in xpaths:
- if not x == not_eq:
+ if x != not_eq:
result.append(x)
count +=1
- if not num == None:
+ if num != None:
try:
result = result[num]
except:
diff --git a/utils/comic/ComicInfo.py b/utils/comic/ComicInfo.py
index 1347ee0..737728a 100644
--- a/utils/comic/ComicInfo.py
+++ b/utils/comic/ComicInfo.py
@@ -168,7 +168,7 @@ class comicInfo():
@classmethod
def getDirComic(cls):
- if not cls.str_comicName == None:
+ if cls.str_comicName != None:
return os.path.join(pathStr.base_comic_img, cls.str_comicName)
else:
print("comicName不存在,退出中")
@@ -176,7 +176,7 @@ class comicInfo():
@classmethod
def getDirComicChapter(cls):
- if not cls.str_comicName == None and not cls.str_chapter == None:
+ if cls.str_comicName != None and cls.str_chapter != None:
return os.path.join(pathStr.base_comic_img,cls.str_comicName,cls.str_chapter)
else:
print("comicName与chapter 不存在,退出中")
diff --git a/utils/comic/PathStr.py b/utils/comic/PathStr.py
index db94199..5e840e3 100644
--- a/utils/comic/PathStr.py
+++ b/utils/comic/PathStr.py
@@ -1,7 +1,11 @@
-import os
-
+import os,datetime
+from time import strftime
class pathStr:
base_comic_out = "COMICOUT"
base_CBZ = os.path.join(base_comic_out,"CBZ")
base_comic_img = os.path.join(base_comic_out,"outputComic")
- base_conf_path = os.path.join(base_comic_out,".conf")
\ No newline at end of file
+ base_conf_path = os.path.join(base_comic_out,".conf")
+
+ date = datetime.datetime.now()
+ date = date.strftime("%Y%m%d")
+ base_html_data = os.path.join(base_comic_out,"html_"+str(date))
\ No newline at end of file
diff --git a/utils/entity/RouMan.py b/utils/entity/RouMan.py
index 441fe41..d15b0ea 100644
--- a/utils/entity/RouMan.py
+++ b/utils/entity/RouMan.py
@@ -25,22 +25,22 @@ class comicEntity:
x = cls.baseComicData(url)
books = x.get("books")
len_books = len(books)
- baseUrl = comicInfo.getBaseUrl(url)
+ base_url = comicInfo.getBaseUrl(url)
for x in range(0, len_books):
book = books[x]
book_id = book.get("id")
book_name = book.get("name")
- updatedAt = book.get("updatedAt")
- comicHref = baseUrl+"/books/"+book_id
+ updated = book.get("updatedAt")
+ comic_href = base_url+"/books/"+book_id
random_int = random.randint(5,20)
comicInfo.setComicName(book_name)
- dirConfComic = comicInfo.getDirConfComic()
- if not os.path.exists(dirConfComic):
+ dir_conf_comic = comicInfo.getDirConfComic()
+ if not os.path.exists(dir_conf_comic):
ntfy.sendMsg(f"{random_int}秒后开始下载 漫画:{book_name}")
time.sleep(random_int)
else:
ntfy.sendMsg(f"已存在 漫画:{book_name}")
- cls.oneComic(comicHref, random.uniform(0,3))
+ cls.oneComic(comic_href, random.uniform(0,3))
print(books)
#for comicHref in comicsHref: