PyComicPackRouMan/utils/HtmlUtils.py
2022-12-20 16:07:14 +08:00

109 lines
3.4 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

from fake_useragent import UserAgent
import requests,os
from lxml import html
import traceback
import time
from urllib3.util.retry import Retry
from requests.adapters import HTTPAdapter
from utils.Ntfy import ntfy
import re
from utils.comic.PathStr import pathStr
import json
class htmlUtils:
headers = {'User-Agent': UserAgent().random}
url_data = {}
@classmethod
def getPathSaveHtml(cls,url,type=None):
rstr = r"[\/\\\:\*\?\"\<\>\|\.]" #  '/ \ : * ? " < > |'
file_url = re.sub(rstr, "_", url)
file_path = os.path.join(pathStr.base_html_data,file_url)
if type == "new":
return file_path
if os.path.exists(file_path):
if type == "read":
with open(file_path,"r",encoding="utf-8") as fs:
return fs.read()
return file_path
else:
return None
@classmethod
def saveHtml(cls,url,data):
file_path = cls.getPathSaveHtml(url,type="new")
dir_name = os.path.dirname(file_path)
if not os.path.exists(dir_name):
os.makedirs(dir_name)
with open(file_path,"w",encoding="utf-8") as fs:
fs.write(str(data))
@classmethod
def getHTML(cls, curl,type=None):
retries = Retry(total=3,
backoff_factor=0.1,
status_forcelist=[ 500, 502, 503, 504 ])
s = requests.Session()
s.mount('http://', HTTPAdapter(max_retries=retries))
s.mount('https://', HTTPAdapter(max_retries=retries))
count = 1
#数据为空则获取数据
url_text = cls.getPathSaveHtml(curl,"read")
if url_text != None:
return html.fromstring(url_text)
while url_text == None:
if count <= 10:
try:
print(f"请求地址:{curl}")
res = s.get(curl,stream=True, headers=cls.headers, timeout=180)
time.sleep(3)
if type == "bytes":
url_text = res
if type == "json":
url_text = res.json()
else:
url_text = html.fromstring(res.text)
cls.saveHtml(curl,res.text)
except:
print(f'Retry! 第{count}')
time.sleep(3)
traceback.print_exc()
count += 1
continue
else:
ntfy.sendMsg(f"fail 请求失败:{curl}")
exit()
break
return url_text
@classmethod
def getBytes(cls, url):
return cls.getHTML(url,type="bytes")
@classmethod
def getJSON(cls,url):
return cls.getHTML(url,type="json")
@classmethod
def xpathData(cls,c_xpath,url=None,num=None,not_eq=None):
if url == None:
url = cls.temp_url
else:
cls.temp_url = url
result = []
#获取html实体数据
et = cls.getHTML(url)
#比对数据
count = 1
xpaths = et.xpath(c_xpath)
for x in xpaths:
if x != not_eq:
result.append(x)
count +=1
if num != None:
try:
result = result[num]
except:
result = None
return result