初始化
This commit is contained in:
104
utils/formats.py
Normal file
104
utils/formats.py
Normal file
@@ -0,0 +1,104 @@
|
||||
import hashlib
|
||||
from bs4 import BeautifulSoup
|
||||
import time
|
||||
import random
|
||||
import datetime
|
||||
import re
|
||||
|
||||
def url_to_spu(url: str, length: int = 36) -> str:
|
||||
h = hashlib.sha256(url.encode('utf-8')).digest()
|
||||
int_val = int.from_bytes(h, 'big')
|
||||
chars = '0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ'
|
||||
base36 = ''
|
||||
while int_val > 0:
|
||||
int_val, idx = divmod(int_val, 36)
|
||||
base36 = chars[idx] + base36
|
||||
base36 = base36.zfill(50)
|
||||
return base36[:length]
|
||||
|
||||
# 工具函数,几乎每个站点的采集都会用到
|
||||
def remove_unwanted_tags(html_content):
|
||||
html_content = str(html_content)
|
||||
soup = BeautifulSoup(html_content, 'html.parser')
|
||||
for tag in soup(['button', 'img', 'a', 'script', 'svg', 'video']):
|
||||
tag.decompose()
|
||||
return str(soup)
|
||||
|
||||
def clean_html(html):
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
|
||||
# 遍历所有标签
|
||||
for tag in soup(['button', 'a', 'script']):
|
||||
tag.attrs = {} # 直接清空所有属性
|
||||
|
||||
return str(soup)
|
||||
|
||||
def de_repeat_urls(filepath):
|
||||
index = {}
|
||||
with open(filepath, 'r', encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
old_data = text.split('\n')
|
||||
with open(f"{filepath}_old.txt", 'w', encoding='utf-8') as f:
|
||||
f.write('\n'.join(old_data))
|
||||
|
||||
new_data = []
|
||||
for line in old_data:
|
||||
line = '#'.join(line.split('#')[:-1])
|
||||
if line in index:
|
||||
continue
|
||||
new_data.append(line)
|
||||
index[line] = None
|
||||
with open(filepath, 'w', encoding='utf-8') as f:
|
||||
f.write('\n'.join(new_data))
|
||||
|
||||
def getTime(timeStamp = False, format: str = '%Y-%m-%d %H:%M:%S'):
|
||||
"""
|
||||
获取时间
|
||||
Author Cerys
|
||||
ChangeTime 2023-11-10
|
||||
|
||||
@param timeStamp 指定时间戳
|
||||
@param fotmat 指定格式
|
||||
|
||||
return str
|
||||
"""
|
||||
|
||||
if timeStamp != False:
|
||||
time = datetime.datetime.fromtimestamp(timeStamp)
|
||||
else:
|
||||
time = datetime.datetime.now()
|
||||
formatTime = time.strftime(format)
|
||||
|
||||
return formatTime
|
||||
|
||||
def formatName(name: str, types: bool = False):
|
||||
"""
|
||||
下划线字符串命名转大驼峰或小驼峰
|
||||
Author Cerys
|
||||
ChangeTime 2023-11-10
|
||||
|
||||
@param name 要转换的字符串
|
||||
@param types 是否转为大驼峰
|
||||
|
||||
return str
|
||||
"""
|
||||
|
||||
strLists = name.split('_')
|
||||
if types:
|
||||
return strLists[0].title() + ''.join(x.title() for x in strLists[1:])
|
||||
else:
|
||||
return strLists[0] + ''.join(x.title() for x in strLists[1:])
|
||||
|
||||
def formatMd5(text: str = '', is_random: bool = False, attach: str = ''):
|
||||
if is_random:
|
||||
text = f"{time.time()}_{random.randint(10000, 99999)}"
|
||||
md5 = hashlib.md5(f"{text}_{attach}".encode()).hexdigest()
|
||||
return md5
|
||||
|
||||
def re_search(text: str, rule: str) -> str:
|
||||
match = re.search(rule, text)
|
||||
if match:
|
||||
url_path = match.group(1)
|
||||
return url_path
|
||||
else:
|
||||
return ""
|
||||
Reference in New Issue
Block a user