上传文件至「py」
This commit is contained in:
+350
@@ -0,0 +1,350 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# 本资源来源于互联网公开渠道,仅可用于个人学习爬虫技术。
|
||||
# 严禁将其用于任何商业用途,下载后请于 24 小时内删除,搜索结果均来自源站,本人不承担任何责任。
|
||||
#junyouyun
|
||||
|
||||
import sys
|
||||
import json
|
||||
import time
|
||||
import base64
|
||||
import hashlib
|
||||
import urllib3
|
||||
import concurrent.futures
|
||||
from urllib.parse import quote
|
||||
from base.spider import Spider
|
||||
from Crypto.PublicKey import RSA
|
||||
from Crypto.Cipher import PKCS1_v1_5
|
||||
|
||||
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
||||
sys.path.append('..')
|
||||
|
||||
class Spider(Spider):
|
||||
host, userid, episode_list = '', '', []
|
||||
|
||||
# ---------- 加密与签名相关常量 ----------
|
||||
PUB_KEY_B64 = "MIGfMA0GCSqGSIb3DQEBAQUAA4GNADCBiQKBgQCoYt0BP77U+DM08BiI/QbSRIfxijXo85BTPqIM1Ow8BNwhLETzRIZ+dEwdWDbydG/PspgBAfRpGaYVdJYtvaC2JnoO8+Ik6qMWojfEJxSFLa0Pb0A892tun4gsxoEMjcreZ+YGyaBxAfqX0BSMfdrOgIYaZQjYrw9TRLlUT31QoQIDAQAB"
|
||||
APP_SIGN_SHA1 = "09a8dc51639a31801af5f6418caebfabc695eb24"
|
||||
DEVICE_ID = "2d590b9842d064a1"
|
||||
|
||||
# RSA 私钥(用于解密响应)
|
||||
PRIV_KEY_B64 = """MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCquQQ5r6+yJI8CDFkXRp8vUsdD45ov8EP12ooLs56ca2DQXaSNGS9910bAPVA9chkp0mKIvKqjAsHz5Tl9EeNPblarGEeJUIxpxZtiSqNTpvtiD/TjhpzuHYic7RAfQ/h7p/ypE8ymU42pYjsB5t26Mv6XgkLV+jzrSf73HlCuS0iMyLmt6zz3Mw9izM13EpB8iFLtfbbYymycKTx4RAmPQLwhNGex/AlUIYxXP4R2yyaa4W6mEtc6aME2QuzJFxPgP3HJ9NBx/LWVn4skxWjZ7zg+VRQRHnjyVaSLu3Z5gN5ITWCyE32qaHJa6WBahZj5jWhRyAG1bQ+xKJa8lBL5AgMBAAECggEAUwv9SjJ0PSwbhNuM2w23kcWquROWhYtTA91zGY4esehqB/IFgb2mpIh8Gje5OKqwIu/8jpd4SiOlRYdUF8sD0DfUYRZGdj2AkFNX6tBz8tVfo6wvbB6naA1lzzBij1L5JO3qsjS3cJFkb+kg2yP66AC2Z+0tpfk8eRhdtshAZwfcd1DEGt1uAvYL1eaUK9HRvpt9lPeGcHERDl2hBd4uyaF0K1O+zF9y59nYbTySWPxRZq3sFEE85xRMlstD7YZi7W2gKvMFRD4/FKmrZ3m7aKJRITtyKOyyPcYmepNv3Qv7kk59Pg38n2WWQ0Ra/bCH3E48YNCnQvZMpitkTfJhoQKBgQDbnROOYTP8OTJ6f/qhoGjxeO3x1VOaOp8l0x7b0SCfoqNGS0Cyiqj72BmJtPMPqSTjn6MmNzqbg1KOdhXyzNozs+i5ccW1M56j96mr5I/Z0FpE3oyIHNfDDBlf9M8YQqEF9oYxniYYft9oapO7cRQkHER6qpvnHTavwlv4m78CXwKBgQDHAjs2YlpKDdI1lcbZJCc7TwtH+Pd2bUki8YXafWNcPhITQHbOZjr310eK1QJC6GJncjkOqbX7yv3ivvTO35FZTQhuA1xEG1P00FG8bE0tHYPIwQHi9y0eA5cieMdo8E6XYria1mw/3fqSQEsfZyJlR32JQIoGAipM8iO1X2nZpwKBgDkMFIhnt5lNQk+P7wsNIDWZtDWdtJnboHuy29E+Abt2A/O+mI/IdRz2hau/1WO8DFkUnszOi+rZshhPlGP90rCbi1igtTrcrdjp/KkqNjPea5R4OwkgdOu1uOG0NheXNzzVTQaWjk7Opjn5dWa7eP/oV+GFb/oZHJuLYVizHGsBAoGADA7rjZEKDYCm4w5PPSr+oY5ZjaPdQrS+gLqHtMRyN82fBMGcMUdqfUfzEstzVqCEDeaS5HuOBlK3bXzKkppjUTjksN3NQmcxgBz7RuJ9DqXCLXDcb2cwuafYCYOt+YLOEEgwDVm+t2P44dG5e46hO+fICH/7nP+WlpD5buz4GfMCgYB57r3g/6hi9WUDnfc7ZAzWMqR0EhJVYKYy+KFEtdIPzhkkIHq5RASe88E9kzoGoZFdb3tIjvGZWcHerirrqWkMsuQtP/Qi0zjieid5tAPj+r4kbiCVTw0E0jnmPBzGInQi7lpeTTKnG1fbyS5lBS+WmHfIuzpECgCkxhaT+LJJkg=="""
|
||||
|
||||
headers = {
|
||||
'User-Agent': "okhttp/4.12.0",
|
||||
'Connection': "Keep-Alive",
|
||||
'Accept-Encoding': "gzip",
|
||||
'Content-Type': "application/json;charset=UTF-8",
|
||||
'Cache-Control': "no-cache",
|
||||
'token': "",
|
||||
'deviceId': DEVICE_ID,
|
||||
'client': "app",
|
||||
'deviceType': "Android"
|
||||
}
|
||||
|
||||
# ---------- RSA 加密 ----------
|
||||
def rsa_encrypt(self, data: str) -> str:
|
||||
key = RSA.import_key(base64.b64decode(self.PUB_KEY_B64))
|
||||
cipher = PKCS1_v1_5.new(key)
|
||||
encrypted = cipher.encrypt(data.encode('utf-8'))
|
||||
return base64.b64encode(encrypted).decode('utf-8')
|
||||
|
||||
# ---------- RSA 解密(支持分块) ----------
|
||||
def rsa_decrypt(self, encrypted_b64: str) -> str:
|
||||
key = RSA.import_key(base64.b64decode(self.PRIV_KEY_B64))
|
||||
cipher = PKCS1_v1_5.new(key)
|
||||
encrypted_bytes = base64.b64decode(encrypted_b64)
|
||||
block_size = 256
|
||||
decrypted_parts = []
|
||||
for i in range(0, len(encrypted_bytes), block_size):
|
||||
block = encrypted_bytes[i:i+block_size]
|
||||
decrypted_parts.append(cipher.decrypt(block, None))
|
||||
return b''.join(decrypted_parts).decode('utf-8')
|
||||
|
||||
# ---------- 构建签名参数 ----------
|
||||
def build_params_string(self, episode_id="", episode_index="", vid="", player_id="", type_id="", user_id=""):
|
||||
return (f"episodeId{episode_id}"
|
||||
f"episodeIndex{episode_index}"
|
||||
f"id{vid}"
|
||||
f"playerId{player_id}"
|
||||
f"source0"
|
||||
f"typeId{type_id}"
|
||||
f"userId{user_id}")
|
||||
|
||||
def generate_sign(self, timestamp: str, params_str: str, device_id: str) -> str:
|
||||
raw = f"SaltLSFBTimestamp{timestamp}Params{params_str}ClientappDeviceId{device_id}"
|
||||
b64 = base64.b64encode(raw.encode('utf-8')).decode('utf-8')
|
||||
md5 = hashlib.md5(b64.encode('utf-8')).hexdigest().upper()
|
||||
return md5
|
||||
|
||||
def build_encrypted_headers(self, body_json: str, params_str: str) -> dict:
|
||||
timestamp = str(int(time.time()))
|
||||
encrypted_key = self.rsa_encrypt(body_json)
|
||||
snjm = self.rsa_encrypt("113")
|
||||
appsign = self.rsa_encrypt(self.APP_SIGN_SHA1)
|
||||
sign = self.generate_sign(timestamp, params_str, self.DEVICE_ID)
|
||||
|
||||
headers = {
|
||||
"snjm": snjm,
|
||||
"appsign": appsign,
|
||||
"timestamp": timestamp,
|
||||
"sign": sign,
|
||||
"deviceId": self.DEVICE_ID,
|
||||
"token": self.headers.get('token', ''),
|
||||
"client": "app",
|
||||
"deviceType": "Android",
|
||||
"Content-Type": "application/json;charset=UTF-8",
|
||||
"Cache-Control": "no-cache",
|
||||
"User-Agent": "okhttp/4.12.0"
|
||||
}
|
||||
return headers, {"key": encrypted_key}
|
||||
|
||||
# ---------- 原有接口(保持不变) ----------
|
||||
def init(self, extend=''):
|
||||
self.headers['deviceId'] = self.DEVICE_ID
|
||||
self.host = 'http://qkys.qukanwh.com'
|
||||
response = self.fetch(f'{self.host}/api/v1/app/user/visitorInfo', headers=self.headers).json()
|
||||
self.userid = response['data']['id']
|
||||
token = response['data']['token']
|
||||
self.headers['token'] = token
|
||||
|
||||
def homeContent(self, filter):
|
||||
response = self.post(f'{self.host}/api/v1/app/screen/screenType', headers=self.headers).json()
|
||||
data = response['data']
|
||||
classes = []
|
||||
for i in data:
|
||||
classes.append({'type_id': i['id'], 'type_name': i['name']})
|
||||
return {'class': classes}
|
||||
|
||||
def homeVideoContent(self):
|
||||
response = self.post(f'{self.host}/api/v1/app/recommend/recommendList', headers=self.headers).json()
|
||||
data = response['data']
|
||||
videos = []
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
future_to_id = {
|
||||
executor.submit(
|
||||
self.post,
|
||||
f'{self.host}/api/v1/app/recommend/recommendSubList',
|
||||
data=json.dumps({
|
||||
"condition": item['id'],
|
||||
"pageNum": 1,
|
||||
"pageSize": 6
|
||||
}),
|
||||
headers=self.headers
|
||||
): item['id'] for item in data
|
||||
}
|
||||
for future in concurrent.futures.as_completed(future_to_id):
|
||||
try:
|
||||
response = future.result().json()
|
||||
for video in response['data']['records']:
|
||||
videos.append({
|
||||
"vod_id": video['id'],
|
||||
"vod_name": video['name'],
|
||||
"vod_pic": video['cover']
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"Request failed for item {future_to_id[future]}: {str(e)}")
|
||||
return {'list': videos}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
payload = {
|
||||
"condition": {
|
||||
"classify": "",
|
||||
"region": "",
|
||||
"sreecnTypeEnum": "NEWEST",
|
||||
"typeId": tid,
|
||||
"year": ""
|
||||
},
|
||||
"pageNum": pg,
|
||||
"pageSize": 40
|
||||
}
|
||||
response = self.post(f'{self.host}/api/v1/app/screen/screenMovie', data=json.dumps(payload), headers=self.headers).json()
|
||||
videos = []
|
||||
for i in response['data']['records']:
|
||||
videos.append({
|
||||
"vod_id": i['id'],
|
||||
"vod_name": i['name'],
|
||||
"vod_pic": i['cover'],
|
||||
"vod_remarks": i['area'],
|
||||
"vod_year": i['year']
|
||||
})
|
||||
return {'list': videos, 'page': pg}
|
||||
|
||||
def searchContent(self, key, quick, pg='1'):
|
||||
payload = {
|
||||
"condition": {
|
||||
"value": key
|
||||
},
|
||||
"pageNum": pg,
|
||||
"pageSize": 40
|
||||
}
|
||||
response = self.post(f'{self.host}/api/v1/app/search/searchMovie', data=json.dumps(payload), headers=self.headers).json()
|
||||
videos = []
|
||||
for i in response['data']['records']:
|
||||
videos.append({
|
||||
'vod_id': i['id'],
|
||||
'vod_name': i['name'],
|
||||
'vod_pic': i['cover'],
|
||||
'vod_remarks': i['area'],
|
||||
'vod_year': i['year'],
|
||||
'vod_area': i['area'],
|
||||
'vod_content': i['desc']
|
||||
})
|
||||
return {'list': videos, 'page': pg}
|
||||
|
||||
# ---------- 详情页(已集成解密) ----------
|
||||
def detailContent(self, ids):
|
||||
type_id = "M15" # 注意:原脚本写死为 M17,可根据需要修改
|
||||
vid = ids[0]
|
||||
body = {
|
||||
"id": vid,
|
||||
"source": 0,
|
||||
"typeId": type_id,
|
||||
"userId": self.userid,
|
||||
"episodeId": "",
|
||||
"episodeIndex": "",
|
||||
"playerId": ""
|
||||
}
|
||||
body_json = json.dumps(body, separators=(',', ':'))
|
||||
params_str = self.build_params_string(
|
||||
episode_id="",
|
||||
episode_index="",
|
||||
vid=str(vid),
|
||||
player_id="",
|
||||
type_id=type_id,
|
||||
user_id=str(self.userid)
|
||||
)
|
||||
headers, payload = self.build_encrypted_headers(body_json, params_str)
|
||||
|
||||
# 发送加密请求
|
||||
resp_raw = self.post(f'{self.host}/api/v1/app/play/movieDetails', data=json.dumps(payload), headers=headers).json()
|
||||
encrypted_data = resp_raw.get('data')
|
||||
if not encrypted_data:
|
||||
raise Exception("响应中 data 为空")
|
||||
# 解密 data 字段
|
||||
decrypted_json_str = self.rsa_decrypt(encrypted_data)
|
||||
data = json.loads(decrypted_json_str)
|
||||
|
||||
# 后续处理与原脚本相同
|
||||
currentplayerid = data['playerId']
|
||||
play_urls = []
|
||||
play_url = []
|
||||
show = []
|
||||
for i in data['episodeList']:
|
||||
play_url.append(f"{i['episode']}${ids[0]}@{currentplayerid}@{i['id']}@episode")
|
||||
play_urls.append('#'.join(play_url))
|
||||
moviePlayerList = data['moviePlayerList']
|
||||
for i2 in moviePlayerList:
|
||||
if i2['id'] == currentplayerid:
|
||||
show.append(i2['moviePlayerName'])
|
||||
for j in moviePlayerList:
|
||||
playerid = j['id']
|
||||
episodeTotal = j.get('episodeTotal')
|
||||
if playerid == currentplayerid or episodeTotal is None:
|
||||
continue
|
||||
play_url = []
|
||||
for k in range(1, episodeTotal + 1):
|
||||
play_url.append(f"第{k}集${k}@{playerid}@{ids[0]}@virtual")
|
||||
play_urls.append('#'.join(play_url))
|
||||
if j['moviePlayerName'] not in show:
|
||||
show.append(j['moviePlayerName'])
|
||||
|
||||
# 获取简介(此接口可能无需加密,保持原样)
|
||||
payload_desc = {
|
||||
"id": ids[0],
|
||||
"typeId": type_id
|
||||
}
|
||||
response_desc = self.post(f'{self.host}/api/v1/app/play/movieDesc', data=json.dumps(payload_desc), headers=self.headers).json()
|
||||
data2 = response_desc['data']
|
||||
|
||||
video = {
|
||||
'vod_id': data2['id'],
|
||||
'vod_name': data2['name'],
|
||||
'vod_pic': data2['cover'],
|
||||
'vod_content': data2['introduce'],
|
||||
'vod_year': data2['year'],
|
||||
'vod_area': data2['area'],
|
||||
'vod_remarks': '',
|
||||
'vod_score': data2['score'],
|
||||
'type_name': data2['classify'],
|
||||
'vod_director': data2['director'],
|
||||
'vod_actor': data2['star'],
|
||||
'vod_play_from': '$$$'.join(show),
|
||||
'vod_play_url': '$$$'.join(play_urls)
|
||||
}
|
||||
return {'list': [video]}
|
||||
|
||||
# ---------- 播放页(已集成解密) ----------
|
||||
def playerContent(self, flag, id, vipflags):
|
||||
param, playerid, param2, param3 = id.split('@')
|
||||
if param3 == 'virtual':
|
||||
payload = {
|
||||
"episodeIndex": str(int(param) - 1),
|
||||
"id": int(param2),
|
||||
"playerId": playerid,
|
||||
"source": 0,
|
||||
"typeId": "M15",
|
||||
"userId": self.userid,
|
||||
"episodeId": ""
|
||||
}
|
||||
else:
|
||||
payload = {
|
||||
"episodeId": param2,
|
||||
"id": int(param),
|
||||
"playerId": playerid,
|
||||
"source": 0,
|
||||
"typeId": "M15",
|
||||
"userId": self.userid,
|
||||
"episodeIndex": ""
|
||||
}
|
||||
body_json = json.dumps(payload, separators=(',', ':'))
|
||||
print(body_json)
|
||||
params_str = self.build_params_string(
|
||||
episode_id=payload.get("episodeId", ""),
|
||||
episode_index=payload.get("episodeIndex", ""),
|
||||
vid=str(payload["id"]),
|
||||
player_id=payload["playerId"],
|
||||
type_id=payload["typeId"],
|
||||
user_id=str(payload["userId"])
|
||||
)
|
||||
print(params_str)
|
||||
headers, encrypted_payload = self.build_encrypted_headers(body_json, params_str)
|
||||
print(headers)
|
||||
print(encrypted_payload)
|
||||
# 获取播放信息(加密响应)
|
||||
resp_raw = self.post(f'{self.host}/api/v1/app/play/movieDetails', data=json.dumps(encrypted_payload), headers=headers).json()
|
||||
encrypted_data = resp_raw.get('data')
|
||||
if not encrypted_data:
|
||||
raise Exception("响应中 data 为空")
|
||||
decrypted_json_str = self.rsa_decrypt(encrypted_data)
|
||||
data = json.loads(decrypted_json_str)
|
||||
print(data)
|
||||
parse_url = data['url']
|
||||
playerid = data['playerId']
|
||||
|
||||
# 调用分析接口(注:analysisMovieUrl 的响应可能也是加密的,但原脚本直接取 data,这里暂不做额外解密)
|
||||
analysis_body = {
|
||||
"playerUrl": parse_url,
|
||||
"playerId": playerid
|
||||
}
|
||||
analysis_json = json.dumps(analysis_body, separators=(',', ':'))
|
||||
# analysisMovieUrl 接口的参数拼接?理论上也需要签名,但原脚本是 GET 方式,为了兼容,我们沿用原脚本的 GET 方式
|
||||
# 原脚本使用 fetch GET 带参数,并未加密。这里也采用 GET 方式,不使用加密 headers
|
||||
resp_analysis = self.fetch(f"{self.host}/api/v1/app/play/analysisMovieUrl?playerUrl={quote(parse_url,safe='')}&playerId={playerid}", headers=self.headers).json()
|
||||
url = resp_analysis.get('data')
|
||||
|
||||
return {'jx': '0', 'parse': '0', 'url': url, 'header': {'User-Agent': 'Mozilla/5.0 (iPhone; CPU iPhone OS 13_2_3 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/13.0.3 Mobile/15E148 Safari/604.1'}}
|
||||
|
||||
def getName(self):
|
||||
pass
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
pass
|
||||
|
||||
def manualVideoCheck(self):
|
||||
pass
|
||||
|
||||
def destroy(self):
|
||||
pass
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
+545
@@ -0,0 +1,545 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
悟空影视爬虫
|
||||
站点: https://www.yikucun.com
|
||||
基于 MacCMS (苹果CMS) 程序,HTML 解析方式
|
||||
OK影视 / 海阔视界 兼容版
|
||||
"""
|
||||
|
||||
import re
|
||||
import time
|
||||
import json
|
||||
import urllib.parse
|
||||
|
||||
import requests
|
||||
|
||||
try:
|
||||
from base.spider import Spider as BaseSpider
|
||||
except ImportError:
|
||||
class BaseSpider:
|
||||
pass
|
||||
|
||||
|
||||
class Spider(BaseSpider):
|
||||
BASE_URL = "https://www.yikucun.com"
|
||||
|
||||
# 分类映射
|
||||
TYPE_MAP = {
|
||||
"1": "电影",
|
||||
"2": "电视剧",
|
||||
"3": "综艺",
|
||||
"4": "动漫",
|
||||
"5": "短剧",
|
||||
}
|
||||
|
||||
# 筛选条件 - 各分类的类型和地区
|
||||
FILTERS = {
|
||||
"1": { # 电影
|
||||
"类型": ["全部", "动作", "喜剧", "爱情", "科幻", "恐怖", "剧情", "战争", "犯罪", "奇幻", "悬疑", "动画", "恐怖", "纪录片", "其他"],
|
||||
"地区": ["全部", "大陆", "香港", "台湾", "日本", "韩国", "美国", "法国", "英国", "德国", "泰国", "印度", "其他"],
|
||||
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015", "2014", "2013", "2012", "2011", "2010"],
|
||||
},
|
||||
"2": { # 电视剧
|
||||
"类型": ["全部", "古装", "战争", "青春偶像", "喜剧", "家庭", "犯罪", "动作", "奇幻", "剧情", "历史", "经典", "乡村", "情景", "商战", "网剧", "其他"],
|
||||
"地区": ["全部", "内地", "韩国", "香港", "台湾", "日本", "美国", "泰国", "英国", "新加坡", "其他"],
|
||||
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015", "2014", "2013", "2012", "2011", "2010"],
|
||||
},
|
||||
"3": { # 综艺
|
||||
"类型": ["全部", "真人秀", "脱口秀", "访谈", "美食", "旅游", "选秀", "情感", "音乐", "舞蹈", "其他"],
|
||||
"地区": ["全部", "大陆", "香港", "台湾", "日本", "韩国", "欧美", "其他"],
|
||||
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015"],
|
||||
},
|
||||
"4": { # 动漫
|
||||
"类型": ["全部", "日本动漫", "国产动漫", "欧美动漫", "其他"],
|
||||
"地区": ["全部", "日本", "大陆", "美国", "其他"],
|
||||
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015"],
|
||||
},
|
||||
"5": { # 短剧
|
||||
"类型": ["全部", "其他"],
|
||||
"地区": ["全部", "其他"],
|
||||
"年份": ["全部", "2026", "2025", "2024", "2023"],
|
||||
},
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.name = ""
|
||||
self.error_play_url = "https://kjjsaas-sh.oss-cn-shanghai.aliyuncs.com/u/3401405881/20240818-936952-fc31b16575e80a7562cdb1f81a39c6b0.mp4"
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update({
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
"Referer": "https://www.yikucun.com/",
|
||||
"Connection": "keep-alive",
|
||||
})
|
||||
self._init_cookies()
|
||||
|
||||
def getName(self):
|
||||
return self.name
|
||||
|
||||
def init(self, extend="{}"):
|
||||
try:
|
||||
self.extend = json.loads(extend)
|
||||
self.name = self.extend.get("name", "")
|
||||
except Exception as e:
|
||||
print(e)
|
||||
self.extend = {}
|
||||
|
||||
def _init_cookies(self):
|
||||
"""初始化 cookies(处理 508 反爬)"""
|
||||
try:
|
||||
r = self.session.get(self.BASE_URL, timeout=10, verify=False)
|
||||
if r.status_code == 508:
|
||||
time.sleep(1)
|
||||
self.session.get(self.BASE_URL, timeout=10, verify=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _get(self, url, **kwargs):
|
||||
"""GET 请求,自动处理 508"""
|
||||
try:
|
||||
r = self.session.get(url, timeout=15, verify=False, **kwargs)
|
||||
if r.status_code == 508:
|
||||
time.sleep(0.5)
|
||||
r = self.session.get(url, timeout=15, verify=False, **kwargs)
|
||||
r.encoding = "utf-8"
|
||||
return r
|
||||
except Exception as e:
|
||||
class FakeResp:
|
||||
status_code = 500
|
||||
text = str(e)
|
||||
return FakeResp()
|
||||
|
||||
def homeContent(self, filter):
|
||||
"""首页 - 返回分类和推荐"""
|
||||
result = {
|
||||
"class": [],
|
||||
"filters": {},
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
# 分类列表
|
||||
for tid, name in self.TYPE_MAP.items():
|
||||
result["class"].append({
|
||||
"type_id": tid,
|
||||
"type_name": name,
|
||||
})
|
||||
|
||||
# 筛选条件
|
||||
for tid, flist in self.FILTERS.items():
|
||||
result["filters"][tid] = []
|
||||
for fname, fvalues in flist.items():
|
||||
result["filters"][tid].append({
|
||||
"key": fname,
|
||||
"name": fname,
|
||||
"value": [{"n": v, "v": v} for v in fvalues],
|
||||
})
|
||||
|
||||
# 首页推荐
|
||||
try:
|
||||
r = self._get(f"{self.BASE_URL}/")
|
||||
if r.status_code == 200:
|
||||
result["list"] = self._parse_list(r.text)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
"""分类页"""
|
||||
result = {
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
# URL 格式: 12个位置,用 - 分隔
|
||||
# 位置: 1=tid, 2=地区, 3=空, 4=类型, 5=空, 6=空, 7=空, 8=空, 9=页码, 10=空, 11=空, 12=年份
|
||||
area = extend.get("地区", "") if isinstance(extend, dict) else ""
|
||||
type_val = extend.get("类型", "") if isinstance(extend, dict) else ""
|
||||
year = extend.get("年份", "") if isinstance(extend, dict) else ""
|
||||
|
||||
if area == "全部":
|
||||
area = ""
|
||||
if type_val == "全部":
|
||||
type_val = ""
|
||||
if year == "全部":
|
||||
year = ""
|
||||
|
||||
area_enc = urllib.parse.quote(area) if area else ""
|
||||
type_enc = urllib.parse.quote(type_val) if type_val else ""
|
||||
|
||||
# 12个位置
|
||||
parts = [
|
||||
tid, # 1
|
||||
area_enc, # 2
|
||||
"", # 3
|
||||
type_enc, # 4
|
||||
"", # 5
|
||||
"", # 6
|
||||
"", # 7
|
||||
"", # 8
|
||||
str(pg), # 9 页码
|
||||
"", # 10
|
||||
"", # 11
|
||||
year, # 12 年份
|
||||
]
|
||||
|
||||
url = f"{self.BASE_URL}/ucusw/{'-'.join(parts)}.html"
|
||||
|
||||
try:
|
||||
r = self._get(url)
|
||||
if r.status_code == 200:
|
||||
result["list"] = self._parse_list(r.text)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
def detailContent(self, ids):
|
||||
"""详情页"""
|
||||
result = {
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
vid = ids[0]
|
||||
url = f"{self.BASE_URL}/ucudt/{vid}.html"
|
||||
|
||||
try:
|
||||
r = self._get(url)
|
||||
if r.status_code == 200:
|
||||
detail = self._parse_detail(r.text, vid)
|
||||
if detail:
|
||||
result["list"].append(detail)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
"""搜索"""
|
||||
result = {
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
try:
|
||||
# 搜索 URL: /ucusc/-------------.html?wd=关键词
|
||||
# 分页: /ucusc/-------------(页码).html?wd=关键词
|
||||
if int(pg) > 1:
|
||||
search_url = f"{self.BASE_URL}/ucusc/-------------{pg}.html?wd={urllib.parse.quote(key)}"
|
||||
else:
|
||||
search_url = f"{self.BASE_URL}/ucusc/-------------.html?wd={urllib.parse.quote(key)}"
|
||||
r = self._get(search_url)
|
||||
if r.status_code == 200:
|
||||
result["list"] = self._parse_list(r.text)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
"""播放页 - 获取播放地址"""
|
||||
result = {
|
||||
"url": self.error_play_url,
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
"header": {},
|
||||
}
|
||||
|
||||
try:
|
||||
# id 格式: 视频ID-线路ID-集数ID
|
||||
parts = id.split("-")
|
||||
if len(parts) >= 3:
|
||||
vid, sid, nid = parts[0], parts[1], parts[2]
|
||||
play_url = f"{self.BASE_URL}/ucupy/{vid}-{sid}-{nid}.html"
|
||||
r = self._get(play_url)
|
||||
if r.status_code == 200:
|
||||
url = self._parse_play_url(r.text)
|
||||
if url:
|
||||
result["url"] = url
|
||||
result["parse"] = 0
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return result
|
||||
|
||||
def localProxy(self, params):
|
||||
return 0
|
||||
|
||||
def _parse_list(self, html):
|
||||
"""解析列表页"""
|
||||
items = []
|
||||
|
||||
# 找到所有 class="name" 的标题,然后往前找最近的图片
|
||||
name_pattern = r'class="name"[^>]*>\s*<a[^>]*href="/ucudt/(\d+)\.html"[^>]*>([^<]+)</a>'
|
||||
|
||||
# 先找出所有 name 的位置
|
||||
name_matches = list(re.finditer(name_pattern, html))
|
||||
|
||||
seen = set()
|
||||
for i, m in enumerate(name_matches):
|
||||
vid = m.group(1)
|
||||
name = m.group(2).strip()
|
||||
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
|
||||
# 往前找最近的图片(在这个 name 之前)
|
||||
pos = m.start()
|
||||
start = max(0, pos - 2000)
|
||||
segment = html[start:pos]
|
||||
|
||||
# 找所有图片
|
||||
img_matches = re.findall(r'<img[^>]*src="([^"]+)"', segment)
|
||||
pic = img_matches[-1] if img_matches else ""
|
||||
|
||||
if pic and not pic.startswith("http"):
|
||||
pic = "https:" + pic if pic.startswith("//") else pic
|
||||
|
||||
# 找备注
|
||||
remarks = ""
|
||||
rgba_match = re.search(r'class="rgba[^"]*"[^>]*>([^<]+)</span>', segment)
|
||||
if rgba_match:
|
||||
remarks = rgba_match.group(1).strip()
|
||||
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": name,
|
||||
"vod_pic": pic,
|
||||
"vod_remarks": remarks,
|
||||
})
|
||||
|
||||
return items[:30]
|
||||
|
||||
def _parse_detail(self, html, vid):
|
||||
"""解析详情页"""
|
||||
detail = {
|
||||
"vod_id": vid,
|
||||
"vod_name": "",
|
||||
"vod_pic": "",
|
||||
"type_name": "",
|
||||
"vod_year": "",
|
||||
"vod_area": "",
|
||||
"vod_remarks": "",
|
||||
"vod_actor": "",
|
||||
"vod_director": "",
|
||||
"vod_content": "",
|
||||
"vod_play_from": "",
|
||||
"vod_play_url": "",
|
||||
}
|
||||
|
||||
# 标题
|
||||
title_match = re.search(r'<title>《([^》]+)》', html)
|
||||
if title_match:
|
||||
detail["vod_name"] = title_match.group(1).strip()
|
||||
|
||||
# 图片
|
||||
pic_match = re.search(r'og:image"[^>]*content="([^"]+)"', html)
|
||||
if pic_match:
|
||||
pic = pic_match.group(1)
|
||||
if not pic.startswith("http"):
|
||||
pic = "https:" + pic if pic.startswith("//") else pic
|
||||
detail["vod_pic"] = pic
|
||||
|
||||
# 主演
|
||||
actor_match = re.search(r'<dt>主演[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if actor_match:
|
||||
actors = re.sub(r'<[^>]+>', '', actor_match.group(1)).strip()
|
||||
actors = actors.replace(' ', ' ').replace('\xa0', ' ')
|
||||
detail["vod_actor"] = actors
|
||||
|
||||
# 导演
|
||||
dir_match = re.search(r'<dt>导演[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if dir_match:
|
||||
directors = re.sub(r'<[^>]+>', '', dir_match.group(1)).strip()
|
||||
directors = directors.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||
# 隐晦水印
|
||||
import base64
|
||||
wm = base64.b64decode(b'5pif5rKz').decode('utf-8')
|
||||
if directors:
|
||||
directors = directors + ' ' + wm
|
||||
else:
|
||||
directors = wm
|
||||
detail["vod_director"] = directors
|
||||
|
||||
# 类型
|
||||
type_match = re.search(r'<dt>类型[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if type_match:
|
||||
type_name = re.sub(r'<[^>]+>', '', type_match.group(1)).strip()
|
||||
type_name = type_name.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||
detail["type_name"] = type_name
|
||||
|
||||
# 地区
|
||||
area_match = re.search(r'<dt>地区[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if area_match:
|
||||
area = re.sub(r'<[^>]+>', '', area_match.group(1)).strip()
|
||||
area = area.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||
detail["vod_area"] = area
|
||||
|
||||
# 年代
|
||||
year_match = re.search(r'<dt>年代[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if year_match:
|
||||
year = re.sub(r'<[^>]+>', '', year_match.group(1)).strip()
|
||||
year = year.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||
detail["vod_year"] = year
|
||||
|
||||
# 备注/状态
|
||||
remark_match = re.search(r'<dt>备注[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if remark_match:
|
||||
detail["vod_remarks"] = re.sub(r'<[^>]+>', '', remark_match.group(1)).strip()
|
||||
|
||||
# 简介
|
||||
desc_match = re.search(r'<dt>剧情[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||
if desc_match:
|
||||
content = re.sub(r'<[^>]+>', '', desc_match.group(1)).strip()
|
||||
content = content.replace("详细", "").strip()
|
||||
detail["vod_content"] = content
|
||||
|
||||
# 播放源和播放地址
|
||||
# 1. 从 tab2 提取线路 id -> 名称 映射 (按顺序)
|
||||
source_order = []
|
||||
tab_match = re.search(r'class="tab2">(.*?)</dt>', html, re.DOTALL)
|
||||
if tab_match:
|
||||
tab_html = tab_match.group(1)
|
||||
source_pattern = r'<span[^>]*id="([^"]+)"[^>]*>([^<]+)</span>'
|
||||
source_matches = re.findall(source_pattern, tab_html)
|
||||
for sid, sname in source_matches:
|
||||
source_order.append((sid, sname.strip()))
|
||||
|
||||
# 2. 定位到 content 区域
|
||||
source_eps = {}
|
||||
content_match = re.search(r'<div[^>]*id="content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
|
||||
if content_match:
|
||||
content = content_match.group(1)
|
||||
|
||||
# 用 <dd 分割,逐个处理
|
||||
dd_parts = re.split(r'<dd\s+', content)
|
||||
for part in dd_parts[1:]:
|
||||
class_match = re.search(r'class="([^"]+)"', part)
|
||||
if not class_match:
|
||||
continue
|
||||
dd_class = class_match.group(1)
|
||||
|
||||
if 'mod' not in dd_class:
|
||||
continue
|
||||
|
||||
source_key = dd_class.replace('mod', '').strip()
|
||||
if not source_key:
|
||||
continue
|
||||
|
||||
# 找这个 dd 里的所有集数
|
||||
ep_pattern = r'href="/ucupy/(\d+)-(\d+)-(\d+)\.html"[^>]*>([^<]+)<'
|
||||
ep_matches = re.findall(ep_pattern, part)
|
||||
|
||||
if ep_matches:
|
||||
eps = []
|
||||
for evid, esid, enid, ename in ep_matches:
|
||||
eps.append(f"{ename.strip()}${evid}-{esid}-{enid}")
|
||||
source_eps[source_key] = "#".join(eps)
|
||||
|
||||
# 3. 按 source_order 的顺序组装结果
|
||||
play_from_list = []
|
||||
play_url_list = []
|
||||
for source_key, source_name in source_order:
|
||||
if source_key in source_eps:
|
||||
play_from_list.append(source_name)
|
||||
play_url_list.append(source_eps[source_key])
|
||||
|
||||
if play_from_list:
|
||||
detail["vod_play_from"] = "$$$".join(play_from_list)
|
||||
detail["vod_play_url"] = "$$$".join(play_url_list)
|
||||
|
||||
# 如果没找到播放列表,用默认线路名
|
||||
if not detail["vod_play_from"]:
|
||||
detail["vod_play_from"] = "速播大屏"
|
||||
detail["vod_play_url"] = f"第01集${vid}-1-1"
|
||||
|
||||
return detail
|
||||
|
||||
def _parse_play_url(self, html):
|
||||
"""解析播放地址"""
|
||||
# 从 player_aaaa 变量中提取
|
||||
idx = html.find('player_aaaa=')
|
||||
if idx >= 0:
|
||||
eq_idx = html.find('=', idx)
|
||||
if eq_idx >= 0:
|
||||
end_idx = html.find('</script>', eq_idx)
|
||||
if end_idx > 0:
|
||||
json_str = html[eq_idx+1:end_idx].strip()
|
||||
if json_str.endswith(';'):
|
||||
json_str = json_str[:-1].strip()
|
||||
try:
|
||||
data = json.loads(json_str)
|
||||
url = data.get("url", "")
|
||||
if url:
|
||||
return url
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 备用:直接找 m3u8
|
||||
url_match = re.search(r'"url"\s*:\s*"(https?://[^"]+\.m3u8[^"]*)"', html)
|
||||
if url_match:
|
||||
url = url_match.group(1).replace("\\/", "/").replace("\\", "")
|
||||
return url
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def main():
|
||||
"""测试用"""
|
||||
spider = Spider()
|
||||
spider.init('{}')
|
||||
|
||||
print("=== homeContent 测试 ===")
|
||||
result = spider.homeContent({})
|
||||
print(f"分类数: {len(result['class'])}")
|
||||
print(f"首页推荐: {len(result['list'])} 个")
|
||||
for item in result['list'][:3]:
|
||||
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||
|
||||
print()
|
||||
print("=== categoryContent 测试 (电视剧) ===")
|
||||
result = spider.categoryContent("2", "1", "", {})
|
||||
print(f"结果数: {len(result['list'])} 个")
|
||||
for item in result['list'][:3]:
|
||||
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||
|
||||
print()
|
||||
print("=== searchContent 测试 (千香) ===")
|
||||
result = spider.searchContent("千香", False, "1")
|
||||
print(f"结果数: {len(result['list'])} 个")
|
||||
for item in result['list'][:3]:
|
||||
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||
|
||||
if result["list"]:
|
||||
vid = result["list"][0]["vod_id"]
|
||||
print(f"\n=== detailContent 测试 ({vid}) ===")
|
||||
detail_result = spider.detailContent([vid])
|
||||
if detail_result["list"]:
|
||||
d = detail_result["list"][0]
|
||||
print(f" 标题: {d['vod_name']}")
|
||||
print(f" 主演: {d['vod_actor'][:50]}...")
|
||||
print(f" 线路: {d['vod_play_from']}")
|
||||
sources = d['vod_play_from'].split('$$$')
|
||||
print(f" 线路数: {len(sources)}")
|
||||
|
||||
# 播放测试
|
||||
urls = d['vod_play_url'].split('$$$')
|
||||
first_ep = urls[0].split('#')[0]
|
||||
ep_id = first_ep.split('$')[1] if '$' in first_ep else first_ep
|
||||
print(f"\n=== playerContent 测试 ({ep_id}) ===")
|
||||
play_result = spider.playerContent(sources[0], ep_id, [])
|
||||
print(f" parse: {play_result['parse']}")
|
||||
print(f" url: {play_result['url'][:80] if play_result['url'] else '无'}...")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+608
@@ -0,0 +1,608 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
import re
|
||||
import json
|
||||
import hashlib
|
||||
import base64
|
||||
import time
|
||||
from urllib.parse import quote
|
||||
from base.spider import Spider
|
||||
|
||||
# ---------- 纯Python实现的RC4和AES(无需第三方库) ----------
|
||||
def rc4_crypt(data, key):
|
||||
S = list(range(256))
|
||||
j = 0
|
||||
for i in range(256):
|
||||
j = (j + S[i] + key[i % len(key)]) % 256
|
||||
S[i], S[j] = S[j], S[i]
|
||||
i = j = 0
|
||||
out = bytearray()
|
||||
for ch in data:
|
||||
i = (i + 1) % 256
|
||||
j = (j + S[i]) % 256
|
||||
S[i], S[j] = S[j], S[i]
|
||||
out.append(ch ^ S[(S[i] + S[j]) % 256])
|
||||
return out
|
||||
|
||||
def aes_cbc_decrypt(data, key, iv):
|
||||
raise NotImplementedError("AES解密需要 pycryptodome 库")
|
||||
|
||||
# 尝试导入官方库
|
||||
try:
|
||||
from Crypto.Cipher import ARC4, AES
|
||||
from Crypto.Util.Padding import unpad
|
||||
def rc4_crypt(data, key):
|
||||
cipher = ARC4.new(key)
|
||||
return cipher.decrypt(data)
|
||||
def aes_cbc_decrypt(data, key, iv):
|
||||
cipher = AES.new(key.encode('utf-8'), AES.MODE_CBC, iv.encode('utf-8'))
|
||||
decrypted = unpad(cipher.decrypt(base64.b64decode(data)), AES.block_size)
|
||||
return decrypted.decode('utf-8')
|
||||
CRYPTO_AVAILABLE = True
|
||||
except ImportError:
|
||||
CRYPTO_AVAILABLE = False
|
||||
print("[歪比影视] 提示: pycryptodome未安装,播放解密可能失败")
|
||||
|
||||
class Spider(Spider):
|
||||
BASE_URL = "https://wbbb1.com"
|
||||
PARSE_DOMAIN = "xn--qvr2v.850088.xyz"
|
||||
|
||||
def_headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/148.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9",
|
||||
"Referer": BASE_URL + "/",
|
||||
"Connection": "keep-alive",
|
||||
}
|
||||
|
||||
play_headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
||||
"Referer": BASE_URL + "/",
|
||||
"Accept": "*/*",
|
||||
}
|
||||
|
||||
CATEGORY_MAP = {"1": "1", "2": "2", "3": "3", "4": "4"}
|
||||
CATEGORY_NAMES = {"1": "电影", "2": "剧集", "3": "动漫", "4": "综艺"}
|
||||
|
||||
_cookies = ""
|
||||
|
||||
# ---------- 加密工具 ----------
|
||||
def _md5(self, s):
|
||||
return hashlib.md5(s.encode('utf-8')).hexdigest()
|
||||
|
||||
def _rc4_encrypt(self, data, key):
|
||||
key_bytes = key.encode('utf-8') if isinstance(key, str) else key
|
||||
data_bytes = data.encode('utf-8') if isinstance(data, str) else data
|
||||
encrypted = rc4_crypt(data_bytes, key_bytes)
|
||||
return base64.b64encode(encrypted).decode('utf-8')
|
||||
|
||||
def _rc4_decrypt(self, data, key):
|
||||
key_bytes = key.encode('utf-8') if isinstance(key, str) else key
|
||||
data_bytes = base64.b64decode(data)
|
||||
decrypted = rc4_crypt(data_bytes, key_bytes)
|
||||
return decrypted.decode('utf-8')
|
||||
|
||||
def _aes_decrypt(self, data, key, iv):
|
||||
if not CRYPTO_AVAILABLE:
|
||||
raise Exception("AES解密需要 pycryptodome 库")
|
||||
return aes_cbc_decrypt(data, key, iv)
|
||||
|
||||
# ---------- 页面请求 ----------
|
||||
def _fetch_cookies(self):
|
||||
try:
|
||||
headers = {"User-Agent": self.def_headers["User-Agent"], "Accept": "text/html", "Referer": self.BASE_URL + "/"}
|
||||
resp = self.fetch(self.BASE_URL, headers=headers)
|
||||
cookie_list = []
|
||||
# 尝试多种方式获取Set-Cookie
|
||||
if hasattr(resp, 'cookies') and resp.cookies:
|
||||
try:
|
||||
for cookie in resp.cookies:
|
||||
cookie_list.append(f"{cookie.name}={cookie.value}")
|
||||
except:
|
||||
pass
|
||||
if not cookie_list:
|
||||
if hasattr(resp.headers, "get_all"):
|
||||
raw = resp.headers.get_all("Set-Cookie")
|
||||
for c in raw:
|
||||
cookie_list.append(c.split(";")[0])
|
||||
elif "Set-Cookie" in resp.headers:
|
||||
raw = resp.headers["Set-Cookie"]
|
||||
if isinstance(raw, list):
|
||||
for c in raw:
|
||||
cookie_list.append(c.split(";")[0])
|
||||
else:
|
||||
cookie_list.append(raw.split(";")[0])
|
||||
self._cookies = "; ".join(cookie_list)
|
||||
if self._cookies:
|
||||
print(f"[歪比影视] Cookie获取成功")
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] Cookie获取失败: {e}")
|
||||
self._cookies = ""
|
||||
|
||||
def _fetch_html(self, url):
|
||||
headers = self.def_headers.copy()
|
||||
if self._cookies:
|
||||
headers["Cookie"] = self._cookies
|
||||
try:
|
||||
resp = self.fetch(url, headers=headers)
|
||||
return resp.text
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] 请求失败: {url}, {e}")
|
||||
# 如果失败且没有cookie,尝试重新获取cookie后重试
|
||||
if not self._cookies:
|
||||
print("[歪比影视] 尝试重新获取Cookie...")
|
||||
self._fetch_cookies()
|
||||
if self._cookies:
|
||||
headers["Cookie"] = self._cookies
|
||||
try:
|
||||
resp = self.fetch(url, headers=headers)
|
||||
return resp.text
|
||||
except Exception as e2:
|
||||
print(f"[歪比影视] 重试失败: {e2}")
|
||||
return ""
|
||||
|
||||
# ---------- 工具函数:清理HTML标签 ----------
|
||||
def _clean_html(self, text):
|
||||
if not text:
|
||||
return ""
|
||||
text = re.sub(r'<[^>]+>', '', text)
|
||||
text = re.sub(r'\s+', ' ', text).strip()
|
||||
return text
|
||||
|
||||
# ---------- 解析函数 ----------
|
||||
def _parse_video_list(self, html):
|
||||
"""通用解析(用于首页/分类页)"""
|
||||
videos = []
|
||||
# 使用更宽松的正则匹配,兼容class顺序变化
|
||||
pattern = r'<a[^>]*href="/detail/(\d+\.html)"[^>]*class="[^"]*module-poster-item[^"]*"[^>]*>.*?<div[^>]*class="[^"]*module-item-note[^"]*"[^>]*>([^<]*)</div>.*?<img[^>]*data-original="([^"]+)"[^>]*>.*?<div[^>]*class="[^"]*module-poster-item-title[^"]*"[^>]*>([^<]*)</div>'
|
||||
for match in re.finditer(pattern, html, re.DOTALL):
|
||||
detail_url = match.group(1)
|
||||
vod_id = detail_url.replace(".html", "")
|
||||
vod_pic = match.group(3).strip()
|
||||
# 处理协议相对URL
|
||||
if vod_pic.startswith("//"):
|
||||
vod_pic = "https:" + vod_pic
|
||||
videos.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": match.group(4).strip(),
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": match.group(2).strip(),
|
||||
})
|
||||
return videos
|
||||
|
||||
def _parse_search_list(self, html):
|
||||
"""专门解析搜索页(参考JS选择器)"""
|
||||
videos = []
|
||||
# 使用更宽松的正则匹配
|
||||
items = re.finditer(r'<div[^>]*class="[^"]*module-card-item[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*module-card-item|$)', html, re.DOTALL)
|
||||
for item in items:
|
||||
block = item.group(1)
|
||||
# 提取链接
|
||||
link_match = re.search(r'<a[^>]*href="(/detail/[^"]+\.html)"', block)
|
||||
if not link_match:
|
||||
continue
|
||||
vod_id = link_match.group(1).replace("/detail/", "").replace(".html", "")
|
||||
# 提取标题
|
||||
title_match = re.search(r'<div[^>]*class="[^"]*module-card-item-title[^"]*"[^>]*>.*?<strong>([^<]*)</strong>', block, re.DOTALL)
|
||||
vod_name = title_match.group(1).strip() if title_match else "未知"
|
||||
# 提取图片
|
||||
pic_match = re.search(r'<img[^>]*data-original="([^"]+)"', block)
|
||||
vod_pic = pic_match.group(1) if pic_match else ""
|
||||
if vod_pic and vod_pic.startswith("//"):
|
||||
vod_pic = "https:" + vod_pic
|
||||
# 提取备注
|
||||
note_match = re.search(r'<div[^>]*class="[^"]*module-item-note[^"]*"[^>]*>([^<]*)</div>', block)
|
||||
vod_remarks = note_match.group(1).strip() if note_match else ""
|
||||
videos.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": vod_remarks,
|
||||
})
|
||||
return videos
|
||||
|
||||
def _parse_play_sources(self, html, vod_id):
|
||||
"""
|
||||
提取每个线路的集数,线路名称准确提取
|
||||
"""
|
||||
sources = []
|
||||
|
||||
# 1. 提取线路名称(按顺序)- 优先使用 data-dropdown-value
|
||||
source_names = []
|
||||
for m in re.finditer(r'data-dropdown-value="([^"]+)"', html):
|
||||
name = m.group(1).strip()
|
||||
if name and name not in source_names:
|
||||
source_names.append(name)
|
||||
if not source_names:
|
||||
tab_box = re.search(r'<div[^>]*class="[^"]*module-tab-items-box[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||
if tab_box:
|
||||
for m in re.finditer(r'<div[^>]*class="[^"]*tab-item[^"]*"[^>]*>.*?<span[^>]*>([^<]*)</span>', tab_box.group(1)):
|
||||
name = m.group(1).strip()
|
||||
if name and name not in source_names:
|
||||
source_names.append(name)
|
||||
if not source_names:
|
||||
for m in re.finditer(r'<span[^>]*class="[^"]*module-tab-value"[^>]*>([^<]*)</span>', html):
|
||||
name = m.group(1).strip()
|
||||
if name and name not in source_names:
|
||||
source_names.append(name)
|
||||
if not source_names:
|
||||
source_names = ["默认"]
|
||||
|
||||
# 2. 提取所有 module-list 块
|
||||
list_blocks = []
|
||||
start_tag = '<div class="module-list sort-list tab-list his-tab-list" id="panel1">'
|
||||
pos = 0
|
||||
while True:
|
||||
start = html.find(start_tag, pos)
|
||||
if start == -1:
|
||||
break
|
||||
start += len(start_tag)
|
||||
depth = 0
|
||||
end = None
|
||||
i = start
|
||||
while i < len(html):
|
||||
if html[i:i+5] == '<div ':
|
||||
depth += 1
|
||||
i += 5
|
||||
elif html[i:i+6] == '</div>':
|
||||
if depth == 0:
|
||||
end = i + 6
|
||||
break
|
||||
else:
|
||||
depth -= 1
|
||||
i += 6
|
||||
else:
|
||||
i += 1
|
||||
if end is not None:
|
||||
block_html = html[start:end]
|
||||
if '/vplay/' in block_html:
|
||||
list_blocks.append(block_html)
|
||||
pos = end
|
||||
else:
|
||||
pos = start + 1
|
||||
|
||||
if not list_blocks:
|
||||
blocks = []
|
||||
start_tag2 = '<div class="module-play-list"'
|
||||
pos = 0
|
||||
while True:
|
||||
start = html.find(start_tag2, pos)
|
||||
if start == -1:
|
||||
break
|
||||
start += len(start_tag2)
|
||||
depth = 0
|
||||
end = None
|
||||
i = start
|
||||
while i < len(html):
|
||||
if html[i:i+5] == '<div ':
|
||||
depth += 1
|
||||
i += 5
|
||||
elif html[i:i+6] == '</div>':
|
||||
if depth == 0:
|
||||
end = i + 6
|
||||
break
|
||||
else:
|
||||
depth -= 1
|
||||
i += 6
|
||||
else:
|
||||
i += 1
|
||||
if end is not None:
|
||||
block_html = html[start:end]
|
||||
if '/vplay/' in block_html:
|
||||
list_blocks.append(block_html)
|
||||
pos = end
|
||||
else:
|
||||
pos = start + 1
|
||||
|
||||
if not list_blocks:
|
||||
simple = re.findall(r'<div class="module-play-list"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||
for b in simple:
|
||||
if '/vplay/' in b:
|
||||
list_blocks.append(b)
|
||||
|
||||
# 3. 对齐名称与块数量
|
||||
if len(source_names) > len(list_blocks):
|
||||
source_names = source_names[:len(list_blocks)]
|
||||
while len(source_names) < len(list_blocks):
|
||||
source_names.append(f"源{len(source_names)+1}")
|
||||
|
||||
# 4. 解析每个块的集数
|
||||
for idx, block in enumerate(list_blocks):
|
||||
eps = []
|
||||
for m in re.finditer(r'<a[^>]*href="(/vplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>.*?<span>([^<]*)</span>', block):
|
||||
link = m.group(1)
|
||||
id_ = m.group(2)
|
||||
sid = m.group(3)
|
||||
nid = m.group(4)
|
||||
name = m.group(5).strip()
|
||||
eps.append({"name": name, "link": f"{id_}-{sid}-{nid}"})
|
||||
if eps:
|
||||
sources.append({
|
||||
"source_name": source_names[idx] if idx < len(source_names) else f"源{idx+1}",
|
||||
"episodes": eps
|
||||
})
|
||||
|
||||
if not sources:
|
||||
eps = []
|
||||
for m in re.finditer(r'<a[^>]*href="(/vplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>.*?<span>([^<]*)</span>', html):
|
||||
link = m.group(1)
|
||||
id_ = m.group(2)
|
||||
sid = m.group(3)
|
||||
nid = m.group(4)
|
||||
name = m.group(5).strip()
|
||||
eps.append({"name": name, "link": f"{id_}-{sid}-{nid}"})
|
||||
if eps:
|
||||
sources.append({"source_name": "默认", "episodes": eps})
|
||||
|
||||
return sources
|
||||
|
||||
# ---------- 播放地址获取 ----------
|
||||
def _get_play_url(self, vod_id, sid, nid):
|
||||
try:
|
||||
play_page = f"{self.BASE_URL}/vplay/{vod_id}-{sid}-{nid}.html"
|
||||
html = self._fetch_html(play_page)
|
||||
if not html:
|
||||
return None
|
||||
|
||||
# 检查iframe
|
||||
iframe_match = re.search(r'<iframe[^>]*src="([^"]+)"', html)
|
||||
if iframe_match:
|
||||
iframe_url = iframe_match.group(1)
|
||||
if iframe_url.startswith('//'):
|
||||
iframe_url = 'https:' + iframe_url
|
||||
elif iframe_url.startswith('/'):
|
||||
iframe_url = self.BASE_URL + iframe_url
|
||||
print(f"[歪比影视] 发现iframe: {iframe_url}")
|
||||
return iframe_url
|
||||
|
||||
# 提取加密URL - 使用更安全的大括号匹配算法,支持嵌套JSON
|
||||
enc_url = None
|
||||
m1 = re.search(r'var\s+player_aaaa\s*=\s*', html)
|
||||
if m1:
|
||||
start = m1.end()
|
||||
# 跳过空白字符
|
||||
while start < len(html) and html[start] in ' \t\n\r':
|
||||
start += 1
|
||||
# 找到匹配的大括号
|
||||
if start < len(html) and html[start] == '{':
|
||||
depth = 1
|
||||
i = start + 1
|
||||
while i < len(html) and depth > 0:
|
||||
if html[i] == '{':
|
||||
depth += 1
|
||||
elif html[i] == '}':
|
||||
depth -= 1
|
||||
i += 1
|
||||
if depth == 0:
|
||||
json_str = html[start:i]
|
||||
try:
|
||||
player_data = json.loads(json_str)
|
||||
enc_url = player_data.get("url", "")
|
||||
print(f"[歪比影视] 提取到player_aaaa.url: {enc_url[:50]}...")
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] player_aaaa JSON解析失败: {e}")
|
||||
|
||||
# 兜底:直接搜索url字段
|
||||
if not enc_url:
|
||||
m2 = re.search(r'"url"\s*:\s*"([^"]+)"', html)
|
||||
if m2:
|
||||
enc_url = m2.group(1)
|
||||
print(f"[歪比影视] 兜底提取到url: {enc_url[:50]}...")
|
||||
|
||||
if not enc_url:
|
||||
print("[歪比影视] 未能提取到加密URL")
|
||||
return None
|
||||
|
||||
# 如果无加密库,返回解析页面
|
||||
if not CRYPTO_AVAILABLE:
|
||||
fallback_url = f"https://{self.PARSE_DOMAIN}/player/?url={enc_url}&next=//&title="
|
||||
print(f"[歪比影视] 无加密库,使用解析页面: {fallback_url}")
|
||||
return fallback_url
|
||||
|
||||
try:
|
||||
domain = self.PARSE_DOMAIN
|
||||
l = (self._md5(enc_url) + " P")[-22:]
|
||||
key = l.encode('utf-8')
|
||||
h = self._rc4_encrypt(self._md5(enc_url + "stray"), key)
|
||||
timestamp = str(int(time.time()))
|
||||
u = self._rc4_encrypt(timestamp + self._md5(key.decode('utf-8') + "stray"), key)
|
||||
y = self._rc4_encrypt(self._md5(domain + "stray"), key)
|
||||
|
||||
api_url = f"https://{domain}/player/api.php"
|
||||
headers = {
|
||||
"User-Agent": self.play_headers["User-Agent"],
|
||||
"Accept": "application/json, text/javascript, */*; q=0.01",
|
||||
"Origin": f"https://{domain}",
|
||||
"Referer": f"https://{domain}/player/?url={enc_url}",
|
||||
"X-Requested-With": "XMLHttpRequest",
|
||||
"Content-Type": "application/x-www-form-urlencoded"
|
||||
}
|
||||
if self._cookies:
|
||||
headers["Cookie"] = self._cookies
|
||||
|
||||
post_data = {"url": enc_url, "key": h, "vkey": u, "ckey": y}
|
||||
print(f"[歪比影视] 请求API: {api_url}")
|
||||
resp = self.post(api_url, data=post_data, headers=headers)
|
||||
result = json.loads(resp.text)
|
||||
print(f"[歪比影视] API返回: code={result.get('code')}")
|
||||
|
||||
if result.get("code") == 200:
|
||||
aes_key = self._rc4_decrypt(result["aes_key"], key)
|
||||
aes_iv = self._rc4_decrypt(result["aes_iv"], key)
|
||||
enc_play = result["url"]
|
||||
play_url = self._aes_decrypt(enc_play, aes_key, aes_iv)
|
||||
print(f"[歪比影视] 解密成功: {play_url[:80]}...")
|
||||
return play_url
|
||||
else:
|
||||
fallback_url = f"https://{domain}/player/?url={enc_url}&next=//&title="
|
||||
print(f"[歪比影视] API返回非200,使用解析页面: {fallback_url}")
|
||||
return fallback_url
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] 解密异常: {e}")
|
||||
fallback_url = f"https://{self.PARSE_DOMAIN}/player/?url={enc_url}&next=//&title="
|
||||
return fallback_url
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] 获取播放地址异常: {e}")
|
||||
return None
|
||||
|
||||
# ---------- TVBox接口 ----------
|
||||
def init(self, extend=''):
|
||||
self._fetch_cookies()
|
||||
print("[歪比影视] 初始化完成")
|
||||
|
||||
def homeContent(self, filter):
|
||||
return {"class": [{"type_id": tid, "type_name": name} for tid, name in self.CATEGORY_NAMES.items()]}
|
||||
|
||||
def homeVideoContent(self):
|
||||
try:
|
||||
html = self._fetch_html(self.BASE_URL)
|
||||
if not html:
|
||||
return {"list": []}
|
||||
# 尝试匹配"正在热映"
|
||||
block = re.search(r'<div class="module">.*?<h2[^>]*class="[^"]*module-title[^"]*"[^>]*>正在热映.*?</div>(.*?)</div>\s*<div class="module">', html, re.DOTALL)
|
||||
if not block:
|
||||
# 兜底:尝试匹配第一个module块
|
||||
block = re.search(r'<div class="module">(.*?)</div>\s*<div class="module">', html, re.DOTALL)
|
||||
if not block:
|
||||
return {"list": []}
|
||||
videos = self._parse_video_list(block.group(1))
|
||||
return {"list": videos[:20]}
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] homeVideoContent 异常: {e}")
|
||||
return {"list": []}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
try:
|
||||
pg = int(pg)
|
||||
if tid not in self.CATEGORY_MAP:
|
||||
return {"list": [], "pagecount": 1, "page": pg}
|
||||
if pg == 1:
|
||||
url = f"{self.BASE_URL}/show/{tid}-----------.html"
|
||||
else:
|
||||
url = f"{self.BASE_URL}/show/{tid}--------{pg}---.html"
|
||||
print(f"[歪比影视] 分类请求: {url}")
|
||||
html = self._fetch_html(url)
|
||||
if not html:
|
||||
return {"list": [], "pagecount": 1, "page": pg}
|
||||
videos = self._parse_video_list(html)
|
||||
last = re.search(r'<a[^>]*href="/show/\d+--------(\d+)---\.html"[^>]*>尾页</a>', html)
|
||||
pagecount = int(last.group(1)) if last else 1
|
||||
return {"list": videos, "pagecount": pagecount, "page": pg}
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] categoryContent 异常: {e}")
|
||||
return {"list": [], "pagecount": 1, "page": pg}
|
||||
|
||||
def searchContent(self, key, quick, pg='1'):
|
||||
try:
|
||||
pg = int(pg)
|
||||
# 对搜索关键词进行URL编码,修复中文搜索失败
|
||||
encoded_key = quote(key)
|
||||
url = f"{self.BASE_URL}/search/{encoded_key}-------------.html"
|
||||
print(f"[歪比影视] 搜索请求: {url}")
|
||||
html = self._fetch_html(url)
|
||||
if not html:
|
||||
return {"list": [], "page": pg}
|
||||
videos = self._parse_search_list(html)
|
||||
return {"list": videos, "page": pg}
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] searchContent 异常: {e}")
|
||||
return {"list": [], "page": pg}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
vod_id = ids[0]
|
||||
url = f"{self.BASE_URL}/detail/{vod_id}.html"
|
||||
print(f"[歪比影视] 详情请求: {url}")
|
||||
html = self._fetch_html(url)
|
||||
if not html:
|
||||
return {"list": []}
|
||||
|
||||
title = re.search(r'<h1>([^<]*)</h1>', html)
|
||||
vod_name = title.group(1).strip() if title else "未知"
|
||||
|
||||
# 更宽松的图片匹配,兼容class变化
|
||||
pic = re.search(r'<div[^>]*class="[^"]*module-item-pic[^"]*"[^>]*>.*?<img[^>]*data-original="([^"]+)"', html, re.DOTALL)
|
||||
vod_pic = pic.group(1) if pic else ""
|
||||
if vod_pic.startswith("//"):
|
||||
vod_pic = "https:" + vod_pic
|
||||
|
||||
desc = re.search(r'<div[^>]*class="[^"]*module-info-introduction-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||
vod_content = self._clean_html(desc.group(1)) if desc else ""
|
||||
|
||||
actor = re.search(r'主演:</span>.*?<div[^>]*class="[^"]*module-info-item-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||
vod_actor = self._clean_html(actor.group(1)) if actor else ""
|
||||
|
||||
director = re.search(r'导演:</span>.*?<div[^>]*class="[^"]*module-info-item-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||
vod_director = self._clean_html(director.group(1)) if director else ""
|
||||
|
||||
year = re.search(r'<a[^>]*title="(\d{4})"', html)
|
||||
vod_year = year.group(1) if year else ""
|
||||
|
||||
sources = self._parse_play_sources(html, vod_id)
|
||||
if not sources:
|
||||
print(f"[歪比影视] 未能解析到播放源")
|
||||
return {"list": []}
|
||||
|
||||
from_list = []
|
||||
url_list = []
|
||||
for src in sources:
|
||||
from_list.append(src["source_name"])
|
||||
eps_str = "#".join([f"{ep['name']}${ep['link']}" for ep in src["episodes"]])
|
||||
url_list.append(eps_str)
|
||||
|
||||
vod_play_from = "$$$".join(from_list)
|
||||
vod_play_url = "$$$".join(url_list)
|
||||
|
||||
video = {
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_year": vod_year,
|
||||
"vod_area": "",
|
||||
"vod_actor": vod_actor,
|
||||
"vod_director": vod_director,
|
||||
"vod_content": vod_content,
|
||||
"vod_play_from": vod_play_from,
|
||||
"vod_play_url": vod_play_url,
|
||||
}
|
||||
print(f"[歪比影视] 详情解析成功: {vod_name}, 线路: {vod_play_from}")
|
||||
return {"list": [video]}
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] detailContent 异常: {e}")
|
||||
return {"list": []}
|
||||
|
||||
def playerContent(self, flag, vid, vip_flags):
|
||||
try:
|
||||
parts = vid.split("-")
|
||||
if len(parts) != 3:
|
||||
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||
vod_id, sid, nid = parts
|
||||
play_url = self._get_play_url(vod_id, sid, nid)
|
||||
if play_url:
|
||||
# 判断URL类型,决定parse标志
|
||||
# 如果是直接的视频文件或m3u8,parse=0;否则parse=1(需要TVBox嗅探/解析)
|
||||
is_direct = any(ext in play_url.lower() for ext in ['.m3u8', '.mp4', '.flv', '.ts', '.mkv', '.avi', '.mov', '.wmv'])
|
||||
is_direct = is_direct or 'm3u8' in play_url.lower() or 'mp4' in play_url.lower()
|
||||
parse_flag = 0 if is_direct else 1
|
||||
# header必须是JSON字符串
|
||||
header_str = json.dumps(self.play_headers)
|
||||
print(f"[歪比影视] 播放URL: {play_url[:80]}..., parse={parse_flag}")
|
||||
return {"jx": 0, "parse": parse_flag, "url": play_url, "header": header_str}
|
||||
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||
except Exception as e:
|
||||
print(f"[歪比影视] playerContent 异常: {e}")
|
||||
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||
|
||||
def getName(self):
|
||||
return "歪比影视"
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
pass
|
||||
|
||||
def manualVideoCheck(self):
|
||||
pass
|
||||
|
||||
def destroy(self):
|
||||
pass
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
@@ -0,0 +1,385 @@
|
||||
# coding = utf-8
|
||||
#!/usr/bin/python
|
||||
import re
|
||||
import sys
|
||||
import json
|
||||
import time
|
||||
import urllib.parse
|
||||
from base.spider import Spider
|
||||
|
||||
sys.path.append('..')
|
||||
|
||||
class Spider(Spider):
|
||||
def __init__(self):
|
||||
self.name = "糖豆广场舞"
|
||||
self.host = 'https://api-h5.tangdou.com'
|
||||
self.img_host = 'https://bimg.tangdou.com' # 图片域名前缀
|
||||
self.header = {
|
||||
'Accept': 'application/json, text/plain, */*',
|
||||
'Accept-Encoding': 'gzip, deflate, br',
|
||||
'Accept-Language': 'zh,zh-CN;q=0.9',
|
||||
'Connection': 'keep-alive',
|
||||
'Host': 'api-h5.tangdou.com',
|
||||
'Referer': 'https://www.tangdou.com/',
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
|
||||
}
|
||||
# 缓存机制
|
||||
self.cache = {}
|
||||
self.cache_timeout = 300 # 5分钟缓存
|
||||
# 生成UUID (时间戳_随机数格式)
|
||||
self.uuid = f"{int(time.time() * 1000)}_{int(time.time() % 100000)}"
|
||||
|
||||
def getName(self):
|
||||
return self.name
|
||||
|
||||
def init(self, extend=''):
|
||||
pass
|
||||
|
||||
def homeContent(self, filter):
|
||||
result = {}
|
||||
# 糖豆广场舞分类
|
||||
classes = [
|
||||
|
||||
{"type_name": "广场舞", "type_id": "1"},
|
||||
{"type_name": "民族舞", "type_id": "2"},
|
||||
{"type_name": "jazz/现代舞", "type_id": "3"},
|
||||
{"type_name": "健身", "type_id": "4"},
|
||||
{"type_name": "双人舞", "type_id": "5"},
|
||||
{"type_name": "步法", "type_id": "6"},
|
||||
{"type_name": "气球", "type_id": "7"},
|
||||
{"type_name": "瑜伽", "type_id": "8"},
|
||||
{"type_name": "二人转", "type_id": "9"}
|
||||
]
|
||||
|
||||
result['class'] = classes
|
||||
|
||||
# 筛选条件 - 主要按年份和难度筛选
|
||||
filters = {}
|
||||
for cate in classes:
|
||||
tid = cate['type_id']
|
||||
filters[tid] = [
|
||||
{"key": "year", "name": "年份", "value": [
|
||||
{"n": "全部", "v": "0"},
|
||||
{"n": "2026", "v": "2026"},
|
||||
{"n": "2025", "v": "2025"},
|
||||
{"n": "2024", "v": "2024"},
|
||||
{"n": "2023", "v": "2023"},
|
||||
{"n": "2022", "v": "2022"},
|
||||
{"n": "2021", "v": "2021"},
|
||||
{"n": "2020", "v": "2020"},
|
||||
{"n": "2019", "v": "2019"},
|
||||
{"n": "2018", "v": "2018"},
|
||||
{"n": "2017及以前", "v": "2017"}
|
||||
]},
|
||||
{"key": "sort", "name": "排序", "value": [
|
||||
{"n": "最新", "v": "new"},
|
||||
{"n": "最热", "v": "hot"},
|
||||
{"n": "推荐", "v": "rec"}
|
||||
]}
|
||||
]
|
||||
|
||||
result['filters'] = filters
|
||||
return result
|
||||
|
||||
def homeVideoContent(self):
|
||||
# 首页推荐 - 获取feed流
|
||||
videos = []
|
||||
try:
|
||||
cache_key = "home_feed"
|
||||
data = self.get_cached_data(cache_key, 1, 20)
|
||||
|
||||
if data and 'data' in data:
|
||||
for item in data['data']:
|
||||
video = self._parse_video_item(item)
|
||||
if video:
|
||||
videos.append(video)
|
||||
except Exception as e:
|
||||
print(f"获取首页推荐失败: {e}")
|
||||
|
||||
return {'list': videos}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
videos = []
|
||||
try:
|
||||
# 构建请求参数
|
||||
page_size = 30
|
||||
# 糖豆API: type 0=推荐, 1=广场舞, etc.
|
||||
api_url = f"{self.host}/mtangdou/home/feed?page={pg}&num={page_size}&uuid={self.uuid}"
|
||||
|
||||
# 如果有分类ID且不是推荐,添加分类参数
|
||||
if tid != "0":
|
||||
api_url += f"&type={tid}"
|
||||
|
||||
# 排序参数
|
||||
sort = extend.get('sort', 'new')
|
||||
if sort == 'hot':
|
||||
api_url += "&sort=hot"
|
||||
elif sort == 'rec':
|
||||
api_url += "&sort=rec"
|
||||
|
||||
cache_key = f"category_{tid}_{pg}_{sort}"
|
||||
data = self.fetchData(api_url, cache_key)
|
||||
|
||||
if data and 'data' in data:
|
||||
for item in data['data']:
|
||||
video = self._parse_video_item(item)
|
||||
if video:
|
||||
videos.append(video)
|
||||
except Exception as e:
|
||||
print(f"获取分类内容失败: {e}")
|
||||
|
||||
return {
|
||||
'list': videos,
|
||||
'page': int(pg),
|
||||
'pagecount': 9999, # 糖豆没有明确页数限制
|
||||
'limit': 30,
|
||||
'total': 999999
|
||||
}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
vid = ids[0].split('||')[0] if '||' in ids[0] else ids[0]
|
||||
|
||||
# 获取视频详情和播放链接
|
||||
play_url_api = f"{self.host}/mtangdou/video/play?vid={vid}&uuid={self.uuid}"
|
||||
share_api = f"{self.host}/sample/share/main?vid={vid}"
|
||||
|
||||
# 尝试获取播放链接
|
||||
play_data = self.fetchData(play_url_api, f"play_{vid}", use_cache=False)
|
||||
|
||||
# 获取详情信息
|
||||
share_data = self.fetchData(share_api, f"share_{vid}", use_cache=False)
|
||||
|
||||
if not share_data or 'data' not in share_data:
|
||||
return {'list': []}
|
||||
|
||||
data_info = share_data['data']
|
||||
|
||||
# 获取简介内容,优先使用API返回的desc或description,否则使用默认简介
|
||||
content = data_info.get('desc', data_info.get('description', '')).strip()
|
||||
if not content:
|
||||
content = '醉卧东风祝您身体健康'
|
||||
|
||||
# 修正:拼接完整缩略图URL
|
||||
cover_path = data_info.get('cover', data_info.get('img', ''))
|
||||
if cover_path and not cover_path.startswith('http'):
|
||||
cover_url = self.img_host + cover_path
|
||||
else:
|
||||
cover_url = cover_path
|
||||
|
||||
# 构建视频详情对象
|
||||
video_detail = {
|
||||
"vod_id": vid,
|
||||
"vod_name": data_info.get('title', '').strip(),
|
||||
"vod_pic": cover_url,
|
||||
"vod_year": str(data_info.get('year', '')),
|
||||
"vod_area": data_info.get('area', '大陆'),
|
||||
"vod_actor": data_info.get('teacher', data_info.get('author', '')),
|
||||
"vod_director": "",
|
||||
"vod_content": content,
|
||||
"vod_play_from": "糖豆播放",
|
||||
"vod_remarks": f"时长: {data_info.get('duration_str', '未知')}" if 'duration_str' in data_info else ""
|
||||
}
|
||||
|
||||
# 构建播放链接 - 糖豆是单视频,没有多集
|
||||
play_url = ""
|
||||
if play_data and 'data' in play_data:
|
||||
play_url = play_data['data'].get('play_url', '')
|
||||
|
||||
# 尝试从share接口获取video_url作为备选
|
||||
if not play_url and 'video_url' in data_info:
|
||||
play_url = data_info['video_url']
|
||||
|
||||
if play_url:
|
||||
# 糖豆视频直接播放,需要处理Referer
|
||||
video_detail["vod_play_url"] = f"{video_detail['vod_name']}${vid}||{play_url}"
|
||||
else:
|
||||
video_detail["vod_play_url"] = ""
|
||||
|
||||
return {'list': [video_detail]}
|
||||
|
||||
except Exception as e:
|
||||
print(f"获取详情失败: {e}")
|
||||
return {'list': []}
|
||||
|
||||
def searchContent(self, key, quick, pg=1):
|
||||
videos = []
|
||||
try:
|
||||
# 糖豆搜索API
|
||||
search_api = f"{self.host}/mtangdou/search?word={urllib.parse.quote(key)}&page={pg}&num=30&uuid={self.uuid}"
|
||||
|
||||
# 搜索不使用缓存,确保实时性
|
||||
data = self.fetchData(search_api, use_cache=False)
|
||||
|
||||
if data and 'data' in data:
|
||||
for item in data['data']:
|
||||
video = self._parse_video_item(item)
|
||||
if video:
|
||||
videos.append(video)
|
||||
else:
|
||||
# 如果搜索API返回空,尝试从首页feed中过滤(仅第一页)
|
||||
if pg == 1:
|
||||
feed_data = self.get_cached_data("search_feed", 1, 100)
|
||||
if feed_data and 'data' in feed_data:
|
||||
key_lower = key.lower()
|
||||
for item in feed_data['data']:
|
||||
title = item.get('title', '').lower()
|
||||
if key_lower in title:
|
||||
video = self._parse_video_item(item)
|
||||
if video:
|
||||
videos.append(video)
|
||||
except Exception as e:
|
||||
print(f"搜索失败: {e}")
|
||||
|
||||
return {
|
||||
'list': videos,
|
||||
'page': int(pg),
|
||||
'pagecount': 9999,
|
||||
'limit': 30,
|
||||
'total': 999999
|
||||
}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
try:
|
||||
# 解析传入的id: vid||play_url
|
||||
if '||' in id:
|
||||
parts = id.split('||')
|
||||
vid = parts[0]
|
||||
play_url = parts[1] if len(parts) > 1 else ""
|
||||
else:
|
||||
vid = id
|
||||
play_url = ""
|
||||
|
||||
# 如果没有play_url,实时获取
|
||||
if not play_url:
|
||||
play_api = f"{self.host}/mtangdou/video/play?vid={vid}&uuid={self.uuid}"
|
||||
data = self.fetchData(play_api, use_cache=False)
|
||||
if data and 'data' in data:
|
||||
play_url = data['data'].get('play_url', '')
|
||||
|
||||
if play_url:
|
||||
# 糖豆视频需要Referer才能正常播放
|
||||
headers = {
|
||||
"Referer": "https://www.tangdou.com/",
|
||||
"User-Agent": self.header['User-Agent']
|
||||
}
|
||||
return {
|
||||
"parse": 0, # 直接播放,不需要解析
|
||||
"playUrl": "",
|
||||
"url": play_url,
|
||||
"header": json.dumps(headers)
|
||||
}
|
||||
else:
|
||||
return {"parse": 0, "playUrl": "", "url": ""}
|
||||
|
||||
except Exception as e:
|
||||
print(f"播放解析失败: {e}")
|
||||
return {"parse": 0, "playUrl": "", "url": ""}
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
video_formats = ['.m3u8', '.mp4', '.avi', '.mkv', '.flv', '.ts', '.mov']
|
||||
return any(url.lower().endswith(fmt) for fmt in video_formats)
|
||||
|
||||
def manualVideoCheck(self):
|
||||
pass
|
||||
|
||||
def localProxy(self, params):
|
||||
return None
|
||||
|
||||
def _parse_video_item(self, item):
|
||||
"""解析视频列表项为统一格式"""
|
||||
try:
|
||||
vid = str(item.get('vid', ''))
|
||||
if not vid:
|
||||
return None
|
||||
|
||||
title = item.get('title', '').strip()
|
||||
|
||||
# 修正:拼接完整缩略图URL,cover是相对路径,需要加域名前缀
|
||||
cover_path = item.get('cover', item.get('img', item.get('video_img', '')))
|
||||
if cover_path and not cover_path.startswith('http'):
|
||||
img = self.img_host + cover_path
|
||||
else:
|
||||
img = cover_path
|
||||
|
||||
duration = item.get('duration_str', '')
|
||||
teacher = item.get('teacher', item.get('author', ''))
|
||||
|
||||
# 是否有附属信息(如老师名字)
|
||||
remarks = duration if duration else teacher
|
||||
|
||||
return {
|
||||
"vod_id": vid,
|
||||
"vod_name": title,
|
||||
"vod_pic": img,
|
||||
"vod_remarks": remarks
|
||||
}
|
||||
except Exception as e:
|
||||
print(f"解析视频项失败: {e}")
|
||||
return None
|
||||
|
||||
def get_cached_data(self, cache_key, page=1, num=30):
|
||||
"""获取首页feed缓存"""
|
||||
current_time = time.time()
|
||||
if cache_key in self.cache:
|
||||
cached_data, timestamp = self.cache[cache_key]
|
||||
if current_time - timestamp < self.cache_timeout:
|
||||
return cached_data
|
||||
|
||||
# 缓存不存在或已过期,重新获取
|
||||
api_url = f"{self.host}/mtangdou/home/feed?page={page}&num={num}&uuid={self.uuid}"
|
||||
result = self.fetchData(api_url, cache_key)
|
||||
if result:
|
||||
self.cache[cache_key] = (result, current_time)
|
||||
return result
|
||||
|
||||
def fetchData(self, url, cache_key=None, use_cache=True):
|
||||
"""封装的数据获取方法,支持缓存"""
|
||||
try:
|
||||
# 如果启用缓存且存在有效缓存
|
||||
current_time = time.time()
|
||||
if use_cache and cache_key and cache_key in self.cache:
|
||||
cached_data, timestamp = self.cache[cache_key]
|
||||
if current_time - timestamp < self.cache_timeout:
|
||||
return cached_data
|
||||
|
||||
start_time = time.time()
|
||||
response = self.fetch(url, headers=self.header)
|
||||
end_time = time.time()
|
||||
|
||||
print(f"请求耗时: {end_time - start_time:.2f}秒, URL: {url[:60]}...")
|
||||
|
||||
if response.status_code != 200:
|
||||
print(f"API请求失败: {response.status_code}")
|
||||
return None
|
||||
|
||||
data = json.loads(response.text)
|
||||
|
||||
# 缓存结果
|
||||
if use_cache and cache_key:
|
||||
self.cache[cache_key] = (data, current_time)
|
||||
|
||||
return data
|
||||
|
||||
except Exception as e:
|
||||
print(f"获取数据失败: {e}, URL: {url}")
|
||||
return None
|
||||
|
||||
def fetch(self, url, headers=None):
|
||||
"""发送HTTP GET请求"""
|
||||
import requests
|
||||
try:
|
||||
if headers is None:
|
||||
headers = self.header
|
||||
response = requests.get(url, headers=headers, timeout=10)
|
||||
return response
|
||||
except Exception as e:
|
||||
print(f"请求异常: {e}")
|
||||
# 返回一个模拟的response对象
|
||||
class FakeResponse:
|
||||
status_code = 0
|
||||
text = ""
|
||||
return FakeResponse()
|
||||
|
||||
if __name__ == '__main__':
|
||||
pass
|
||||
+336
@@ -0,0 +1,336 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# by @PyramidStore AutoGen
|
||||
import re
|
||||
import sys
|
||||
sys.path.append('..')
|
||||
import json
|
||||
from urllib.parse import quote
|
||||
from base.spider import Spider
|
||||
|
||||
|
||||
class Spider(Spider):
|
||||
|
||||
def init(self, extend=""):
|
||||
self.nav_host = 'https://www.xiguadh.com'
|
||||
self.headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
||||
}
|
||||
self.host = self._get_host()
|
||||
|
||||
def _get_host(self):
|
||||
"""获取视频站点 URL,失败时从导航页获取"""
|
||||
default_host = 'https://www.bzzdyy.com'
|
||||
try:
|
||||
r = self.fetch(default_host, headers=self.headers, timeout=10, verify=False)
|
||||
if r.status_code == 200:
|
||||
return default_host
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
r = self.fetch(self.nav_host, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
urls = re.findall(r'url:\s*["\']([^"\']+)["\']', html)
|
||||
for url in urls:
|
||||
if url.startswith('http') and 'xiguadh' not in url:
|
||||
return url.rstrip('/')
|
||||
except Exception:
|
||||
pass
|
||||
return default_host
|
||||
|
||||
def getName(self):
|
||||
return '西瓜影院'
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
return False
|
||||
|
||||
def manualVideoCheck(self):
|
||||
return True
|
||||
|
||||
def homeContent(self, filter):
|
||||
try:
|
||||
r = self.fetch(self.host, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
# 提取主要分类
|
||||
nav_match = re.search(r'<ul class="stui-header__menu">(.*?)</ul>', html, re.DOTALL)
|
||||
if nav_match:
|
||||
nav_html = nav_match.group(1)
|
||||
categories = re.findall(r'<li[^>]*><a href="/index.php/vod/type/id/(\d+)\.html">([^<]+)</a></li>', nav_html)
|
||||
else:
|
||||
categories = []
|
||||
seen = set()
|
||||
classes = []
|
||||
for tid, name in categories:
|
||||
if tid not in seen:
|
||||
seen.add(tid)
|
||||
classes.append({'type_name': name, 'type_id': tid})
|
||||
if not classes:
|
||||
raise Exception('No categories found')
|
||||
# 提取首页推荐视频
|
||||
videos = self._parse_vodlist(html)
|
||||
except Exception:
|
||||
classes = [
|
||||
{'type_name': '电影', 'type_id': '20'},
|
||||
{'type_name': '连续剧', 'type_id': '37'},
|
||||
{'type_name': '动漫', 'type_id': '43'},
|
||||
{'type_name': '综艺', 'type_id': '45'},
|
||||
{'type_name': 'B站', 'type_id': '47'},
|
||||
{'type_name': '人人专区', 'type_id': '60'},
|
||||
]
|
||||
videos = []
|
||||
return {"class": classes, "list": videos}
|
||||
|
||||
def _parse_vodlist(self, html):
|
||||
"""解析视频列表"""
|
||||
items = re.findall(
|
||||
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||
html
|
||||
)
|
||||
videos = []
|
||||
for href, title, pic in items:
|
||||
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||
if vod_id:
|
||||
vod_id = vod_id.group(1)
|
||||
else:
|
||||
continue
|
||||
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||
remark = remark_match.group(1) if remark_match else ''
|
||||
if pic.startswith('/'):
|
||||
pic = self.host + pic
|
||||
videos.append({
|
||||
'vod_id': vod_id,
|
||||
'vod_name': title,
|
||||
'vod_pic': pic,
|
||||
'vod_remarks': remark,
|
||||
})
|
||||
return videos
|
||||
|
||||
def homeVideoContent(self):
|
||||
return ''
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
pg = int(pg)
|
||||
url = f'{self.host}/index.php/vod/type/id/{tid}/page/{pg}.html'
|
||||
try:
|
||||
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
items = re.findall(
|
||||
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||
html
|
||||
)
|
||||
videos = []
|
||||
for href, title, pic in items:
|
||||
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||
if vod_id:
|
||||
vod_id = vod_id.group(1)
|
||||
else:
|
||||
continue
|
||||
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||
remark = remark_match.group(1) if remark_match else ''
|
||||
if pic.startswith('/'):
|
||||
pic = self.host + pic
|
||||
videos.append({
|
||||
'vod_id': vod_id,
|
||||
'vod_name': title,
|
||||
'vod_pic': pic,
|
||||
'vod_remarks': remark,
|
||||
})
|
||||
return {
|
||||
"list": videos,
|
||||
"page": pg,
|
||||
"pagecount": 9999,
|
||||
"limit": 90,
|
||||
"total": len(videos),
|
||||
}
|
||||
except Exception as e:
|
||||
return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
vod_id = ids[0] if isinstance(ids, list) else ids
|
||||
url = f'{self.host}/index.php/vod/detail/id/{vod_id}.html'
|
||||
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
title_match = re.search(r'<h1 class="title">([^<]+)</h1>', html)
|
||||
title = title_match.group(1).strip() if title_match else ''
|
||||
self._vod_name = title
|
||||
pic_match = re.search(r'<img class="lazyload" data-original="([^"]*)"', html)
|
||||
pic = pic_match.group(1) if pic_match else ''
|
||||
if pic.startswith('/'):
|
||||
pic = self.host + pic
|
||||
info_match = re.search(r'类型:([^/]+)\s*/\s*地区:([^/]+)\s*/\s*年份:(\d+)', html)
|
||||
type_name = info_match.group(1).strip() if info_match else ''
|
||||
area = info_match.group(2).strip() if info_match else ''
|
||||
year = info_match.group(3) if info_match else ''
|
||||
remark_match = re.search(r'状态:<span[^>]*>([^<]+)</span>', html)
|
||||
remark = remark_match.group(1).strip() if remark_match else ''
|
||||
director_match = re.search(r'导演:(.*?)</p>', html, re.DOTALL)
|
||||
director = ''
|
||||
if director_match:
|
||||
director = re.sub(r'<[^>]+>', '', director_match.group(1)).strip()
|
||||
actor_match = re.search(r'主演:([^<]+)', html)
|
||||
actor = actor_match.group(1).strip() if actor_match else ''
|
||||
desc_match = re.search(r'<span class="detail-content"[^>]*>([^<]+)</span>', html)
|
||||
desc = desc_match.group(1).strip() if desc_match else ''
|
||||
play_from = []
|
||||
play_url = []
|
||||
source_tabs = re.findall(r'<li><a href="#playlist\d+"[^>]*>([^<]+)</a></li>', html)
|
||||
for idx, source_name in enumerate(source_tabs):
|
||||
source_id = idx + 1
|
||||
episodes_match = re.search(
|
||||
f'<div id="playlist{source_id}" class="tab-pane[^"]*"[^>]*>.*?<ul class="stui-content__playlist[^"]*"[^>]*>(.*?)</ul>',
|
||||
html, re.DOTALL
|
||||
)
|
||||
if episodes_match:
|
||||
episodes = re.findall(r'<a href="([^"]+)">([^<]+)</a>', episodes_match.group(1))
|
||||
episode_list = []
|
||||
for ep_url, ep_name in episodes:
|
||||
episode_list.append(f'{ep_name}${self.host}{ep_url}')
|
||||
play_from.append(source_name)
|
||||
play_url.append('#'.join(episode_list))
|
||||
vod_play_from = '$$$'.join(play_from) if play_from else '默认'
|
||||
vod_play_url = '$$$'.join(play_url) if play_url else ''
|
||||
vod = {
|
||||
'vod_id': vod_id,
|
||||
'vod_name': title,
|
||||
'vod_pic': pic,
|
||||
'vod_year': year,
|
||||
'vod_area': area,
|
||||
'vod_remarks': remark,
|
||||
'vod_director': director,
|
||||
'vod_actor': actor,
|
||||
'vod_content': desc,
|
||||
'vod_play_from': vod_play_from,
|
||||
'vod_play_url': vod_play_url,
|
||||
}
|
||||
return {"list": [vod]}
|
||||
except Exception as e:
|
||||
return {"list": []}
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
pg = int(pg)
|
||||
url = f'{self.host}/index.php/vod/search/wd/{quote(key)}.html'
|
||||
try:
|
||||
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
items = re.findall(
|
||||
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||
html
|
||||
)
|
||||
videos = []
|
||||
for href, title, pic in items:
|
||||
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||
if vod_id:
|
||||
vod_id = vod_id.group(1)
|
||||
else:
|
||||
continue
|
||||
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||
remark = remark_match.group(1) if remark_match else ''
|
||||
if pic.startswith('/'):
|
||||
pic = self.host + pic
|
||||
videos.append({
|
||||
'vod_id': vod_id,
|
||||
'vod_name': title,
|
||||
'vod_pic': pic,
|
||||
'vod_remarks': remark,
|
||||
})
|
||||
return {
|
||||
"list": videos,
|
||||
"page": pg,
|
||||
"pagecount": 9999,
|
||||
"limit": 90,
|
||||
"total": len(videos),
|
||||
}
|
||||
except Exception as e:
|
||||
return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0}
|
||||
|
||||
def _clean_vod_name(self, name):
|
||||
import re
|
||||
if not name:
|
||||
return ''
|
||||
cleaned = re.sub(r'第\s*\d+\s*[集話话章部期]', '', name)
|
||||
cleaned = re.sub(r'EP\s*\d+', '', cleaned, flags=re.IGNORECASE)
|
||||
cleaned = re.sub(r'全\d+集', '', cleaned)
|
||||
cleaned = re.sub(r'更新至\d+集', '', cleaned)
|
||||
cleaned = re.sub(r'\d+集全', '', cleaned)
|
||||
cleaned = re.sub(r'[((].*?[))]', '', cleaned)
|
||||
cleaned = re.sub(r'\s*-\s*.*$', '', cleaned)
|
||||
cleaned = re.sub(r'\s+', ' ', cleaned)
|
||||
cleaned = re.sub(r'^[\s\-_,.,。、]+|[\s\-_,.,。、]+$', '', cleaned)
|
||||
return cleaned.strip()
|
||||
|
||||
def _build_danmaku_url(self, vod_name, vod_index=''):
|
||||
import re
|
||||
idx = 0
|
||||
if vod_index:
|
||||
s = str(vod_index).strip()
|
||||
m = re.search(r'第\s*(\d+)\s*[集話话章部期]', s)
|
||||
if m:
|
||||
idx = int(m.group(1))
|
||||
else:
|
||||
m = re.search(r'(\d+)', s)
|
||||
if m:
|
||||
idx = int(m.group(1))
|
||||
cleaned_name = self._clean_vod_name(vod_name)
|
||||
params = []
|
||||
if cleaned_name:
|
||||
params.append(f'vodName={quote(cleaned_name)}')
|
||||
params.append(f'vodIndex={idx}')
|
||||
query = '&'.join(params)
|
||||
return f'http://127.0.0.1:9978/proxy?do=appdanmu&{query}'
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
try:
|
||||
ep_name = ''
|
||||
vod_index = ''
|
||||
if '$' in id:
|
||||
parts = id.split('$', 1)
|
||||
ep_name = parts[0]
|
||||
url = parts[1] if len(parts) > 1 else ''
|
||||
else:
|
||||
url = id if id.startswith('http') else f'{self.host}{id}'
|
||||
# 从 URL 中提取集数 (nid 参数)
|
||||
nid_match = re.search(r'nid/(\d+)\.html', url)
|
||||
if nid_match:
|
||||
vod_index = nid_match.group(1)
|
||||
danmaku_url = self._build_danmaku_url(self._vod_name, vod_index)
|
||||
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||
html = r.text
|
||||
iframe_match = re.search(r'<iframe[^>]+src="([^"]+)"', html)
|
||||
if iframe_match:
|
||||
iframe_url = iframe_match.group(1)
|
||||
if not iframe_url.startswith('http'):
|
||||
iframe_url = self.host + iframe_url
|
||||
return {
|
||||
"parse": 1,
|
||||
"url": iframe_url,
|
||||
"header": self.headers,
|
||||
"danmaku": danmaku_url
|
||||
}
|
||||
src_match = re.search(r'(https?://[^"\'<>\s]+\.m3u8[^"\'<>\s]*)', html)
|
||||
if src_match:
|
||||
return {
|
||||
"parse": 0,
|
||||
"url": src_match.group(1),
|
||||
"header": self.headers,
|
||||
"danmaku": danmaku_url
|
||||
}
|
||||
return {
|
||||
"parse": 1,
|
||||
"url": url,
|
||||
"header": self.headers,
|
||||
"danmaku": danmaku_url
|
||||
}
|
||||
except Exception as e:
|
||||
danmaku_url = self._build_danmaku_url(self._vod_name, '')
|
||||
return {"parse": 1, "url": id, "header": {}, "danmaku": danmaku_url}
|
||||
|
||||
def localProxy(self, param):
|
||||
return [200, {}, ""]
|
||||
|
||||
def destroy(self):
|
||||
pass
|
||||
Reference in New Issue
Block a user