上传文件至「py」
This commit is contained in:
+350
@@ -0,0 +1,350 @@
|
|||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
# 本资源来源于互联网公开渠道,仅可用于个人学习爬虫技术。
|
||||||
|
# 严禁将其用于任何商业用途,下载后请于 24 小时内删除,搜索结果均来自源站,本人不承担任何责任。
|
||||||
|
#junyouyun
|
||||||
|
|
||||||
|
import sys
|
||||||
|
import json
|
||||||
|
import time
|
||||||
|
import base64
|
||||||
|
import hashlib
|
||||||
|
import urllib3
|
||||||
|
import concurrent.futures
|
||||||
|
from urllib.parse import quote
|
||||||
|
from base.spider import Spider
|
||||||
|
from Crypto.PublicKey import RSA
|
||||||
|
from Crypto.Cipher import PKCS1_v1_5
|
||||||
|
|
||||||
|
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
||||||
|
sys.path.append('..')
|
||||||
|
|
||||||
|
class Spider(Spider):
|
||||||
|
host, userid, episode_list = '', '', []
|
||||||
|
|
||||||
|
# ---------- 加密与签名相关常量 ----------
|
||||||
|
PUB_KEY_B64 = "MIGfMA0GCSqGSIb3DQEBAQUAA4GNADCBiQKBgQCoYt0BP77U+DM08BiI/QbSRIfxijXo85BTPqIM1Ow8BNwhLETzRIZ+dEwdWDbydG/PspgBAfRpGaYVdJYtvaC2JnoO8+Ik6qMWojfEJxSFLa0Pb0A892tun4gsxoEMjcreZ+YGyaBxAfqX0BSMfdrOgIYaZQjYrw9TRLlUT31QoQIDAQAB"
|
||||||
|
APP_SIGN_SHA1 = "09a8dc51639a31801af5f6418caebfabc695eb24"
|
||||||
|
DEVICE_ID = "2d590b9842d064a1"
|
||||||
|
|
||||||
|
# RSA 私钥(用于解密响应)
|
||||||
|
PRIV_KEY_B64 = """MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCquQQ5r6+yJI8CDFkXRp8vUsdD45ov8EP12ooLs56ca2DQXaSNGS9910bAPVA9chkp0mKIvKqjAsHz5Tl9EeNPblarGEeJUIxpxZtiSqNTpvtiD/TjhpzuHYic7RAfQ/h7p/ypE8ymU42pYjsB5t26Mv6XgkLV+jzrSf73HlCuS0iMyLmt6zz3Mw9izM13EpB8iFLtfbbYymycKTx4RAmPQLwhNGex/AlUIYxXP4R2yyaa4W6mEtc6aME2QuzJFxPgP3HJ9NBx/LWVn4skxWjZ7zg+VRQRHnjyVaSLu3Z5gN5ITWCyE32qaHJa6WBahZj5jWhRyAG1bQ+xKJa8lBL5AgMBAAECggEAUwv9SjJ0PSwbhNuM2w23kcWquROWhYtTA91zGY4esehqB/IFgb2mpIh8Gje5OKqwIu/8jpd4SiOlRYdUF8sD0DfUYRZGdj2AkFNX6tBz8tVfo6wvbB6naA1lzzBij1L5JO3qsjS3cJFkb+kg2yP66AC2Z+0tpfk8eRhdtshAZwfcd1DEGt1uAvYL1eaUK9HRvpt9lPeGcHERDl2hBd4uyaF0K1O+zF9y59nYbTySWPxRZq3sFEE85xRMlstD7YZi7W2gKvMFRD4/FKmrZ3m7aKJRITtyKOyyPcYmepNv3Qv7kk59Pg38n2WWQ0Ra/bCH3E48YNCnQvZMpitkTfJhoQKBgQDbnROOYTP8OTJ6f/qhoGjxeO3x1VOaOp8l0x7b0SCfoqNGS0Cyiqj72BmJtPMPqSTjn6MmNzqbg1KOdhXyzNozs+i5ccW1M56j96mr5I/Z0FpE3oyIHNfDDBlf9M8YQqEF9oYxniYYft9oapO7cRQkHER6qpvnHTavwlv4m78CXwKBgQDHAjs2YlpKDdI1lcbZJCc7TwtH+Pd2bUki8YXafWNcPhITQHbOZjr310eK1QJC6GJncjkOqbX7yv3ivvTO35FZTQhuA1xEG1P00FG8bE0tHYPIwQHi9y0eA5cieMdo8E6XYria1mw/3fqSQEsfZyJlR32JQIoGAipM8iO1X2nZpwKBgDkMFIhnt5lNQk+P7wsNIDWZtDWdtJnboHuy29E+Abt2A/O+mI/IdRz2hau/1WO8DFkUnszOi+rZshhPlGP90rCbi1igtTrcrdjp/KkqNjPea5R4OwkgdOu1uOG0NheXNzzVTQaWjk7Opjn5dWa7eP/oV+GFb/oZHJuLYVizHGsBAoGADA7rjZEKDYCm4w5PPSr+oY5ZjaPdQrS+gLqHtMRyN82fBMGcMUdqfUfzEstzVqCEDeaS5HuOBlK3bXzKkppjUTjksN3NQmcxgBz7RuJ9DqXCLXDcb2cwuafYCYOt+YLOEEgwDVm+t2P44dG5e46hO+fICH/7nP+WlpD5buz4GfMCgYB57r3g/6hi9WUDnfc7ZAzWMqR0EhJVYKYy+KFEtdIPzhkkIHq5RASe88E9kzoGoZFdb3tIjvGZWcHerirrqWkMsuQtP/Qi0zjieid5tAPj+r4kbiCVTw0E0jnmPBzGInQi7lpeTTKnG1fbyS5lBS+WmHfIuzpECgCkxhaT+LJJkg=="""
|
||||||
|
|
||||||
|
headers = {
|
||||||
|
'User-Agent': "okhttp/4.12.0",
|
||||||
|
'Connection': "Keep-Alive",
|
||||||
|
'Accept-Encoding': "gzip",
|
||||||
|
'Content-Type': "application/json;charset=UTF-8",
|
||||||
|
'Cache-Control': "no-cache",
|
||||||
|
'token': "",
|
||||||
|
'deviceId': DEVICE_ID,
|
||||||
|
'client': "app",
|
||||||
|
'deviceType': "Android"
|
||||||
|
}
|
||||||
|
|
||||||
|
# ---------- RSA 加密 ----------
|
||||||
|
def rsa_encrypt(self, data: str) -> str:
|
||||||
|
key = RSA.import_key(base64.b64decode(self.PUB_KEY_B64))
|
||||||
|
cipher = PKCS1_v1_5.new(key)
|
||||||
|
encrypted = cipher.encrypt(data.encode('utf-8'))
|
||||||
|
return base64.b64encode(encrypted).decode('utf-8')
|
||||||
|
|
||||||
|
# ---------- RSA 解密(支持分块) ----------
|
||||||
|
def rsa_decrypt(self, encrypted_b64: str) -> str:
|
||||||
|
key = RSA.import_key(base64.b64decode(self.PRIV_KEY_B64))
|
||||||
|
cipher = PKCS1_v1_5.new(key)
|
||||||
|
encrypted_bytes = base64.b64decode(encrypted_b64)
|
||||||
|
block_size = 256
|
||||||
|
decrypted_parts = []
|
||||||
|
for i in range(0, len(encrypted_bytes), block_size):
|
||||||
|
block = encrypted_bytes[i:i+block_size]
|
||||||
|
decrypted_parts.append(cipher.decrypt(block, None))
|
||||||
|
return b''.join(decrypted_parts).decode('utf-8')
|
||||||
|
|
||||||
|
# ---------- 构建签名参数 ----------
|
||||||
|
def build_params_string(self, episode_id="", episode_index="", vid="", player_id="", type_id="", user_id=""):
|
||||||
|
return (f"episodeId{episode_id}"
|
||||||
|
f"episodeIndex{episode_index}"
|
||||||
|
f"id{vid}"
|
||||||
|
f"playerId{player_id}"
|
||||||
|
f"source0"
|
||||||
|
f"typeId{type_id}"
|
||||||
|
f"userId{user_id}")
|
||||||
|
|
||||||
|
def generate_sign(self, timestamp: str, params_str: str, device_id: str) -> str:
|
||||||
|
raw = f"SaltLSFBTimestamp{timestamp}Params{params_str}ClientappDeviceId{device_id}"
|
||||||
|
b64 = base64.b64encode(raw.encode('utf-8')).decode('utf-8')
|
||||||
|
md5 = hashlib.md5(b64.encode('utf-8')).hexdigest().upper()
|
||||||
|
return md5
|
||||||
|
|
||||||
|
def build_encrypted_headers(self, body_json: str, params_str: str) -> dict:
|
||||||
|
timestamp = str(int(time.time()))
|
||||||
|
encrypted_key = self.rsa_encrypt(body_json)
|
||||||
|
snjm = self.rsa_encrypt("113")
|
||||||
|
appsign = self.rsa_encrypt(self.APP_SIGN_SHA1)
|
||||||
|
sign = self.generate_sign(timestamp, params_str, self.DEVICE_ID)
|
||||||
|
|
||||||
|
headers = {
|
||||||
|
"snjm": snjm,
|
||||||
|
"appsign": appsign,
|
||||||
|
"timestamp": timestamp,
|
||||||
|
"sign": sign,
|
||||||
|
"deviceId": self.DEVICE_ID,
|
||||||
|
"token": self.headers.get('token', ''),
|
||||||
|
"client": "app",
|
||||||
|
"deviceType": "Android",
|
||||||
|
"Content-Type": "application/json;charset=UTF-8",
|
||||||
|
"Cache-Control": "no-cache",
|
||||||
|
"User-Agent": "okhttp/4.12.0"
|
||||||
|
}
|
||||||
|
return headers, {"key": encrypted_key}
|
||||||
|
|
||||||
|
# ---------- 原有接口(保持不变) ----------
|
||||||
|
def init(self, extend=''):
|
||||||
|
self.headers['deviceId'] = self.DEVICE_ID
|
||||||
|
self.host = 'http://qkys.qukanwh.com'
|
||||||
|
response = self.fetch(f'{self.host}/api/v1/app/user/visitorInfo', headers=self.headers).json()
|
||||||
|
self.userid = response['data']['id']
|
||||||
|
token = response['data']['token']
|
||||||
|
self.headers['token'] = token
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
response = self.post(f'{self.host}/api/v1/app/screen/screenType', headers=self.headers).json()
|
||||||
|
data = response['data']
|
||||||
|
classes = []
|
||||||
|
for i in data:
|
||||||
|
classes.append({'type_id': i['id'], 'type_name': i['name']})
|
||||||
|
return {'class': classes}
|
||||||
|
|
||||||
|
def homeVideoContent(self):
|
||||||
|
response = self.post(f'{self.host}/api/v1/app/recommend/recommendList', headers=self.headers).json()
|
||||||
|
data = response['data']
|
||||||
|
videos = []
|
||||||
|
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||||
|
future_to_id = {
|
||||||
|
executor.submit(
|
||||||
|
self.post,
|
||||||
|
f'{self.host}/api/v1/app/recommend/recommendSubList',
|
||||||
|
data=json.dumps({
|
||||||
|
"condition": item['id'],
|
||||||
|
"pageNum": 1,
|
||||||
|
"pageSize": 6
|
||||||
|
}),
|
||||||
|
headers=self.headers
|
||||||
|
): item['id'] for item in data
|
||||||
|
}
|
||||||
|
for future in concurrent.futures.as_completed(future_to_id):
|
||||||
|
try:
|
||||||
|
response = future.result().json()
|
||||||
|
for video in response['data']['records']:
|
||||||
|
videos.append({
|
||||||
|
"vod_id": video['id'],
|
||||||
|
"vod_name": video['name'],
|
||||||
|
"vod_pic": video['cover']
|
||||||
|
})
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Request failed for item {future_to_id[future]}: {str(e)}")
|
||||||
|
return {'list': videos}
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
payload = {
|
||||||
|
"condition": {
|
||||||
|
"classify": "",
|
||||||
|
"region": "",
|
||||||
|
"sreecnTypeEnum": "NEWEST",
|
||||||
|
"typeId": tid,
|
||||||
|
"year": ""
|
||||||
|
},
|
||||||
|
"pageNum": pg,
|
||||||
|
"pageSize": 40
|
||||||
|
}
|
||||||
|
response = self.post(f'{self.host}/api/v1/app/screen/screenMovie', data=json.dumps(payload), headers=self.headers).json()
|
||||||
|
videos = []
|
||||||
|
for i in response['data']['records']:
|
||||||
|
videos.append({
|
||||||
|
"vod_id": i['id'],
|
||||||
|
"vod_name": i['name'],
|
||||||
|
"vod_pic": i['cover'],
|
||||||
|
"vod_remarks": i['area'],
|
||||||
|
"vod_year": i['year']
|
||||||
|
})
|
||||||
|
return {'list': videos, 'page': pg}
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg='1'):
|
||||||
|
payload = {
|
||||||
|
"condition": {
|
||||||
|
"value": key
|
||||||
|
},
|
||||||
|
"pageNum": pg,
|
||||||
|
"pageSize": 40
|
||||||
|
}
|
||||||
|
response = self.post(f'{self.host}/api/v1/app/search/searchMovie', data=json.dumps(payload), headers=self.headers).json()
|
||||||
|
videos = []
|
||||||
|
for i in response['data']['records']:
|
||||||
|
videos.append({
|
||||||
|
'vod_id': i['id'],
|
||||||
|
'vod_name': i['name'],
|
||||||
|
'vod_pic': i['cover'],
|
||||||
|
'vod_remarks': i['area'],
|
||||||
|
'vod_year': i['year'],
|
||||||
|
'vod_area': i['area'],
|
||||||
|
'vod_content': i['desc']
|
||||||
|
})
|
||||||
|
return {'list': videos, 'page': pg}
|
||||||
|
|
||||||
|
# ---------- 详情页(已集成解密) ----------
|
||||||
|
def detailContent(self, ids):
|
||||||
|
type_id = "M15" # 注意:原脚本写死为 M17,可根据需要修改
|
||||||
|
vid = ids[0]
|
||||||
|
body = {
|
||||||
|
"id": vid,
|
||||||
|
"source": 0,
|
||||||
|
"typeId": type_id,
|
||||||
|
"userId": self.userid,
|
||||||
|
"episodeId": "",
|
||||||
|
"episodeIndex": "",
|
||||||
|
"playerId": ""
|
||||||
|
}
|
||||||
|
body_json = json.dumps(body, separators=(',', ':'))
|
||||||
|
params_str = self.build_params_string(
|
||||||
|
episode_id="",
|
||||||
|
episode_index="",
|
||||||
|
vid=str(vid),
|
||||||
|
player_id="",
|
||||||
|
type_id=type_id,
|
||||||
|
user_id=str(self.userid)
|
||||||
|
)
|
||||||
|
headers, payload = self.build_encrypted_headers(body_json, params_str)
|
||||||
|
|
||||||
|
# 发送加密请求
|
||||||
|
resp_raw = self.post(f'{self.host}/api/v1/app/play/movieDetails', data=json.dumps(payload), headers=headers).json()
|
||||||
|
encrypted_data = resp_raw.get('data')
|
||||||
|
if not encrypted_data:
|
||||||
|
raise Exception("响应中 data 为空")
|
||||||
|
# 解密 data 字段
|
||||||
|
decrypted_json_str = self.rsa_decrypt(encrypted_data)
|
||||||
|
data = json.loads(decrypted_json_str)
|
||||||
|
|
||||||
|
# 后续处理与原脚本相同
|
||||||
|
currentplayerid = data['playerId']
|
||||||
|
play_urls = []
|
||||||
|
play_url = []
|
||||||
|
show = []
|
||||||
|
for i in data['episodeList']:
|
||||||
|
play_url.append(f"{i['episode']}${ids[0]}@{currentplayerid}@{i['id']}@episode")
|
||||||
|
play_urls.append('#'.join(play_url))
|
||||||
|
moviePlayerList = data['moviePlayerList']
|
||||||
|
for i2 in moviePlayerList:
|
||||||
|
if i2['id'] == currentplayerid:
|
||||||
|
show.append(i2['moviePlayerName'])
|
||||||
|
for j in moviePlayerList:
|
||||||
|
playerid = j['id']
|
||||||
|
episodeTotal = j.get('episodeTotal')
|
||||||
|
if playerid == currentplayerid or episodeTotal is None:
|
||||||
|
continue
|
||||||
|
play_url = []
|
||||||
|
for k in range(1, episodeTotal + 1):
|
||||||
|
play_url.append(f"第{k}集${k}@{playerid}@{ids[0]}@virtual")
|
||||||
|
play_urls.append('#'.join(play_url))
|
||||||
|
if j['moviePlayerName'] not in show:
|
||||||
|
show.append(j['moviePlayerName'])
|
||||||
|
|
||||||
|
# 获取简介(此接口可能无需加密,保持原样)
|
||||||
|
payload_desc = {
|
||||||
|
"id": ids[0],
|
||||||
|
"typeId": type_id
|
||||||
|
}
|
||||||
|
response_desc = self.post(f'{self.host}/api/v1/app/play/movieDesc', data=json.dumps(payload_desc), headers=self.headers).json()
|
||||||
|
data2 = response_desc['data']
|
||||||
|
|
||||||
|
video = {
|
||||||
|
'vod_id': data2['id'],
|
||||||
|
'vod_name': data2['name'],
|
||||||
|
'vod_pic': data2['cover'],
|
||||||
|
'vod_content': data2['introduce'],
|
||||||
|
'vod_year': data2['year'],
|
||||||
|
'vod_area': data2['area'],
|
||||||
|
'vod_remarks': '',
|
||||||
|
'vod_score': data2['score'],
|
||||||
|
'type_name': data2['classify'],
|
||||||
|
'vod_director': data2['director'],
|
||||||
|
'vod_actor': data2['star'],
|
||||||
|
'vod_play_from': '$$$'.join(show),
|
||||||
|
'vod_play_url': '$$$'.join(play_urls)
|
||||||
|
}
|
||||||
|
return {'list': [video]}
|
||||||
|
|
||||||
|
# ---------- 播放页(已集成解密) ----------
|
||||||
|
def playerContent(self, flag, id, vipflags):
|
||||||
|
param, playerid, param2, param3 = id.split('@')
|
||||||
|
if param3 == 'virtual':
|
||||||
|
payload = {
|
||||||
|
"episodeIndex": str(int(param) - 1),
|
||||||
|
"id": int(param2),
|
||||||
|
"playerId": playerid,
|
||||||
|
"source": 0,
|
||||||
|
"typeId": "M15",
|
||||||
|
"userId": self.userid,
|
||||||
|
"episodeId": ""
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
payload = {
|
||||||
|
"episodeId": param2,
|
||||||
|
"id": int(param),
|
||||||
|
"playerId": playerid,
|
||||||
|
"source": 0,
|
||||||
|
"typeId": "M15",
|
||||||
|
"userId": self.userid,
|
||||||
|
"episodeIndex": ""
|
||||||
|
}
|
||||||
|
body_json = json.dumps(payload, separators=(',', ':'))
|
||||||
|
print(body_json)
|
||||||
|
params_str = self.build_params_string(
|
||||||
|
episode_id=payload.get("episodeId", ""),
|
||||||
|
episode_index=payload.get("episodeIndex", ""),
|
||||||
|
vid=str(payload["id"]),
|
||||||
|
player_id=payload["playerId"],
|
||||||
|
type_id=payload["typeId"],
|
||||||
|
user_id=str(payload["userId"])
|
||||||
|
)
|
||||||
|
print(params_str)
|
||||||
|
headers, encrypted_payload = self.build_encrypted_headers(body_json, params_str)
|
||||||
|
print(headers)
|
||||||
|
print(encrypted_payload)
|
||||||
|
# 获取播放信息(加密响应)
|
||||||
|
resp_raw = self.post(f'{self.host}/api/v1/app/play/movieDetails', data=json.dumps(encrypted_payload), headers=headers).json()
|
||||||
|
encrypted_data = resp_raw.get('data')
|
||||||
|
if not encrypted_data:
|
||||||
|
raise Exception("响应中 data 为空")
|
||||||
|
decrypted_json_str = self.rsa_decrypt(encrypted_data)
|
||||||
|
data = json.loads(decrypted_json_str)
|
||||||
|
print(data)
|
||||||
|
parse_url = data['url']
|
||||||
|
playerid = data['playerId']
|
||||||
|
|
||||||
|
# 调用分析接口(注:analysisMovieUrl 的响应可能也是加密的,但原脚本直接取 data,这里暂不做额外解密)
|
||||||
|
analysis_body = {
|
||||||
|
"playerUrl": parse_url,
|
||||||
|
"playerId": playerid
|
||||||
|
}
|
||||||
|
analysis_json = json.dumps(analysis_body, separators=(',', ':'))
|
||||||
|
# analysisMovieUrl 接口的参数拼接?理论上也需要签名,但原脚本是 GET 方式,为了兼容,我们沿用原脚本的 GET 方式
|
||||||
|
# 原脚本使用 fetch GET 带参数,并未加密。这里也采用 GET 方式,不使用加密 headers
|
||||||
|
resp_analysis = self.fetch(f"{self.host}/api/v1/app/play/analysisMovieUrl?playerUrl={quote(parse_url,safe='')}&playerId={playerid}", headers=self.headers).json()
|
||||||
|
url = resp_analysis.get('data')
|
||||||
|
|
||||||
|
return {'jx': '0', 'parse': '0', 'url': url, 'header': {'User-Agent': 'Mozilla/5.0 (iPhone; CPU iPhone OS 13_2_3 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/13.0.3 Mobile/15E148 Safari/604.1'}}
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def isVideoFormat(self, url):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def manualVideoCheck(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def destroy(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def localProxy(self, param):
|
||||||
|
pass
|
||||||
+545
@@ -0,0 +1,545 @@
|
|||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
"""
|
||||||
|
悟空影视爬虫
|
||||||
|
站点: https://www.yikucun.com
|
||||||
|
基于 MacCMS (苹果CMS) 程序,HTML 解析方式
|
||||||
|
OK影视 / 海阔视界 兼容版
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
import time
|
||||||
|
import json
|
||||||
|
import urllib.parse
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
try:
|
||||||
|
from base.spider import Spider as BaseSpider
|
||||||
|
except ImportError:
|
||||||
|
class BaseSpider:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class Spider(BaseSpider):
|
||||||
|
BASE_URL = "https://www.yikucun.com"
|
||||||
|
|
||||||
|
# 分类映射
|
||||||
|
TYPE_MAP = {
|
||||||
|
"1": "电影",
|
||||||
|
"2": "电视剧",
|
||||||
|
"3": "综艺",
|
||||||
|
"4": "动漫",
|
||||||
|
"5": "短剧",
|
||||||
|
}
|
||||||
|
|
||||||
|
# 筛选条件 - 各分类的类型和地区
|
||||||
|
FILTERS = {
|
||||||
|
"1": { # 电影
|
||||||
|
"类型": ["全部", "动作", "喜剧", "爱情", "科幻", "恐怖", "剧情", "战争", "犯罪", "奇幻", "悬疑", "动画", "恐怖", "纪录片", "其他"],
|
||||||
|
"地区": ["全部", "大陆", "香港", "台湾", "日本", "韩国", "美国", "法国", "英国", "德国", "泰国", "印度", "其他"],
|
||||||
|
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015", "2014", "2013", "2012", "2011", "2010"],
|
||||||
|
},
|
||||||
|
"2": { # 电视剧
|
||||||
|
"类型": ["全部", "古装", "战争", "青春偶像", "喜剧", "家庭", "犯罪", "动作", "奇幻", "剧情", "历史", "经典", "乡村", "情景", "商战", "网剧", "其他"],
|
||||||
|
"地区": ["全部", "内地", "韩国", "香港", "台湾", "日本", "美国", "泰国", "英国", "新加坡", "其他"],
|
||||||
|
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015", "2014", "2013", "2012", "2011", "2010"],
|
||||||
|
},
|
||||||
|
"3": { # 综艺
|
||||||
|
"类型": ["全部", "真人秀", "脱口秀", "访谈", "美食", "旅游", "选秀", "情感", "音乐", "舞蹈", "其他"],
|
||||||
|
"地区": ["全部", "大陆", "香港", "台湾", "日本", "韩国", "欧美", "其他"],
|
||||||
|
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015"],
|
||||||
|
},
|
||||||
|
"4": { # 动漫
|
||||||
|
"类型": ["全部", "日本动漫", "国产动漫", "欧美动漫", "其他"],
|
||||||
|
"地区": ["全部", "日本", "大陆", "美国", "其他"],
|
||||||
|
"年份": ["全部", "2026", "2025", "2024", "2023", "2022", "2021", "2020", "2019", "2018", "2017", "2016", "2015"],
|
||||||
|
},
|
||||||
|
"5": { # 短剧
|
||||||
|
"类型": ["全部", "其他"],
|
||||||
|
"地区": ["全部", "其他"],
|
||||||
|
"年份": ["全部", "2026", "2025", "2024", "2023"],
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
super().__init__()
|
||||||
|
self.name = ""
|
||||||
|
self.error_play_url = "https://kjjsaas-sh.oss-cn-shanghai.aliyuncs.com/u/3401405881/20240818-936952-fc31b16575e80a7562cdb1f81a39c6b0.mp4"
|
||||||
|
self.session = requests.Session()
|
||||||
|
self.session.headers.update({
|
||||||
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||||
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
|
||||||
|
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||||
|
"Referer": "https://www.yikucun.com/",
|
||||||
|
"Connection": "keep-alive",
|
||||||
|
})
|
||||||
|
self._init_cookies()
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
return self.name
|
||||||
|
|
||||||
|
def init(self, extend="{}"):
|
||||||
|
try:
|
||||||
|
self.extend = json.loads(extend)
|
||||||
|
self.name = self.extend.get("name", "")
|
||||||
|
except Exception as e:
|
||||||
|
print(e)
|
||||||
|
self.extend = {}
|
||||||
|
|
||||||
|
def _init_cookies(self):
|
||||||
|
"""初始化 cookies(处理 508 反爬)"""
|
||||||
|
try:
|
||||||
|
r = self.session.get(self.BASE_URL, timeout=10, verify=False)
|
||||||
|
if r.status_code == 508:
|
||||||
|
time.sleep(1)
|
||||||
|
self.session.get(self.BASE_URL, timeout=10, verify=False)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _get(self, url, **kwargs):
|
||||||
|
"""GET 请求,自动处理 508"""
|
||||||
|
try:
|
||||||
|
r = self.session.get(url, timeout=15, verify=False, **kwargs)
|
||||||
|
if r.status_code == 508:
|
||||||
|
time.sleep(0.5)
|
||||||
|
r = self.session.get(url, timeout=15, verify=False, **kwargs)
|
||||||
|
r.encoding = "utf-8"
|
||||||
|
return r
|
||||||
|
except Exception as e:
|
||||||
|
class FakeResp:
|
||||||
|
status_code = 500
|
||||||
|
text = str(e)
|
||||||
|
return FakeResp()
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
"""首页 - 返回分类和推荐"""
|
||||||
|
result = {
|
||||||
|
"class": [],
|
||||||
|
"filters": {},
|
||||||
|
"list": [],
|
||||||
|
"parse": 0,
|
||||||
|
"jx": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
# 分类列表
|
||||||
|
for tid, name in self.TYPE_MAP.items():
|
||||||
|
result["class"].append({
|
||||||
|
"type_id": tid,
|
||||||
|
"type_name": name,
|
||||||
|
})
|
||||||
|
|
||||||
|
# 筛选条件
|
||||||
|
for tid, flist in self.FILTERS.items():
|
||||||
|
result["filters"][tid] = []
|
||||||
|
for fname, fvalues in flist.items():
|
||||||
|
result["filters"][tid].append({
|
||||||
|
"key": fname,
|
||||||
|
"name": fname,
|
||||||
|
"value": [{"n": v, "v": v} for v in fvalues],
|
||||||
|
})
|
||||||
|
|
||||||
|
# 首页推荐
|
||||||
|
try:
|
||||||
|
r = self._get(f"{self.BASE_URL}/")
|
||||||
|
if r.status_code == 200:
|
||||||
|
result["list"] = self._parse_list(r.text)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
"""分类页"""
|
||||||
|
result = {
|
||||||
|
"list": [],
|
||||||
|
"parse": 0,
|
||||||
|
"jx": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
# URL 格式: 12个位置,用 - 分隔
|
||||||
|
# 位置: 1=tid, 2=地区, 3=空, 4=类型, 5=空, 6=空, 7=空, 8=空, 9=页码, 10=空, 11=空, 12=年份
|
||||||
|
area = extend.get("地区", "") if isinstance(extend, dict) else ""
|
||||||
|
type_val = extend.get("类型", "") if isinstance(extend, dict) else ""
|
||||||
|
year = extend.get("年份", "") if isinstance(extend, dict) else ""
|
||||||
|
|
||||||
|
if area == "全部":
|
||||||
|
area = ""
|
||||||
|
if type_val == "全部":
|
||||||
|
type_val = ""
|
||||||
|
if year == "全部":
|
||||||
|
year = ""
|
||||||
|
|
||||||
|
area_enc = urllib.parse.quote(area) if area else ""
|
||||||
|
type_enc = urllib.parse.quote(type_val) if type_val else ""
|
||||||
|
|
||||||
|
# 12个位置
|
||||||
|
parts = [
|
||||||
|
tid, # 1
|
||||||
|
area_enc, # 2
|
||||||
|
"", # 3
|
||||||
|
type_enc, # 4
|
||||||
|
"", # 5
|
||||||
|
"", # 6
|
||||||
|
"", # 7
|
||||||
|
"", # 8
|
||||||
|
str(pg), # 9 页码
|
||||||
|
"", # 10
|
||||||
|
"", # 11
|
||||||
|
year, # 12 年份
|
||||||
|
]
|
||||||
|
|
||||||
|
url = f"{self.BASE_URL}/ucusw/{'-'.join(parts)}.html"
|
||||||
|
|
||||||
|
try:
|
||||||
|
r = self._get(url)
|
||||||
|
if r.status_code == 200:
|
||||||
|
result["list"] = self._parse_list(r.text)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def detailContent(self, ids):
|
||||||
|
"""详情页"""
|
||||||
|
result = {
|
||||||
|
"list": [],
|
||||||
|
"parse": 0,
|
||||||
|
"jx": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
vid = ids[0]
|
||||||
|
url = f"{self.BASE_URL}/ucudt/{vid}.html"
|
||||||
|
|
||||||
|
try:
|
||||||
|
r = self._get(url)
|
||||||
|
if r.status_code == 200:
|
||||||
|
detail = self._parse_detail(r.text, vid)
|
||||||
|
if detail:
|
||||||
|
result["list"].append(detail)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg="1"):
|
||||||
|
"""搜索"""
|
||||||
|
result = {
|
||||||
|
"list": [],
|
||||||
|
"parse": 0,
|
||||||
|
"jx": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
try:
|
||||||
|
# 搜索 URL: /ucusc/-------------.html?wd=关键词
|
||||||
|
# 分页: /ucusc/-------------(页码).html?wd=关键词
|
||||||
|
if int(pg) > 1:
|
||||||
|
search_url = f"{self.BASE_URL}/ucusc/-------------{pg}.html?wd={urllib.parse.quote(key)}"
|
||||||
|
else:
|
||||||
|
search_url = f"{self.BASE_URL}/ucusc/-------------.html?wd={urllib.parse.quote(key)}"
|
||||||
|
r = self._get(search_url)
|
||||||
|
if r.status_code == 200:
|
||||||
|
result["list"] = self._parse_list(r.text)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def playerContent(self, flag, id, vipFlags):
|
||||||
|
"""播放页 - 获取播放地址"""
|
||||||
|
result = {
|
||||||
|
"url": self.error_play_url,
|
||||||
|
"parse": 0,
|
||||||
|
"jx": 0,
|
||||||
|
"header": {},
|
||||||
|
}
|
||||||
|
|
||||||
|
try:
|
||||||
|
# id 格式: 视频ID-线路ID-集数ID
|
||||||
|
parts = id.split("-")
|
||||||
|
if len(parts) >= 3:
|
||||||
|
vid, sid, nid = parts[0], parts[1], parts[2]
|
||||||
|
play_url = f"{self.BASE_URL}/ucupy/{vid}-{sid}-{nid}.html"
|
||||||
|
r = self._get(play_url)
|
||||||
|
if r.status_code == 200:
|
||||||
|
url = self._parse_play_url(r.text)
|
||||||
|
if url:
|
||||||
|
result["url"] = url
|
||||||
|
result["parse"] = 0
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
def localProxy(self, params):
|
||||||
|
return 0
|
||||||
|
|
||||||
|
def _parse_list(self, html):
|
||||||
|
"""解析列表页"""
|
||||||
|
items = []
|
||||||
|
|
||||||
|
# 找到所有 class="name" 的标题,然后往前找最近的图片
|
||||||
|
name_pattern = r'class="name"[^>]*>\s*<a[^>]*href="/ucudt/(\d+)\.html"[^>]*>([^<]+)</a>'
|
||||||
|
|
||||||
|
# 先找出所有 name 的位置
|
||||||
|
name_matches = list(re.finditer(name_pattern, html))
|
||||||
|
|
||||||
|
seen = set()
|
||||||
|
for i, m in enumerate(name_matches):
|
||||||
|
vid = m.group(1)
|
||||||
|
name = m.group(2).strip()
|
||||||
|
|
||||||
|
if vid in seen:
|
||||||
|
continue
|
||||||
|
seen.add(vid)
|
||||||
|
|
||||||
|
# 往前找最近的图片(在这个 name 之前)
|
||||||
|
pos = m.start()
|
||||||
|
start = max(0, pos - 2000)
|
||||||
|
segment = html[start:pos]
|
||||||
|
|
||||||
|
# 找所有图片
|
||||||
|
img_matches = re.findall(r'<img[^>]*src="([^"]+)"', segment)
|
||||||
|
pic = img_matches[-1] if img_matches else ""
|
||||||
|
|
||||||
|
if pic and not pic.startswith("http"):
|
||||||
|
pic = "https:" + pic if pic.startswith("//") else pic
|
||||||
|
|
||||||
|
# 找备注
|
||||||
|
remarks = ""
|
||||||
|
rgba_match = re.search(r'class="rgba[^"]*"[^>]*>([^<]+)</span>', segment)
|
||||||
|
if rgba_match:
|
||||||
|
remarks = rgba_match.group(1).strip()
|
||||||
|
|
||||||
|
items.append({
|
||||||
|
"vod_id": vid,
|
||||||
|
"vod_name": name,
|
||||||
|
"vod_pic": pic,
|
||||||
|
"vod_remarks": remarks,
|
||||||
|
})
|
||||||
|
|
||||||
|
return items[:30]
|
||||||
|
|
||||||
|
def _parse_detail(self, html, vid):
|
||||||
|
"""解析详情页"""
|
||||||
|
detail = {
|
||||||
|
"vod_id": vid,
|
||||||
|
"vod_name": "",
|
||||||
|
"vod_pic": "",
|
||||||
|
"type_name": "",
|
||||||
|
"vod_year": "",
|
||||||
|
"vod_area": "",
|
||||||
|
"vod_remarks": "",
|
||||||
|
"vod_actor": "",
|
||||||
|
"vod_director": "",
|
||||||
|
"vod_content": "",
|
||||||
|
"vod_play_from": "",
|
||||||
|
"vod_play_url": "",
|
||||||
|
}
|
||||||
|
|
||||||
|
# 标题
|
||||||
|
title_match = re.search(r'<title>《([^》]+)》', html)
|
||||||
|
if title_match:
|
||||||
|
detail["vod_name"] = title_match.group(1).strip()
|
||||||
|
|
||||||
|
# 图片
|
||||||
|
pic_match = re.search(r'og:image"[^>]*content="([^"]+)"', html)
|
||||||
|
if pic_match:
|
||||||
|
pic = pic_match.group(1)
|
||||||
|
if not pic.startswith("http"):
|
||||||
|
pic = "https:" + pic if pic.startswith("//") else pic
|
||||||
|
detail["vod_pic"] = pic
|
||||||
|
|
||||||
|
# 主演
|
||||||
|
actor_match = re.search(r'<dt>主演[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if actor_match:
|
||||||
|
actors = re.sub(r'<[^>]+>', '', actor_match.group(1)).strip()
|
||||||
|
actors = actors.replace(' ', ' ').replace('\xa0', ' ')
|
||||||
|
detail["vod_actor"] = actors
|
||||||
|
|
||||||
|
# 导演
|
||||||
|
dir_match = re.search(r'<dt>导演[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if dir_match:
|
||||||
|
directors = re.sub(r'<[^>]+>', '', dir_match.group(1)).strip()
|
||||||
|
directors = directors.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||||
|
# 隐晦水印
|
||||||
|
import base64
|
||||||
|
wm = base64.b64decode(b'5pif5rKz').decode('utf-8')
|
||||||
|
if directors:
|
||||||
|
directors = directors + ' ' + wm
|
||||||
|
else:
|
||||||
|
directors = wm
|
||||||
|
detail["vod_director"] = directors
|
||||||
|
|
||||||
|
# 类型
|
||||||
|
type_match = re.search(r'<dt>类型[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if type_match:
|
||||||
|
type_name = re.sub(r'<[^>]+>', '', type_match.group(1)).strip()
|
||||||
|
type_name = type_name.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||||
|
detail["type_name"] = type_name
|
||||||
|
|
||||||
|
# 地区
|
||||||
|
area_match = re.search(r'<dt>地区[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if area_match:
|
||||||
|
area = re.sub(r'<[^>]+>', '', area_match.group(1)).strip()
|
||||||
|
area = area.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||||
|
detail["vod_area"] = area
|
||||||
|
|
||||||
|
# 年代
|
||||||
|
year_match = re.search(r'<dt>年代[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if year_match:
|
||||||
|
year = re.sub(r'<[^>]+>', '', year_match.group(1)).strip()
|
||||||
|
year = year.replace(' ', ' ').replace('\xa0', ' ').strip()
|
||||||
|
detail["vod_year"] = year
|
||||||
|
|
||||||
|
# 备注/状态
|
||||||
|
remark_match = re.search(r'<dt>备注[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if remark_match:
|
||||||
|
detail["vod_remarks"] = re.sub(r'<[^>]+>', '', remark_match.group(1)).strip()
|
||||||
|
|
||||||
|
# 简介
|
||||||
|
desc_match = re.search(r'<dt>剧情[::]</dt>\s*<dd>(.*?)</dd>', html, re.DOTALL)
|
||||||
|
if desc_match:
|
||||||
|
content = re.sub(r'<[^>]+>', '', desc_match.group(1)).strip()
|
||||||
|
content = content.replace("详细", "").strip()
|
||||||
|
detail["vod_content"] = content
|
||||||
|
|
||||||
|
# 播放源和播放地址
|
||||||
|
# 1. 从 tab2 提取线路 id -> 名称 映射 (按顺序)
|
||||||
|
source_order = []
|
||||||
|
tab_match = re.search(r'class="tab2">(.*?)</dt>', html, re.DOTALL)
|
||||||
|
if tab_match:
|
||||||
|
tab_html = tab_match.group(1)
|
||||||
|
source_pattern = r'<span[^>]*id="([^"]+)"[^>]*>([^<]+)</span>'
|
||||||
|
source_matches = re.findall(source_pattern, tab_html)
|
||||||
|
for sid, sname in source_matches:
|
||||||
|
source_order.append((sid, sname.strip()))
|
||||||
|
|
||||||
|
# 2. 定位到 content 区域
|
||||||
|
source_eps = {}
|
||||||
|
content_match = re.search(r'<div[^>]*id="content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
|
||||||
|
if content_match:
|
||||||
|
content = content_match.group(1)
|
||||||
|
|
||||||
|
# 用 <dd 分割,逐个处理
|
||||||
|
dd_parts = re.split(r'<dd\s+', content)
|
||||||
|
for part in dd_parts[1:]:
|
||||||
|
class_match = re.search(r'class="([^"]+)"', part)
|
||||||
|
if not class_match:
|
||||||
|
continue
|
||||||
|
dd_class = class_match.group(1)
|
||||||
|
|
||||||
|
if 'mod' not in dd_class:
|
||||||
|
continue
|
||||||
|
|
||||||
|
source_key = dd_class.replace('mod', '').strip()
|
||||||
|
if not source_key:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# 找这个 dd 里的所有集数
|
||||||
|
ep_pattern = r'href="/ucupy/(\d+)-(\d+)-(\d+)\.html"[^>]*>([^<]+)<'
|
||||||
|
ep_matches = re.findall(ep_pattern, part)
|
||||||
|
|
||||||
|
if ep_matches:
|
||||||
|
eps = []
|
||||||
|
for evid, esid, enid, ename in ep_matches:
|
||||||
|
eps.append(f"{ename.strip()}${evid}-{esid}-{enid}")
|
||||||
|
source_eps[source_key] = "#".join(eps)
|
||||||
|
|
||||||
|
# 3. 按 source_order 的顺序组装结果
|
||||||
|
play_from_list = []
|
||||||
|
play_url_list = []
|
||||||
|
for source_key, source_name in source_order:
|
||||||
|
if source_key in source_eps:
|
||||||
|
play_from_list.append(source_name)
|
||||||
|
play_url_list.append(source_eps[source_key])
|
||||||
|
|
||||||
|
if play_from_list:
|
||||||
|
detail["vod_play_from"] = "$$$".join(play_from_list)
|
||||||
|
detail["vod_play_url"] = "$$$".join(play_url_list)
|
||||||
|
|
||||||
|
# 如果没找到播放列表,用默认线路名
|
||||||
|
if not detail["vod_play_from"]:
|
||||||
|
detail["vod_play_from"] = "速播大屏"
|
||||||
|
detail["vod_play_url"] = f"第01集${vid}-1-1"
|
||||||
|
|
||||||
|
return detail
|
||||||
|
|
||||||
|
def _parse_play_url(self, html):
|
||||||
|
"""解析播放地址"""
|
||||||
|
# 从 player_aaaa 变量中提取
|
||||||
|
idx = html.find('player_aaaa=')
|
||||||
|
if idx >= 0:
|
||||||
|
eq_idx = html.find('=', idx)
|
||||||
|
if eq_idx >= 0:
|
||||||
|
end_idx = html.find('</script>', eq_idx)
|
||||||
|
if end_idx > 0:
|
||||||
|
json_str = html[eq_idx+1:end_idx].strip()
|
||||||
|
if json_str.endswith(';'):
|
||||||
|
json_str = json_str[:-1].strip()
|
||||||
|
try:
|
||||||
|
data = json.loads(json_str)
|
||||||
|
url = data.get("url", "")
|
||||||
|
if url:
|
||||||
|
return url
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# 备用:直接找 m3u8
|
||||||
|
url_match = re.search(r'"url"\s*:\s*"(https?://[^"]+\.m3u8[^"]*)"', html)
|
||||||
|
if url_match:
|
||||||
|
url = url_match.group(1).replace("\\/", "/").replace("\\", "")
|
||||||
|
return url
|
||||||
|
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""测试用"""
|
||||||
|
spider = Spider()
|
||||||
|
spider.init('{}')
|
||||||
|
|
||||||
|
print("=== homeContent 测试 ===")
|
||||||
|
result = spider.homeContent({})
|
||||||
|
print(f"分类数: {len(result['class'])}")
|
||||||
|
print(f"首页推荐: {len(result['list'])} 个")
|
||||||
|
for item in result['list'][:3]:
|
||||||
|
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("=== categoryContent 测试 (电视剧) ===")
|
||||||
|
result = spider.categoryContent("2", "1", "", {})
|
||||||
|
print(f"结果数: {len(result['list'])} 个")
|
||||||
|
for item in result['list'][:3]:
|
||||||
|
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||||
|
|
||||||
|
print()
|
||||||
|
print("=== searchContent 测试 (千香) ===")
|
||||||
|
result = spider.searchContent("千香", False, "1")
|
||||||
|
print(f"结果数: {len(result['list'])} 个")
|
||||||
|
for item in result['list'][:3]:
|
||||||
|
print(f" {item['vod_id']}: {item['vod_name']}")
|
||||||
|
|
||||||
|
if result["list"]:
|
||||||
|
vid = result["list"][0]["vod_id"]
|
||||||
|
print(f"\n=== detailContent 测试 ({vid}) ===")
|
||||||
|
detail_result = spider.detailContent([vid])
|
||||||
|
if detail_result["list"]:
|
||||||
|
d = detail_result["list"][0]
|
||||||
|
print(f" 标题: {d['vod_name']}")
|
||||||
|
print(f" 主演: {d['vod_actor'][:50]}...")
|
||||||
|
print(f" 线路: {d['vod_play_from']}")
|
||||||
|
sources = d['vod_play_from'].split('$$$')
|
||||||
|
print(f" 线路数: {len(sources)}")
|
||||||
|
|
||||||
|
# 播放测试
|
||||||
|
urls = d['vod_play_url'].split('$$$')
|
||||||
|
first_ep = urls[0].split('#')[0]
|
||||||
|
ep_id = first_ep.split('$')[1] if '$' in first_ep else first_ep
|
||||||
|
print(f"\n=== playerContent 测试 ({ep_id}) ===")
|
||||||
|
play_result = spider.playerContent(sources[0], ep_id, [])
|
||||||
|
print(f" parse: {play_result['parse']}")
|
||||||
|
print(f" url: {play_result['url'][:80] if play_result['url'] else '无'}...")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
+608
@@ -0,0 +1,608 @@
|
|||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
import re
|
||||||
|
import json
|
||||||
|
import hashlib
|
||||||
|
import base64
|
||||||
|
import time
|
||||||
|
from urllib.parse import quote
|
||||||
|
from base.spider import Spider
|
||||||
|
|
||||||
|
# ---------- 纯Python实现的RC4和AES(无需第三方库) ----------
|
||||||
|
def rc4_crypt(data, key):
|
||||||
|
S = list(range(256))
|
||||||
|
j = 0
|
||||||
|
for i in range(256):
|
||||||
|
j = (j + S[i] + key[i % len(key)]) % 256
|
||||||
|
S[i], S[j] = S[j], S[i]
|
||||||
|
i = j = 0
|
||||||
|
out = bytearray()
|
||||||
|
for ch in data:
|
||||||
|
i = (i + 1) % 256
|
||||||
|
j = (j + S[i]) % 256
|
||||||
|
S[i], S[j] = S[j], S[i]
|
||||||
|
out.append(ch ^ S[(S[i] + S[j]) % 256])
|
||||||
|
return out
|
||||||
|
|
||||||
|
def aes_cbc_decrypt(data, key, iv):
|
||||||
|
raise NotImplementedError("AES解密需要 pycryptodome 库")
|
||||||
|
|
||||||
|
# 尝试导入官方库
|
||||||
|
try:
|
||||||
|
from Crypto.Cipher import ARC4, AES
|
||||||
|
from Crypto.Util.Padding import unpad
|
||||||
|
def rc4_crypt(data, key):
|
||||||
|
cipher = ARC4.new(key)
|
||||||
|
return cipher.decrypt(data)
|
||||||
|
def aes_cbc_decrypt(data, key, iv):
|
||||||
|
cipher = AES.new(key.encode('utf-8'), AES.MODE_CBC, iv.encode('utf-8'))
|
||||||
|
decrypted = unpad(cipher.decrypt(base64.b64decode(data)), AES.block_size)
|
||||||
|
return decrypted.decode('utf-8')
|
||||||
|
CRYPTO_AVAILABLE = True
|
||||||
|
except ImportError:
|
||||||
|
CRYPTO_AVAILABLE = False
|
||||||
|
print("[歪比影视] 提示: pycryptodome未安装,播放解密可能失败")
|
||||||
|
|
||||||
|
class Spider(Spider):
|
||||||
|
BASE_URL = "https://wbbb1.com"
|
||||||
|
PARSE_DOMAIN = "xn--qvr2v.850088.xyz"
|
||||||
|
|
||||||
|
def_headers = {
|
||||||
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/148.0.0.0 Safari/537.36",
|
||||||
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
|
||||||
|
"Accept-Language": "zh-CN,zh;q=0.9",
|
||||||
|
"Referer": BASE_URL + "/",
|
||||||
|
"Connection": "keep-alive",
|
||||||
|
}
|
||||||
|
|
||||||
|
play_headers = {
|
||||||
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
||||||
|
"Referer": BASE_URL + "/",
|
||||||
|
"Accept": "*/*",
|
||||||
|
}
|
||||||
|
|
||||||
|
CATEGORY_MAP = {"1": "1", "2": "2", "3": "3", "4": "4"}
|
||||||
|
CATEGORY_NAMES = {"1": "电影", "2": "剧集", "3": "动漫", "4": "综艺"}
|
||||||
|
|
||||||
|
_cookies = ""
|
||||||
|
|
||||||
|
# ---------- 加密工具 ----------
|
||||||
|
def _md5(self, s):
|
||||||
|
return hashlib.md5(s.encode('utf-8')).hexdigest()
|
||||||
|
|
||||||
|
def _rc4_encrypt(self, data, key):
|
||||||
|
key_bytes = key.encode('utf-8') if isinstance(key, str) else key
|
||||||
|
data_bytes = data.encode('utf-8') if isinstance(data, str) else data
|
||||||
|
encrypted = rc4_crypt(data_bytes, key_bytes)
|
||||||
|
return base64.b64encode(encrypted).decode('utf-8')
|
||||||
|
|
||||||
|
def _rc4_decrypt(self, data, key):
|
||||||
|
key_bytes = key.encode('utf-8') if isinstance(key, str) else key
|
||||||
|
data_bytes = base64.b64decode(data)
|
||||||
|
decrypted = rc4_crypt(data_bytes, key_bytes)
|
||||||
|
return decrypted.decode('utf-8')
|
||||||
|
|
||||||
|
def _aes_decrypt(self, data, key, iv):
|
||||||
|
if not CRYPTO_AVAILABLE:
|
||||||
|
raise Exception("AES解密需要 pycryptodome 库")
|
||||||
|
return aes_cbc_decrypt(data, key, iv)
|
||||||
|
|
||||||
|
# ---------- 页面请求 ----------
|
||||||
|
def _fetch_cookies(self):
|
||||||
|
try:
|
||||||
|
headers = {"User-Agent": self.def_headers["User-Agent"], "Accept": "text/html", "Referer": self.BASE_URL + "/"}
|
||||||
|
resp = self.fetch(self.BASE_URL, headers=headers)
|
||||||
|
cookie_list = []
|
||||||
|
# 尝试多种方式获取Set-Cookie
|
||||||
|
if hasattr(resp, 'cookies') and resp.cookies:
|
||||||
|
try:
|
||||||
|
for cookie in resp.cookies:
|
||||||
|
cookie_list.append(f"{cookie.name}={cookie.value}")
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
if not cookie_list:
|
||||||
|
if hasattr(resp.headers, "get_all"):
|
||||||
|
raw = resp.headers.get_all("Set-Cookie")
|
||||||
|
for c in raw:
|
||||||
|
cookie_list.append(c.split(";")[0])
|
||||||
|
elif "Set-Cookie" in resp.headers:
|
||||||
|
raw = resp.headers["Set-Cookie"]
|
||||||
|
if isinstance(raw, list):
|
||||||
|
for c in raw:
|
||||||
|
cookie_list.append(c.split(";")[0])
|
||||||
|
else:
|
||||||
|
cookie_list.append(raw.split(";")[0])
|
||||||
|
self._cookies = "; ".join(cookie_list)
|
||||||
|
if self._cookies:
|
||||||
|
print(f"[歪比影视] Cookie获取成功")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] Cookie获取失败: {e}")
|
||||||
|
self._cookies = ""
|
||||||
|
|
||||||
|
def _fetch_html(self, url):
|
||||||
|
headers = self.def_headers.copy()
|
||||||
|
if self._cookies:
|
||||||
|
headers["Cookie"] = self._cookies
|
||||||
|
try:
|
||||||
|
resp = self.fetch(url, headers=headers)
|
||||||
|
return resp.text
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] 请求失败: {url}, {e}")
|
||||||
|
# 如果失败且没有cookie,尝试重新获取cookie后重试
|
||||||
|
if not self._cookies:
|
||||||
|
print("[歪比影视] 尝试重新获取Cookie...")
|
||||||
|
self._fetch_cookies()
|
||||||
|
if self._cookies:
|
||||||
|
headers["Cookie"] = self._cookies
|
||||||
|
try:
|
||||||
|
resp = self.fetch(url, headers=headers)
|
||||||
|
return resp.text
|
||||||
|
except Exception as e2:
|
||||||
|
print(f"[歪比影视] 重试失败: {e2}")
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# ---------- 工具函数:清理HTML标签 ----------
|
||||||
|
def _clean_html(self, text):
|
||||||
|
if not text:
|
||||||
|
return ""
|
||||||
|
text = re.sub(r'<[^>]+>', '', text)
|
||||||
|
text = re.sub(r'\s+', ' ', text).strip()
|
||||||
|
return text
|
||||||
|
|
||||||
|
# ---------- 解析函数 ----------
|
||||||
|
def _parse_video_list(self, html):
|
||||||
|
"""通用解析(用于首页/分类页)"""
|
||||||
|
videos = []
|
||||||
|
# 使用更宽松的正则匹配,兼容class顺序变化
|
||||||
|
pattern = r'<a[^>]*href="/detail/(\d+\.html)"[^>]*class="[^"]*module-poster-item[^"]*"[^>]*>.*?<div[^>]*class="[^"]*module-item-note[^"]*"[^>]*>([^<]*)</div>.*?<img[^>]*data-original="([^"]+)"[^>]*>.*?<div[^>]*class="[^"]*module-poster-item-title[^"]*"[^>]*>([^<]*)</div>'
|
||||||
|
for match in re.finditer(pattern, html, re.DOTALL):
|
||||||
|
detail_url = match.group(1)
|
||||||
|
vod_id = detail_url.replace(".html", "")
|
||||||
|
vod_pic = match.group(3).strip()
|
||||||
|
# 处理协议相对URL
|
||||||
|
if vod_pic.startswith("//"):
|
||||||
|
vod_pic = "https:" + vod_pic
|
||||||
|
videos.append({
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": match.group(4).strip(),
|
||||||
|
"vod_pic": vod_pic,
|
||||||
|
"vod_remarks": match.group(2).strip(),
|
||||||
|
})
|
||||||
|
return videos
|
||||||
|
|
||||||
|
def _parse_search_list(self, html):
|
||||||
|
"""专门解析搜索页(参考JS选择器)"""
|
||||||
|
videos = []
|
||||||
|
# 使用更宽松的正则匹配
|
||||||
|
items = re.finditer(r'<div[^>]*class="[^"]*module-card-item[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*module-card-item|$)', html, re.DOTALL)
|
||||||
|
for item in items:
|
||||||
|
block = item.group(1)
|
||||||
|
# 提取链接
|
||||||
|
link_match = re.search(r'<a[^>]*href="(/detail/[^"]+\.html)"', block)
|
||||||
|
if not link_match:
|
||||||
|
continue
|
||||||
|
vod_id = link_match.group(1).replace("/detail/", "").replace(".html", "")
|
||||||
|
# 提取标题
|
||||||
|
title_match = re.search(r'<div[^>]*class="[^"]*module-card-item-title[^"]*"[^>]*>.*?<strong>([^<]*)</strong>', block, re.DOTALL)
|
||||||
|
vod_name = title_match.group(1).strip() if title_match else "未知"
|
||||||
|
# 提取图片
|
||||||
|
pic_match = re.search(r'<img[^>]*data-original="([^"]+)"', block)
|
||||||
|
vod_pic = pic_match.group(1) if pic_match else ""
|
||||||
|
if vod_pic and vod_pic.startswith("//"):
|
||||||
|
vod_pic = "https:" + vod_pic
|
||||||
|
# 提取备注
|
||||||
|
note_match = re.search(r'<div[^>]*class="[^"]*module-item-note[^"]*"[^>]*>([^<]*)</div>', block)
|
||||||
|
vod_remarks = note_match.group(1).strip() if note_match else ""
|
||||||
|
videos.append({
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": vod_name,
|
||||||
|
"vod_pic": vod_pic,
|
||||||
|
"vod_remarks": vod_remarks,
|
||||||
|
})
|
||||||
|
return videos
|
||||||
|
|
||||||
|
def _parse_play_sources(self, html, vod_id):
|
||||||
|
"""
|
||||||
|
提取每个线路的集数,线路名称准确提取
|
||||||
|
"""
|
||||||
|
sources = []
|
||||||
|
|
||||||
|
# 1. 提取线路名称(按顺序)- 优先使用 data-dropdown-value
|
||||||
|
source_names = []
|
||||||
|
for m in re.finditer(r'data-dropdown-value="([^"]+)"', html):
|
||||||
|
name = m.group(1).strip()
|
||||||
|
if name and name not in source_names:
|
||||||
|
source_names.append(name)
|
||||||
|
if not source_names:
|
||||||
|
tab_box = re.search(r'<div[^>]*class="[^"]*module-tab-items-box[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||||
|
if tab_box:
|
||||||
|
for m in re.finditer(r'<div[^>]*class="[^"]*tab-item[^"]*"[^>]*>.*?<span[^>]*>([^<]*)</span>', tab_box.group(1)):
|
||||||
|
name = m.group(1).strip()
|
||||||
|
if name and name not in source_names:
|
||||||
|
source_names.append(name)
|
||||||
|
if not source_names:
|
||||||
|
for m in re.finditer(r'<span[^>]*class="[^"]*module-tab-value"[^>]*>([^<]*)</span>', html):
|
||||||
|
name = m.group(1).strip()
|
||||||
|
if name and name not in source_names:
|
||||||
|
source_names.append(name)
|
||||||
|
if not source_names:
|
||||||
|
source_names = ["默认"]
|
||||||
|
|
||||||
|
# 2. 提取所有 module-list 块
|
||||||
|
list_blocks = []
|
||||||
|
start_tag = '<div class="module-list sort-list tab-list his-tab-list" id="panel1">'
|
||||||
|
pos = 0
|
||||||
|
while True:
|
||||||
|
start = html.find(start_tag, pos)
|
||||||
|
if start == -1:
|
||||||
|
break
|
||||||
|
start += len(start_tag)
|
||||||
|
depth = 0
|
||||||
|
end = None
|
||||||
|
i = start
|
||||||
|
while i < len(html):
|
||||||
|
if html[i:i+5] == '<div ':
|
||||||
|
depth += 1
|
||||||
|
i += 5
|
||||||
|
elif html[i:i+6] == '</div>':
|
||||||
|
if depth == 0:
|
||||||
|
end = i + 6
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
depth -= 1
|
||||||
|
i += 6
|
||||||
|
else:
|
||||||
|
i += 1
|
||||||
|
if end is not None:
|
||||||
|
block_html = html[start:end]
|
||||||
|
if '/vplay/' in block_html:
|
||||||
|
list_blocks.append(block_html)
|
||||||
|
pos = end
|
||||||
|
else:
|
||||||
|
pos = start + 1
|
||||||
|
|
||||||
|
if not list_blocks:
|
||||||
|
blocks = []
|
||||||
|
start_tag2 = '<div class="module-play-list"'
|
||||||
|
pos = 0
|
||||||
|
while True:
|
||||||
|
start = html.find(start_tag2, pos)
|
||||||
|
if start == -1:
|
||||||
|
break
|
||||||
|
start += len(start_tag2)
|
||||||
|
depth = 0
|
||||||
|
end = None
|
||||||
|
i = start
|
||||||
|
while i < len(html):
|
||||||
|
if html[i:i+5] == '<div ':
|
||||||
|
depth += 1
|
||||||
|
i += 5
|
||||||
|
elif html[i:i+6] == '</div>':
|
||||||
|
if depth == 0:
|
||||||
|
end = i + 6
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
depth -= 1
|
||||||
|
i += 6
|
||||||
|
else:
|
||||||
|
i += 1
|
||||||
|
if end is not None:
|
||||||
|
block_html = html[start:end]
|
||||||
|
if '/vplay/' in block_html:
|
||||||
|
list_blocks.append(block_html)
|
||||||
|
pos = end
|
||||||
|
else:
|
||||||
|
pos = start + 1
|
||||||
|
|
||||||
|
if not list_blocks:
|
||||||
|
simple = re.findall(r'<div class="module-play-list"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||||
|
for b in simple:
|
||||||
|
if '/vplay/' in b:
|
||||||
|
list_blocks.append(b)
|
||||||
|
|
||||||
|
# 3. 对齐名称与块数量
|
||||||
|
if len(source_names) > len(list_blocks):
|
||||||
|
source_names = source_names[:len(list_blocks)]
|
||||||
|
while len(source_names) < len(list_blocks):
|
||||||
|
source_names.append(f"源{len(source_names)+1}")
|
||||||
|
|
||||||
|
# 4. 解析每个块的集数
|
||||||
|
for idx, block in enumerate(list_blocks):
|
||||||
|
eps = []
|
||||||
|
for m in re.finditer(r'<a[^>]*href="(/vplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>.*?<span>([^<]*)</span>', block):
|
||||||
|
link = m.group(1)
|
||||||
|
id_ = m.group(2)
|
||||||
|
sid = m.group(3)
|
||||||
|
nid = m.group(4)
|
||||||
|
name = m.group(5).strip()
|
||||||
|
eps.append({"name": name, "link": f"{id_}-{sid}-{nid}"})
|
||||||
|
if eps:
|
||||||
|
sources.append({
|
||||||
|
"source_name": source_names[idx] if idx < len(source_names) else f"源{idx+1}",
|
||||||
|
"episodes": eps
|
||||||
|
})
|
||||||
|
|
||||||
|
if not sources:
|
||||||
|
eps = []
|
||||||
|
for m in re.finditer(r'<a[^>]*href="(/vplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>.*?<span>([^<]*)</span>', html):
|
||||||
|
link = m.group(1)
|
||||||
|
id_ = m.group(2)
|
||||||
|
sid = m.group(3)
|
||||||
|
nid = m.group(4)
|
||||||
|
name = m.group(5).strip()
|
||||||
|
eps.append({"name": name, "link": f"{id_}-{sid}-{nid}"})
|
||||||
|
if eps:
|
||||||
|
sources.append({"source_name": "默认", "episodes": eps})
|
||||||
|
|
||||||
|
return sources
|
||||||
|
|
||||||
|
# ---------- 播放地址获取 ----------
|
||||||
|
def _get_play_url(self, vod_id, sid, nid):
|
||||||
|
try:
|
||||||
|
play_page = f"{self.BASE_URL}/vplay/{vod_id}-{sid}-{nid}.html"
|
||||||
|
html = self._fetch_html(play_page)
|
||||||
|
if not html:
|
||||||
|
return None
|
||||||
|
|
||||||
|
# 检查iframe
|
||||||
|
iframe_match = re.search(r'<iframe[^>]*src="([^"]+)"', html)
|
||||||
|
if iframe_match:
|
||||||
|
iframe_url = iframe_match.group(1)
|
||||||
|
if iframe_url.startswith('//'):
|
||||||
|
iframe_url = 'https:' + iframe_url
|
||||||
|
elif iframe_url.startswith('/'):
|
||||||
|
iframe_url = self.BASE_URL + iframe_url
|
||||||
|
print(f"[歪比影视] 发现iframe: {iframe_url}")
|
||||||
|
return iframe_url
|
||||||
|
|
||||||
|
# 提取加密URL - 使用更安全的大括号匹配算法,支持嵌套JSON
|
||||||
|
enc_url = None
|
||||||
|
m1 = re.search(r'var\s+player_aaaa\s*=\s*', html)
|
||||||
|
if m1:
|
||||||
|
start = m1.end()
|
||||||
|
# 跳过空白字符
|
||||||
|
while start < len(html) and html[start] in ' \t\n\r':
|
||||||
|
start += 1
|
||||||
|
# 找到匹配的大括号
|
||||||
|
if start < len(html) and html[start] == '{':
|
||||||
|
depth = 1
|
||||||
|
i = start + 1
|
||||||
|
while i < len(html) and depth > 0:
|
||||||
|
if html[i] == '{':
|
||||||
|
depth += 1
|
||||||
|
elif html[i] == '}':
|
||||||
|
depth -= 1
|
||||||
|
i += 1
|
||||||
|
if depth == 0:
|
||||||
|
json_str = html[start:i]
|
||||||
|
try:
|
||||||
|
player_data = json.loads(json_str)
|
||||||
|
enc_url = player_data.get("url", "")
|
||||||
|
print(f"[歪比影视] 提取到player_aaaa.url: {enc_url[:50]}...")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] player_aaaa JSON解析失败: {e}")
|
||||||
|
|
||||||
|
# 兜底:直接搜索url字段
|
||||||
|
if not enc_url:
|
||||||
|
m2 = re.search(r'"url"\s*:\s*"([^"]+)"', html)
|
||||||
|
if m2:
|
||||||
|
enc_url = m2.group(1)
|
||||||
|
print(f"[歪比影视] 兜底提取到url: {enc_url[:50]}...")
|
||||||
|
|
||||||
|
if not enc_url:
|
||||||
|
print("[歪比影视] 未能提取到加密URL")
|
||||||
|
return None
|
||||||
|
|
||||||
|
# 如果无加密库,返回解析页面
|
||||||
|
if not CRYPTO_AVAILABLE:
|
||||||
|
fallback_url = f"https://{self.PARSE_DOMAIN}/player/?url={enc_url}&next=//&title="
|
||||||
|
print(f"[歪比影视] 无加密库,使用解析页面: {fallback_url}")
|
||||||
|
return fallback_url
|
||||||
|
|
||||||
|
try:
|
||||||
|
domain = self.PARSE_DOMAIN
|
||||||
|
l = (self._md5(enc_url) + " P")[-22:]
|
||||||
|
key = l.encode('utf-8')
|
||||||
|
h = self._rc4_encrypt(self._md5(enc_url + "stray"), key)
|
||||||
|
timestamp = str(int(time.time()))
|
||||||
|
u = self._rc4_encrypt(timestamp + self._md5(key.decode('utf-8') + "stray"), key)
|
||||||
|
y = self._rc4_encrypt(self._md5(domain + "stray"), key)
|
||||||
|
|
||||||
|
api_url = f"https://{domain}/player/api.php"
|
||||||
|
headers = {
|
||||||
|
"User-Agent": self.play_headers["User-Agent"],
|
||||||
|
"Accept": "application/json, text/javascript, */*; q=0.01",
|
||||||
|
"Origin": f"https://{domain}",
|
||||||
|
"Referer": f"https://{domain}/player/?url={enc_url}",
|
||||||
|
"X-Requested-With": "XMLHttpRequest",
|
||||||
|
"Content-Type": "application/x-www-form-urlencoded"
|
||||||
|
}
|
||||||
|
if self._cookies:
|
||||||
|
headers["Cookie"] = self._cookies
|
||||||
|
|
||||||
|
post_data = {"url": enc_url, "key": h, "vkey": u, "ckey": y}
|
||||||
|
print(f"[歪比影视] 请求API: {api_url}")
|
||||||
|
resp = self.post(api_url, data=post_data, headers=headers)
|
||||||
|
result = json.loads(resp.text)
|
||||||
|
print(f"[歪比影视] API返回: code={result.get('code')}")
|
||||||
|
|
||||||
|
if result.get("code") == 200:
|
||||||
|
aes_key = self._rc4_decrypt(result["aes_key"], key)
|
||||||
|
aes_iv = self._rc4_decrypt(result["aes_iv"], key)
|
||||||
|
enc_play = result["url"]
|
||||||
|
play_url = self._aes_decrypt(enc_play, aes_key, aes_iv)
|
||||||
|
print(f"[歪比影视] 解密成功: {play_url[:80]}...")
|
||||||
|
return play_url
|
||||||
|
else:
|
||||||
|
fallback_url = f"https://{domain}/player/?url={enc_url}&next=//&title="
|
||||||
|
print(f"[歪比影视] API返回非200,使用解析页面: {fallback_url}")
|
||||||
|
return fallback_url
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] 解密异常: {e}")
|
||||||
|
fallback_url = f"https://{self.PARSE_DOMAIN}/player/?url={enc_url}&next=//&title="
|
||||||
|
return fallback_url
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] 获取播放地址异常: {e}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
# ---------- TVBox接口 ----------
|
||||||
|
def init(self, extend=''):
|
||||||
|
self._fetch_cookies()
|
||||||
|
print("[歪比影视] 初始化完成")
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
return {"class": [{"type_id": tid, "type_name": name} for tid, name in self.CATEGORY_NAMES.items()]}
|
||||||
|
|
||||||
|
def homeVideoContent(self):
|
||||||
|
try:
|
||||||
|
html = self._fetch_html(self.BASE_URL)
|
||||||
|
if not html:
|
||||||
|
return {"list": []}
|
||||||
|
# 尝试匹配"正在热映"
|
||||||
|
block = re.search(r'<div class="module">.*?<h2[^>]*class="[^"]*module-title[^"]*"[^>]*>正在热映.*?</div>(.*?)</div>\s*<div class="module">', html, re.DOTALL)
|
||||||
|
if not block:
|
||||||
|
# 兜底:尝试匹配第一个module块
|
||||||
|
block = re.search(r'<div class="module">(.*?)</div>\s*<div class="module">', html, re.DOTALL)
|
||||||
|
if not block:
|
||||||
|
return {"list": []}
|
||||||
|
videos = self._parse_video_list(block.group(1))
|
||||||
|
return {"list": videos[:20]}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] homeVideoContent 异常: {e}")
|
||||||
|
return {"list": []}
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
try:
|
||||||
|
pg = int(pg)
|
||||||
|
if tid not in self.CATEGORY_MAP:
|
||||||
|
return {"list": [], "pagecount": 1, "page": pg}
|
||||||
|
if pg == 1:
|
||||||
|
url = f"{self.BASE_URL}/show/{tid}-----------.html"
|
||||||
|
else:
|
||||||
|
url = f"{self.BASE_URL}/show/{tid}--------{pg}---.html"
|
||||||
|
print(f"[歪比影视] 分类请求: {url}")
|
||||||
|
html = self._fetch_html(url)
|
||||||
|
if not html:
|
||||||
|
return {"list": [], "pagecount": 1, "page": pg}
|
||||||
|
videos = self._parse_video_list(html)
|
||||||
|
last = re.search(r'<a[^>]*href="/show/\d+--------(\d+)---\.html"[^>]*>尾页</a>', html)
|
||||||
|
pagecount = int(last.group(1)) if last else 1
|
||||||
|
return {"list": videos, "pagecount": pagecount, "page": pg}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] categoryContent 异常: {e}")
|
||||||
|
return {"list": [], "pagecount": 1, "page": pg}
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg='1'):
|
||||||
|
try:
|
||||||
|
pg = int(pg)
|
||||||
|
# 对搜索关键词进行URL编码,修复中文搜索失败
|
||||||
|
encoded_key = quote(key)
|
||||||
|
url = f"{self.BASE_URL}/search/{encoded_key}-------------.html"
|
||||||
|
print(f"[歪比影视] 搜索请求: {url}")
|
||||||
|
html = self._fetch_html(url)
|
||||||
|
if not html:
|
||||||
|
return {"list": [], "page": pg}
|
||||||
|
videos = self._parse_search_list(html)
|
||||||
|
return {"list": videos, "page": pg}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] searchContent 异常: {e}")
|
||||||
|
return {"list": [], "page": pg}
|
||||||
|
|
||||||
|
def detailContent(self, ids):
|
||||||
|
try:
|
||||||
|
vod_id = ids[0]
|
||||||
|
url = f"{self.BASE_URL}/detail/{vod_id}.html"
|
||||||
|
print(f"[歪比影视] 详情请求: {url}")
|
||||||
|
html = self._fetch_html(url)
|
||||||
|
if not html:
|
||||||
|
return {"list": []}
|
||||||
|
|
||||||
|
title = re.search(r'<h1>([^<]*)</h1>', html)
|
||||||
|
vod_name = title.group(1).strip() if title else "未知"
|
||||||
|
|
||||||
|
# 更宽松的图片匹配,兼容class变化
|
||||||
|
pic = re.search(r'<div[^>]*class="[^"]*module-item-pic[^"]*"[^>]*>.*?<img[^>]*data-original="([^"]+)"', html, re.DOTALL)
|
||||||
|
vod_pic = pic.group(1) if pic else ""
|
||||||
|
if vod_pic.startswith("//"):
|
||||||
|
vod_pic = "https:" + vod_pic
|
||||||
|
|
||||||
|
desc = re.search(r'<div[^>]*class="[^"]*module-info-introduction-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||||
|
vod_content = self._clean_html(desc.group(1)) if desc else ""
|
||||||
|
|
||||||
|
actor = re.search(r'主演:</span>.*?<div[^>]*class="[^"]*module-info-item-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||||
|
vod_actor = self._clean_html(actor.group(1)) if actor else ""
|
||||||
|
|
||||||
|
director = re.search(r'导演:</span>.*?<div[^>]*class="[^"]*module-info-item-content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
|
||||||
|
vod_director = self._clean_html(director.group(1)) if director else ""
|
||||||
|
|
||||||
|
year = re.search(r'<a[^>]*title="(\d{4})"', html)
|
||||||
|
vod_year = year.group(1) if year else ""
|
||||||
|
|
||||||
|
sources = self._parse_play_sources(html, vod_id)
|
||||||
|
if not sources:
|
||||||
|
print(f"[歪比影视] 未能解析到播放源")
|
||||||
|
return {"list": []}
|
||||||
|
|
||||||
|
from_list = []
|
||||||
|
url_list = []
|
||||||
|
for src in sources:
|
||||||
|
from_list.append(src["source_name"])
|
||||||
|
eps_str = "#".join([f"{ep['name']}${ep['link']}" for ep in src["episodes"]])
|
||||||
|
url_list.append(eps_str)
|
||||||
|
|
||||||
|
vod_play_from = "$$$".join(from_list)
|
||||||
|
vod_play_url = "$$$".join(url_list)
|
||||||
|
|
||||||
|
video = {
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": vod_name,
|
||||||
|
"vod_pic": vod_pic,
|
||||||
|
"vod_year": vod_year,
|
||||||
|
"vod_area": "",
|
||||||
|
"vod_actor": vod_actor,
|
||||||
|
"vod_director": vod_director,
|
||||||
|
"vod_content": vod_content,
|
||||||
|
"vod_play_from": vod_play_from,
|
||||||
|
"vod_play_url": vod_play_url,
|
||||||
|
}
|
||||||
|
print(f"[歪比影视] 详情解析成功: {vod_name}, 线路: {vod_play_from}")
|
||||||
|
return {"list": [video]}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] detailContent 异常: {e}")
|
||||||
|
return {"list": []}
|
||||||
|
|
||||||
|
def playerContent(self, flag, vid, vip_flags):
|
||||||
|
try:
|
||||||
|
parts = vid.split("-")
|
||||||
|
if len(parts) != 3:
|
||||||
|
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||||
|
vod_id, sid, nid = parts
|
||||||
|
play_url = self._get_play_url(vod_id, sid, nid)
|
||||||
|
if play_url:
|
||||||
|
# 判断URL类型,决定parse标志
|
||||||
|
# 如果是直接的视频文件或m3u8,parse=0;否则parse=1(需要TVBox嗅探/解析)
|
||||||
|
is_direct = any(ext in play_url.lower() for ext in ['.m3u8', '.mp4', '.flv', '.ts', '.mkv', '.avi', '.mov', '.wmv'])
|
||||||
|
is_direct = is_direct or 'm3u8' in play_url.lower() or 'mp4' in play_url.lower()
|
||||||
|
parse_flag = 0 if is_direct else 1
|
||||||
|
# header必须是JSON字符串
|
||||||
|
header_str = json.dumps(self.play_headers)
|
||||||
|
print(f"[歪比影视] 播放URL: {play_url[:80]}..., parse={parse_flag}")
|
||||||
|
return {"jx": 0, "parse": parse_flag, "url": play_url, "header": header_str}
|
||||||
|
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[歪比影视] playerContent 异常: {e}")
|
||||||
|
return {"jx": 0, "parse": 0, "url": "", "header": ""}
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
return "歪比影视"
|
||||||
|
|
||||||
|
def isVideoFormat(self, url):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def manualVideoCheck(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def destroy(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def localProxy(self, param):
|
||||||
|
pass
|
||||||
@@ -0,0 +1,385 @@
|
|||||||
|
# coding = utf-8
|
||||||
|
#!/usr/bin/python
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
import json
|
||||||
|
import time
|
||||||
|
import urllib.parse
|
||||||
|
from base.spider import Spider
|
||||||
|
|
||||||
|
sys.path.append('..')
|
||||||
|
|
||||||
|
class Spider(Spider):
|
||||||
|
def __init__(self):
|
||||||
|
self.name = "糖豆广场舞"
|
||||||
|
self.host = 'https://api-h5.tangdou.com'
|
||||||
|
self.img_host = 'https://bimg.tangdou.com' # 图片域名前缀
|
||||||
|
self.header = {
|
||||||
|
'Accept': 'application/json, text/plain, */*',
|
||||||
|
'Accept-Encoding': 'gzip, deflate, br',
|
||||||
|
'Accept-Language': 'zh,zh-CN;q=0.9',
|
||||||
|
'Connection': 'keep-alive',
|
||||||
|
'Host': 'api-h5.tangdou.com',
|
||||||
|
'Referer': 'https://www.tangdou.com/',
|
||||||
|
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
|
||||||
|
}
|
||||||
|
# 缓存机制
|
||||||
|
self.cache = {}
|
||||||
|
self.cache_timeout = 300 # 5分钟缓存
|
||||||
|
# 生成UUID (时间戳_随机数格式)
|
||||||
|
self.uuid = f"{int(time.time() * 1000)}_{int(time.time() % 100000)}"
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
return self.name
|
||||||
|
|
||||||
|
def init(self, extend=''):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
result = {}
|
||||||
|
# 糖豆广场舞分类
|
||||||
|
classes = [
|
||||||
|
|
||||||
|
{"type_name": "广场舞", "type_id": "1"},
|
||||||
|
{"type_name": "民族舞", "type_id": "2"},
|
||||||
|
{"type_name": "jazz/现代舞", "type_id": "3"},
|
||||||
|
{"type_name": "健身", "type_id": "4"},
|
||||||
|
{"type_name": "双人舞", "type_id": "5"},
|
||||||
|
{"type_name": "步法", "type_id": "6"},
|
||||||
|
{"type_name": "气球", "type_id": "7"},
|
||||||
|
{"type_name": "瑜伽", "type_id": "8"},
|
||||||
|
{"type_name": "二人转", "type_id": "9"}
|
||||||
|
]
|
||||||
|
|
||||||
|
result['class'] = classes
|
||||||
|
|
||||||
|
# 筛选条件 - 主要按年份和难度筛选
|
||||||
|
filters = {}
|
||||||
|
for cate in classes:
|
||||||
|
tid = cate['type_id']
|
||||||
|
filters[tid] = [
|
||||||
|
{"key": "year", "name": "年份", "value": [
|
||||||
|
{"n": "全部", "v": "0"},
|
||||||
|
{"n": "2026", "v": "2026"},
|
||||||
|
{"n": "2025", "v": "2025"},
|
||||||
|
{"n": "2024", "v": "2024"},
|
||||||
|
{"n": "2023", "v": "2023"},
|
||||||
|
{"n": "2022", "v": "2022"},
|
||||||
|
{"n": "2021", "v": "2021"},
|
||||||
|
{"n": "2020", "v": "2020"},
|
||||||
|
{"n": "2019", "v": "2019"},
|
||||||
|
{"n": "2018", "v": "2018"},
|
||||||
|
{"n": "2017及以前", "v": "2017"}
|
||||||
|
]},
|
||||||
|
{"key": "sort", "name": "排序", "value": [
|
||||||
|
{"n": "最新", "v": "new"},
|
||||||
|
{"n": "最热", "v": "hot"},
|
||||||
|
{"n": "推荐", "v": "rec"}
|
||||||
|
]}
|
||||||
|
]
|
||||||
|
|
||||||
|
result['filters'] = filters
|
||||||
|
return result
|
||||||
|
|
||||||
|
def homeVideoContent(self):
|
||||||
|
# 首页推荐 - 获取feed流
|
||||||
|
videos = []
|
||||||
|
try:
|
||||||
|
cache_key = "home_feed"
|
||||||
|
data = self.get_cached_data(cache_key, 1, 20)
|
||||||
|
|
||||||
|
if data and 'data' in data:
|
||||||
|
for item in data['data']:
|
||||||
|
video = self._parse_video_item(item)
|
||||||
|
if video:
|
||||||
|
videos.append(video)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"获取首页推荐失败: {e}")
|
||||||
|
|
||||||
|
return {'list': videos}
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
videos = []
|
||||||
|
try:
|
||||||
|
# 构建请求参数
|
||||||
|
page_size = 30
|
||||||
|
# 糖豆API: type 0=推荐, 1=广场舞, etc.
|
||||||
|
api_url = f"{self.host}/mtangdou/home/feed?page={pg}&num={page_size}&uuid={self.uuid}"
|
||||||
|
|
||||||
|
# 如果有分类ID且不是推荐,添加分类参数
|
||||||
|
if tid != "0":
|
||||||
|
api_url += f"&type={tid}"
|
||||||
|
|
||||||
|
# 排序参数
|
||||||
|
sort = extend.get('sort', 'new')
|
||||||
|
if sort == 'hot':
|
||||||
|
api_url += "&sort=hot"
|
||||||
|
elif sort == 'rec':
|
||||||
|
api_url += "&sort=rec"
|
||||||
|
|
||||||
|
cache_key = f"category_{tid}_{pg}_{sort}"
|
||||||
|
data = self.fetchData(api_url, cache_key)
|
||||||
|
|
||||||
|
if data and 'data' in data:
|
||||||
|
for item in data['data']:
|
||||||
|
video = self._parse_video_item(item)
|
||||||
|
if video:
|
||||||
|
videos.append(video)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"获取分类内容失败: {e}")
|
||||||
|
|
||||||
|
return {
|
||||||
|
'list': videos,
|
||||||
|
'page': int(pg),
|
||||||
|
'pagecount': 9999, # 糖豆没有明确页数限制
|
||||||
|
'limit': 30,
|
||||||
|
'total': 999999
|
||||||
|
}
|
||||||
|
|
||||||
|
def detailContent(self, ids):
|
||||||
|
try:
|
||||||
|
vid = ids[0].split('||')[0] if '||' in ids[0] else ids[0]
|
||||||
|
|
||||||
|
# 获取视频详情和播放链接
|
||||||
|
play_url_api = f"{self.host}/mtangdou/video/play?vid={vid}&uuid={self.uuid}"
|
||||||
|
share_api = f"{self.host}/sample/share/main?vid={vid}"
|
||||||
|
|
||||||
|
# 尝试获取播放链接
|
||||||
|
play_data = self.fetchData(play_url_api, f"play_{vid}", use_cache=False)
|
||||||
|
|
||||||
|
# 获取详情信息
|
||||||
|
share_data = self.fetchData(share_api, f"share_{vid}", use_cache=False)
|
||||||
|
|
||||||
|
if not share_data or 'data' not in share_data:
|
||||||
|
return {'list': []}
|
||||||
|
|
||||||
|
data_info = share_data['data']
|
||||||
|
|
||||||
|
# 获取简介内容,优先使用API返回的desc或description,否则使用默认简介
|
||||||
|
content = data_info.get('desc', data_info.get('description', '')).strip()
|
||||||
|
if not content:
|
||||||
|
content = '醉卧东风祝您身体健康'
|
||||||
|
|
||||||
|
# 修正:拼接完整缩略图URL
|
||||||
|
cover_path = data_info.get('cover', data_info.get('img', ''))
|
||||||
|
if cover_path and not cover_path.startswith('http'):
|
||||||
|
cover_url = self.img_host + cover_path
|
||||||
|
else:
|
||||||
|
cover_url = cover_path
|
||||||
|
|
||||||
|
# 构建视频详情对象
|
||||||
|
video_detail = {
|
||||||
|
"vod_id": vid,
|
||||||
|
"vod_name": data_info.get('title', '').strip(),
|
||||||
|
"vod_pic": cover_url,
|
||||||
|
"vod_year": str(data_info.get('year', '')),
|
||||||
|
"vod_area": data_info.get('area', '大陆'),
|
||||||
|
"vod_actor": data_info.get('teacher', data_info.get('author', '')),
|
||||||
|
"vod_director": "",
|
||||||
|
"vod_content": content,
|
||||||
|
"vod_play_from": "糖豆播放",
|
||||||
|
"vod_remarks": f"时长: {data_info.get('duration_str', '未知')}" if 'duration_str' in data_info else ""
|
||||||
|
}
|
||||||
|
|
||||||
|
# 构建播放链接 - 糖豆是单视频,没有多集
|
||||||
|
play_url = ""
|
||||||
|
if play_data and 'data' in play_data:
|
||||||
|
play_url = play_data['data'].get('play_url', '')
|
||||||
|
|
||||||
|
# 尝试从share接口获取video_url作为备选
|
||||||
|
if not play_url and 'video_url' in data_info:
|
||||||
|
play_url = data_info['video_url']
|
||||||
|
|
||||||
|
if play_url:
|
||||||
|
# 糖豆视频直接播放,需要处理Referer
|
||||||
|
video_detail["vod_play_url"] = f"{video_detail['vod_name']}${vid}||{play_url}"
|
||||||
|
else:
|
||||||
|
video_detail["vod_play_url"] = ""
|
||||||
|
|
||||||
|
return {'list': [video_detail]}
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"获取详情失败: {e}")
|
||||||
|
return {'list': []}
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg=1):
|
||||||
|
videos = []
|
||||||
|
try:
|
||||||
|
# 糖豆搜索API
|
||||||
|
search_api = f"{self.host}/mtangdou/search?word={urllib.parse.quote(key)}&page={pg}&num=30&uuid={self.uuid}"
|
||||||
|
|
||||||
|
# 搜索不使用缓存,确保实时性
|
||||||
|
data = self.fetchData(search_api, use_cache=False)
|
||||||
|
|
||||||
|
if data and 'data' in data:
|
||||||
|
for item in data['data']:
|
||||||
|
video = self._parse_video_item(item)
|
||||||
|
if video:
|
||||||
|
videos.append(video)
|
||||||
|
else:
|
||||||
|
# 如果搜索API返回空,尝试从首页feed中过滤(仅第一页)
|
||||||
|
if pg == 1:
|
||||||
|
feed_data = self.get_cached_data("search_feed", 1, 100)
|
||||||
|
if feed_data and 'data' in feed_data:
|
||||||
|
key_lower = key.lower()
|
||||||
|
for item in feed_data['data']:
|
||||||
|
title = item.get('title', '').lower()
|
||||||
|
if key_lower in title:
|
||||||
|
video = self._parse_video_item(item)
|
||||||
|
if video:
|
||||||
|
videos.append(video)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"搜索失败: {e}")
|
||||||
|
|
||||||
|
return {
|
||||||
|
'list': videos,
|
||||||
|
'page': int(pg),
|
||||||
|
'pagecount': 9999,
|
||||||
|
'limit': 30,
|
||||||
|
'total': 999999
|
||||||
|
}
|
||||||
|
|
||||||
|
def playerContent(self, flag, id, vipFlags):
|
||||||
|
try:
|
||||||
|
# 解析传入的id: vid||play_url
|
||||||
|
if '||' in id:
|
||||||
|
parts = id.split('||')
|
||||||
|
vid = parts[0]
|
||||||
|
play_url = parts[1] if len(parts) > 1 else ""
|
||||||
|
else:
|
||||||
|
vid = id
|
||||||
|
play_url = ""
|
||||||
|
|
||||||
|
# 如果没有play_url,实时获取
|
||||||
|
if not play_url:
|
||||||
|
play_api = f"{self.host}/mtangdou/video/play?vid={vid}&uuid={self.uuid}"
|
||||||
|
data = self.fetchData(play_api, use_cache=False)
|
||||||
|
if data and 'data' in data:
|
||||||
|
play_url = data['data'].get('play_url', '')
|
||||||
|
|
||||||
|
if play_url:
|
||||||
|
# 糖豆视频需要Referer才能正常播放
|
||||||
|
headers = {
|
||||||
|
"Referer": "https://www.tangdou.com/",
|
||||||
|
"User-Agent": self.header['User-Agent']
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"parse": 0, # 直接播放,不需要解析
|
||||||
|
"playUrl": "",
|
||||||
|
"url": play_url,
|
||||||
|
"header": json.dumps(headers)
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
return {"parse": 0, "playUrl": "", "url": ""}
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"播放解析失败: {e}")
|
||||||
|
return {"parse": 0, "playUrl": "", "url": ""}
|
||||||
|
|
||||||
|
def isVideoFormat(self, url):
|
||||||
|
video_formats = ['.m3u8', '.mp4', '.avi', '.mkv', '.flv', '.ts', '.mov']
|
||||||
|
return any(url.lower().endswith(fmt) for fmt in video_formats)
|
||||||
|
|
||||||
|
def manualVideoCheck(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def localProxy(self, params):
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _parse_video_item(self, item):
|
||||||
|
"""解析视频列表项为统一格式"""
|
||||||
|
try:
|
||||||
|
vid = str(item.get('vid', ''))
|
||||||
|
if not vid:
|
||||||
|
return None
|
||||||
|
|
||||||
|
title = item.get('title', '').strip()
|
||||||
|
|
||||||
|
# 修正:拼接完整缩略图URL,cover是相对路径,需要加域名前缀
|
||||||
|
cover_path = item.get('cover', item.get('img', item.get('video_img', '')))
|
||||||
|
if cover_path and not cover_path.startswith('http'):
|
||||||
|
img = self.img_host + cover_path
|
||||||
|
else:
|
||||||
|
img = cover_path
|
||||||
|
|
||||||
|
duration = item.get('duration_str', '')
|
||||||
|
teacher = item.get('teacher', item.get('author', ''))
|
||||||
|
|
||||||
|
# 是否有附属信息(如老师名字)
|
||||||
|
remarks = duration if duration else teacher
|
||||||
|
|
||||||
|
return {
|
||||||
|
"vod_id": vid,
|
||||||
|
"vod_name": title,
|
||||||
|
"vod_pic": img,
|
||||||
|
"vod_remarks": remarks
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"解析视频项失败: {e}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
def get_cached_data(self, cache_key, page=1, num=30):
|
||||||
|
"""获取首页feed缓存"""
|
||||||
|
current_time = time.time()
|
||||||
|
if cache_key in self.cache:
|
||||||
|
cached_data, timestamp = self.cache[cache_key]
|
||||||
|
if current_time - timestamp < self.cache_timeout:
|
||||||
|
return cached_data
|
||||||
|
|
||||||
|
# 缓存不存在或已过期,重新获取
|
||||||
|
api_url = f"{self.host}/mtangdou/home/feed?page={page}&num={num}&uuid={self.uuid}"
|
||||||
|
result = self.fetchData(api_url, cache_key)
|
||||||
|
if result:
|
||||||
|
self.cache[cache_key] = (result, current_time)
|
||||||
|
return result
|
||||||
|
|
||||||
|
def fetchData(self, url, cache_key=None, use_cache=True):
|
||||||
|
"""封装的数据获取方法,支持缓存"""
|
||||||
|
try:
|
||||||
|
# 如果启用缓存且存在有效缓存
|
||||||
|
current_time = time.time()
|
||||||
|
if use_cache and cache_key and cache_key in self.cache:
|
||||||
|
cached_data, timestamp = self.cache[cache_key]
|
||||||
|
if current_time - timestamp < self.cache_timeout:
|
||||||
|
return cached_data
|
||||||
|
|
||||||
|
start_time = time.time()
|
||||||
|
response = self.fetch(url, headers=self.header)
|
||||||
|
end_time = time.time()
|
||||||
|
|
||||||
|
print(f"请求耗时: {end_time - start_time:.2f}秒, URL: {url[:60]}...")
|
||||||
|
|
||||||
|
if response.status_code != 200:
|
||||||
|
print(f"API请求失败: {response.status_code}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
data = json.loads(response.text)
|
||||||
|
|
||||||
|
# 缓存结果
|
||||||
|
if use_cache and cache_key:
|
||||||
|
self.cache[cache_key] = (data, current_time)
|
||||||
|
|
||||||
|
return data
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"获取数据失败: {e}, URL: {url}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
def fetch(self, url, headers=None):
|
||||||
|
"""发送HTTP GET请求"""
|
||||||
|
import requests
|
||||||
|
try:
|
||||||
|
if headers is None:
|
||||||
|
headers = self.header
|
||||||
|
response = requests.get(url, headers=headers, timeout=10)
|
||||||
|
return response
|
||||||
|
except Exception as e:
|
||||||
|
print(f"请求异常: {e}")
|
||||||
|
# 返回一个模拟的response对象
|
||||||
|
class FakeResponse:
|
||||||
|
status_code = 0
|
||||||
|
text = ""
|
||||||
|
return FakeResponse()
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
pass
|
||||||
+336
@@ -0,0 +1,336 @@
|
|||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
# by @PyramidStore AutoGen
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
sys.path.append('..')
|
||||||
|
import json
|
||||||
|
from urllib.parse import quote
|
||||||
|
from base.spider import Spider
|
||||||
|
|
||||||
|
|
||||||
|
class Spider(Spider):
|
||||||
|
|
||||||
|
def init(self, extend=""):
|
||||||
|
self.nav_host = 'https://www.xiguadh.com'
|
||||||
|
self.headers = {
|
||||||
|
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||||
|
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
|
||||||
|
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
||||||
|
}
|
||||||
|
self.host = self._get_host()
|
||||||
|
|
||||||
|
def _get_host(self):
|
||||||
|
"""获取视频站点 URL,失败时从导航页获取"""
|
||||||
|
default_host = 'https://www.bzzdyy.com'
|
||||||
|
try:
|
||||||
|
r = self.fetch(default_host, headers=self.headers, timeout=10, verify=False)
|
||||||
|
if r.status_code == 200:
|
||||||
|
return default_host
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
r = self.fetch(self.nav_host, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
urls = re.findall(r'url:\s*["\']([^"\']+)["\']', html)
|
||||||
|
for url in urls:
|
||||||
|
if url.startswith('http') and 'xiguadh' not in url:
|
||||||
|
return url.rstrip('/')
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return default_host
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
return '西瓜影院'
|
||||||
|
|
||||||
|
def isVideoFormat(self, url):
|
||||||
|
return False
|
||||||
|
|
||||||
|
def manualVideoCheck(self):
|
||||||
|
return True
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
try:
|
||||||
|
r = self.fetch(self.host, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
# 提取主要分类
|
||||||
|
nav_match = re.search(r'<ul class="stui-header__menu">(.*?)</ul>', html, re.DOTALL)
|
||||||
|
if nav_match:
|
||||||
|
nav_html = nav_match.group(1)
|
||||||
|
categories = re.findall(r'<li[^>]*><a href="/index.php/vod/type/id/(\d+)\.html">([^<]+)</a></li>', nav_html)
|
||||||
|
else:
|
||||||
|
categories = []
|
||||||
|
seen = set()
|
||||||
|
classes = []
|
||||||
|
for tid, name in categories:
|
||||||
|
if tid not in seen:
|
||||||
|
seen.add(tid)
|
||||||
|
classes.append({'type_name': name, 'type_id': tid})
|
||||||
|
if not classes:
|
||||||
|
raise Exception('No categories found')
|
||||||
|
# 提取首页推荐视频
|
||||||
|
videos = self._parse_vodlist(html)
|
||||||
|
except Exception:
|
||||||
|
classes = [
|
||||||
|
{'type_name': '电影', 'type_id': '20'},
|
||||||
|
{'type_name': '连续剧', 'type_id': '37'},
|
||||||
|
{'type_name': '动漫', 'type_id': '43'},
|
||||||
|
{'type_name': '综艺', 'type_id': '45'},
|
||||||
|
{'type_name': 'B站', 'type_id': '47'},
|
||||||
|
{'type_name': '人人专区', 'type_id': '60'},
|
||||||
|
]
|
||||||
|
videos = []
|
||||||
|
return {"class": classes, "list": videos}
|
||||||
|
|
||||||
|
def _parse_vodlist(self, html):
|
||||||
|
"""解析视频列表"""
|
||||||
|
items = re.findall(
|
||||||
|
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||||
|
html
|
||||||
|
)
|
||||||
|
videos = []
|
||||||
|
for href, title, pic in items:
|
||||||
|
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||||
|
if vod_id:
|
||||||
|
vod_id = vod_id.group(1)
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||||
|
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||||
|
remark = remark_match.group(1) if remark_match else ''
|
||||||
|
if pic.startswith('/'):
|
||||||
|
pic = self.host + pic
|
||||||
|
videos.append({
|
||||||
|
'vod_id': vod_id,
|
||||||
|
'vod_name': title,
|
||||||
|
'vod_pic': pic,
|
||||||
|
'vod_remarks': remark,
|
||||||
|
})
|
||||||
|
return videos
|
||||||
|
|
||||||
|
def homeVideoContent(self):
|
||||||
|
return ''
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
pg = int(pg)
|
||||||
|
url = f'{self.host}/index.php/vod/type/id/{tid}/page/{pg}.html'
|
||||||
|
try:
|
||||||
|
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
items = re.findall(
|
||||||
|
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||||
|
html
|
||||||
|
)
|
||||||
|
videos = []
|
||||||
|
for href, title, pic in items:
|
||||||
|
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||||
|
if vod_id:
|
||||||
|
vod_id = vod_id.group(1)
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||||
|
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||||
|
remark = remark_match.group(1) if remark_match else ''
|
||||||
|
if pic.startswith('/'):
|
||||||
|
pic = self.host + pic
|
||||||
|
videos.append({
|
||||||
|
'vod_id': vod_id,
|
||||||
|
'vod_name': title,
|
||||||
|
'vod_pic': pic,
|
||||||
|
'vod_remarks': remark,
|
||||||
|
})
|
||||||
|
return {
|
||||||
|
"list": videos,
|
||||||
|
"page": pg,
|
||||||
|
"pagecount": 9999,
|
||||||
|
"limit": 90,
|
||||||
|
"total": len(videos),
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0}
|
||||||
|
|
||||||
|
def detailContent(self, ids):
|
||||||
|
try:
|
||||||
|
vod_id = ids[0] if isinstance(ids, list) else ids
|
||||||
|
url = f'{self.host}/index.php/vod/detail/id/{vod_id}.html'
|
||||||
|
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
title_match = re.search(r'<h1 class="title">([^<]+)</h1>', html)
|
||||||
|
title = title_match.group(1).strip() if title_match else ''
|
||||||
|
self._vod_name = title
|
||||||
|
pic_match = re.search(r'<img class="lazyload" data-original="([^"]*)"', html)
|
||||||
|
pic = pic_match.group(1) if pic_match else ''
|
||||||
|
if pic.startswith('/'):
|
||||||
|
pic = self.host + pic
|
||||||
|
info_match = re.search(r'类型:([^/]+)\s*/\s*地区:([^/]+)\s*/\s*年份:(\d+)', html)
|
||||||
|
type_name = info_match.group(1).strip() if info_match else ''
|
||||||
|
area = info_match.group(2).strip() if info_match else ''
|
||||||
|
year = info_match.group(3) if info_match else ''
|
||||||
|
remark_match = re.search(r'状态:<span[^>]*>([^<]+)</span>', html)
|
||||||
|
remark = remark_match.group(1).strip() if remark_match else ''
|
||||||
|
director_match = re.search(r'导演:(.*?)</p>', html, re.DOTALL)
|
||||||
|
director = ''
|
||||||
|
if director_match:
|
||||||
|
director = re.sub(r'<[^>]+>', '', director_match.group(1)).strip()
|
||||||
|
actor_match = re.search(r'主演:([^<]+)', html)
|
||||||
|
actor = actor_match.group(1).strip() if actor_match else ''
|
||||||
|
desc_match = re.search(r'<span class="detail-content"[^>]*>([^<]+)</span>', html)
|
||||||
|
desc = desc_match.group(1).strip() if desc_match else ''
|
||||||
|
play_from = []
|
||||||
|
play_url = []
|
||||||
|
source_tabs = re.findall(r'<li><a href="#playlist\d+"[^>]*>([^<]+)</a></li>', html)
|
||||||
|
for idx, source_name in enumerate(source_tabs):
|
||||||
|
source_id = idx + 1
|
||||||
|
episodes_match = re.search(
|
||||||
|
f'<div id="playlist{source_id}" class="tab-pane[^"]*"[^>]*>.*?<ul class="stui-content__playlist[^"]*"[^>]*>(.*?)</ul>',
|
||||||
|
html, re.DOTALL
|
||||||
|
)
|
||||||
|
if episodes_match:
|
||||||
|
episodes = re.findall(r'<a href="([^"]+)">([^<]+)</a>', episodes_match.group(1))
|
||||||
|
episode_list = []
|
||||||
|
for ep_url, ep_name in episodes:
|
||||||
|
episode_list.append(f'{ep_name}${self.host}{ep_url}')
|
||||||
|
play_from.append(source_name)
|
||||||
|
play_url.append('#'.join(episode_list))
|
||||||
|
vod_play_from = '$$$'.join(play_from) if play_from else '默认'
|
||||||
|
vod_play_url = '$$$'.join(play_url) if play_url else ''
|
||||||
|
vod = {
|
||||||
|
'vod_id': vod_id,
|
||||||
|
'vod_name': title,
|
||||||
|
'vod_pic': pic,
|
||||||
|
'vod_year': year,
|
||||||
|
'vod_area': area,
|
||||||
|
'vod_remarks': remark,
|
||||||
|
'vod_director': director,
|
||||||
|
'vod_actor': actor,
|
||||||
|
'vod_content': desc,
|
||||||
|
'vod_play_from': vod_play_from,
|
||||||
|
'vod_play_url': vod_play_url,
|
||||||
|
}
|
||||||
|
return {"list": [vod]}
|
||||||
|
except Exception as e:
|
||||||
|
return {"list": []}
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg="1"):
|
||||||
|
pg = int(pg)
|
||||||
|
url = f'{self.host}/index.php/vod/search/wd/{quote(key)}.html'
|
||||||
|
try:
|
||||||
|
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
items = re.findall(
|
||||||
|
r'<a class="stui-vodlist__thumb[^"]*"\s+href="([^"]+)"\s+title="([^"]*)"[^>]*data-original="([^"]*)"',
|
||||||
|
html
|
||||||
|
)
|
||||||
|
videos = []
|
||||||
|
for href, title, pic in items:
|
||||||
|
vod_id = re.search(r'/id/(\d+)\.html', href)
|
||||||
|
if vod_id:
|
||||||
|
vod_id = vod_id.group(1)
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
remark_match = re.search(r'<span class="pic-text text-right"><b>([^<]+)</b></span>',
|
||||||
|
html[html.find(href):html.find(href)+500] if href in html else '')
|
||||||
|
remark = remark_match.group(1) if remark_match else ''
|
||||||
|
if pic.startswith('/'):
|
||||||
|
pic = self.host + pic
|
||||||
|
videos.append({
|
||||||
|
'vod_id': vod_id,
|
||||||
|
'vod_name': title,
|
||||||
|
'vod_pic': pic,
|
||||||
|
'vod_remarks': remark,
|
||||||
|
})
|
||||||
|
return {
|
||||||
|
"list": videos,
|
||||||
|
"page": pg,
|
||||||
|
"pagecount": 9999,
|
||||||
|
"limit": 90,
|
||||||
|
"total": len(videos),
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0}
|
||||||
|
|
||||||
|
def _clean_vod_name(self, name):
|
||||||
|
import re
|
||||||
|
if not name:
|
||||||
|
return ''
|
||||||
|
cleaned = re.sub(r'第\s*\d+\s*[集話话章部期]', '', name)
|
||||||
|
cleaned = re.sub(r'EP\s*\d+', '', cleaned, flags=re.IGNORECASE)
|
||||||
|
cleaned = re.sub(r'全\d+集', '', cleaned)
|
||||||
|
cleaned = re.sub(r'更新至\d+集', '', cleaned)
|
||||||
|
cleaned = re.sub(r'\d+集全', '', cleaned)
|
||||||
|
cleaned = re.sub(r'[((].*?[))]', '', cleaned)
|
||||||
|
cleaned = re.sub(r'\s*-\s*.*$', '', cleaned)
|
||||||
|
cleaned = re.sub(r'\s+', ' ', cleaned)
|
||||||
|
cleaned = re.sub(r'^[\s\-_,.,。、]+|[\s\-_,.,。、]+$', '', cleaned)
|
||||||
|
return cleaned.strip()
|
||||||
|
|
||||||
|
def _build_danmaku_url(self, vod_name, vod_index=''):
|
||||||
|
import re
|
||||||
|
idx = 0
|
||||||
|
if vod_index:
|
||||||
|
s = str(vod_index).strip()
|
||||||
|
m = re.search(r'第\s*(\d+)\s*[集話话章部期]', s)
|
||||||
|
if m:
|
||||||
|
idx = int(m.group(1))
|
||||||
|
else:
|
||||||
|
m = re.search(r'(\d+)', s)
|
||||||
|
if m:
|
||||||
|
idx = int(m.group(1))
|
||||||
|
cleaned_name = self._clean_vod_name(vod_name)
|
||||||
|
params = []
|
||||||
|
if cleaned_name:
|
||||||
|
params.append(f'vodName={quote(cleaned_name)}')
|
||||||
|
params.append(f'vodIndex={idx}')
|
||||||
|
query = '&'.join(params)
|
||||||
|
return f'http://127.0.0.1:9978/proxy?do=appdanmu&{query}'
|
||||||
|
|
||||||
|
def playerContent(self, flag, id, vipFlags):
|
||||||
|
try:
|
||||||
|
ep_name = ''
|
||||||
|
vod_index = ''
|
||||||
|
if '$' in id:
|
||||||
|
parts = id.split('$', 1)
|
||||||
|
ep_name = parts[0]
|
||||||
|
url = parts[1] if len(parts) > 1 else ''
|
||||||
|
else:
|
||||||
|
url = id if id.startswith('http') else f'{self.host}{id}'
|
||||||
|
# 从 URL 中提取集数 (nid 参数)
|
||||||
|
nid_match = re.search(r'nid/(\d+)\.html', url)
|
||||||
|
if nid_match:
|
||||||
|
vod_index = nid_match.group(1)
|
||||||
|
danmaku_url = self._build_danmaku_url(self._vod_name, vod_index)
|
||||||
|
r = self.fetch(url, headers=self.headers, timeout=15, verify=False)
|
||||||
|
html = r.text
|
||||||
|
iframe_match = re.search(r'<iframe[^>]+src="([^"]+)"', html)
|
||||||
|
if iframe_match:
|
||||||
|
iframe_url = iframe_match.group(1)
|
||||||
|
if not iframe_url.startswith('http'):
|
||||||
|
iframe_url = self.host + iframe_url
|
||||||
|
return {
|
||||||
|
"parse": 1,
|
||||||
|
"url": iframe_url,
|
||||||
|
"header": self.headers,
|
||||||
|
"danmaku": danmaku_url
|
||||||
|
}
|
||||||
|
src_match = re.search(r'(https?://[^"\'<>\s]+\.m3u8[^"\'<>\s]*)', html)
|
||||||
|
if src_match:
|
||||||
|
return {
|
||||||
|
"parse": 0,
|
||||||
|
"url": src_match.group(1),
|
||||||
|
"header": self.headers,
|
||||||
|
"danmaku": danmaku_url
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"parse": 1,
|
||||||
|
"url": url,
|
||||||
|
"header": self.headers,
|
||||||
|
"danmaku": danmaku_url
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
danmaku_url = self._build_danmaku_url(self._vod_name, '')
|
||||||
|
return {"parse": 1, "url": id, "header": {}, "danmaku": danmaku_url}
|
||||||
|
|
||||||
|
def localProxy(self, param):
|
||||||
|
return [200, {}, ""]
|
||||||
|
|
||||||
|
def destroy(self):
|
||||||
|
pass
|
||||||
Reference in New Issue
Block a user