上传文件至「py」
This commit is contained in:
+317
@@ -0,0 +1,317 @@
|
||||
import json
|
||||
import random
|
||||
import sys
|
||||
from base64 import b64encode, b64decode
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
# 引入 RSA 加解密所需模块
|
||||
from Crypto.PublicKey import RSA
|
||||
from Crypto.Cipher import PKCS1_v1_5
|
||||
sys.path.append('..')
|
||||
from base.spider import Spider
|
||||
class Spider(Spider):
|
||||
def init(self, extend=""):
|
||||
did = self.getdid()
|
||||
self.headers.update({'deviceId': did})
|
||||
token = self.gettk()
|
||||
self.headers.update({'token': token})
|
||||
def getName(self):
|
||||
pass
|
||||
def isVideoFormat(self, url):
|
||||
pass
|
||||
def manualVideoCheck(self):
|
||||
pass
|
||||
def destroy(self):
|
||||
pass
|
||||
# 1. 修改为主机域名
|
||||
host = 'http://qkys.qukanwh.com'
|
||||
|
||||
# 2. 同步原脚本的配置请求头
|
||||
headers = {
|
||||
'HOST': 'qkys.qukanwh.com',
|
||||
'User-Agent': 'okhttp/4.12.0',
|
||||
'client': 'app',
|
||||
'deviceType': 'Android',
|
||||
'Referer': ''
|
||||
}
|
||||
# 3. 导入原脚本中的 RSA 密钥对与配置
|
||||
publicKey_str = "-----BEGIN PUBLIC KEY-----\nMIGfMA0GCSqGSIb3DQEBAQUAA4GNADCBiQKBgQCoYt0BP77U+DM08BiI/QbSRIfxijXo85BTPqIM1Ow8BNwhLETzRIZ+dEwdWDbydG/PspgBAfRpGaYVdJYtvaC2JnoO8+Ik6qMWojfEJxSFLa0Pb0A892tun4gsxoEMjcreZ+YGyaBxAfqX0BSMfdrOgIYaZQjYrw9TRLlUT31QoQIDAQAB\n-----END PUBLIC KEY-----"
|
||||
privateKey_str = "-----BEGIN PRIVATE KEY-----\nMIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCquQQ5r6+yJI8CDFkXRp8vUsdD45ov8EP12ooLs56ca2DQXaSNGS9910bAPVA9chkp0mKIvKqjAsHz5Tl9EeNPblarGEeJUIxpxZtiSqNTpvtiD/TjhpzuHYic7RAfQ/h7p/ypE8ymU42pYjsB5t26Mv6XgkLV+jzrSf73HlCuS0iMyLmt6zz3Mw9izM13EpB8iFLtfbbYymycKTx4RAmPQLwhNGex/AlUIYxXP4R2yyaa4W6mEtc6aME2QuzJFxPgP3HJ9NBx/LWVn4skxWjZ7zg+VRQRHnjyVaSLu3Z5gN5ITWCyE32qaHJa6WBahZj5jWhRyAG1bQ+xKJa8lBL5AgMBAAECggEAUwv9SjJ0PSwbhNuM2w23kcWquROWhYtTA91zGY4esehqB/IFgb2mpIh8Gje5OKqwIu/8jpd4SiOlRYdUF8sD0DfUYRZGdj2AkFNX6tBz8tVfo6wvbB6naA1lzzBij1L5JO3qsjS3cJFkb+kg2yP66AC2Z+0tpfk8eRhdtshAZwfcd1DEGt1uAvYL1eaUK9HRvpt9lPeGcHERDl2hBd4uyaF0K1O+zF9y59nYbTySWPxRZq3sFEE85xRMlstD7YZi7W2gKvMFRD4/FKmrZ3m7aKJRITtyKOyyPcYmepNv3Qv7kk59Pg38n2WWQ0Ra/bCH3E48YNCnQvZMpitkTfJhoQKBgQDbnROOYTP8OTJ6f/qhoGjxeO3x1VOaOp8l0x7b0SCfoqNGS0Cyiqj72BmJtPMPqSTjn6MmNzqbg1KOdhXyzNozs+i5ccW1M56j96mr5I/Z0FpE3oyIHNfDDBlf9M8YQqEF9oYxniYYft9oapO7cRQkHER6qpvnHTavwlv4m78CXwKBgQDHAjs2YlpKDdI1lcbZJCc7TwtH+Pd2bUki8YXafWNcPhITQHbOZjr310eK1QJC6GJncjkOqbX7yv3ivvTO35FZTQhuA1xEG1P00FG8bE0tHYPIwQHi9y0eA5cieMdo8E6XYria1mw/3fqSQEsfZyJlR32JQIoGAipM8iO1X2nZpwKBgDkMFIhnt5lNQk+P7wsNIDWZtDWdtJnboHuy29E+Abt2A/O+mI/IdRz2hau/1WO8DFkUnszOi+rZshhPlGP90rCbi1igtTrcrdjp/KkqNjPea5R4OwkgdOu1uOG0NheXNzzVTQaWjk7Opjn5dWa7eP/oV+GFb/oZHJuLYVizHGsBAoGADA7rjZEKDYCm4w5PPSr+oY5ZjaPdQrS+gLqHtMRyN82fBMGcMUdqfUfzEstzVqCEDeaS5HuOBlK3bXzKkppjUTjksN3NQmcxgBz7RuJ9DqXCLXDcb2cwuafYCYOt+YLOEEgwDVm+t2P44dG5e46hO+fICH/7nP+WlpD5buz4GfMCgYB57r3g/6hi9WUDnfc7ZAzWMqR0EhJVYKYy+KFEtdIPzhkkIHq5RASe88E9kzoGoZFdb3tIjvGZWcHerirrqWkMsuQtP/Qi0zjieid5tAPj+r4kbiCVTw0E0jnmPBzGInQi7lpeTTKnG1fbyS5lBS+WmHfIuzpECgCkxhaT+LJJkg==\n-----END PRIVATE KEY-----"
|
||||
# RSA 公钥加密实现
|
||||
def rsa_encrypt(self, text):
|
||||
try:
|
||||
key = RSA.import_key(self.publicKey_str)
|
||||
cipher = PKCS1_v1_5.new(key)
|
||||
cipher_text = cipher.encrypt(text.encode('utf-8'))
|
||||
return b64encode(cipher_text).decode('utf-8')
|
||||
except Exception as e:
|
||||
print(f"RSA加密失败: {e}")
|
||||
return ""
|
||||
|
||||
# RSA 私钥解密实现
|
||||
def rsa_decrypt(self, text):
|
||||
try:
|
||||
key = RSA.import_key(self.privateKey_str)
|
||||
cipher = PKCS1_v1_5.new(key)
|
||||
raw_bytes = b64decode(text.encode('utf-8'))
|
||||
|
||||
decrypted = b""
|
||||
offset = 0
|
||||
while offset < len(raw_bytes):
|
||||
chunk = raw_bytes[offset:offset + 256]
|
||||
decrypted += cipher.decrypt(chunk, None)
|
||||
offset += 256
|
||||
return decrypted.decode('utf-8')
|
||||
except Exception as e:
|
||||
print(f"RSA解密失败: {e}")
|
||||
return ""
|
||||
|
||||
def homeContent(self, filter):
|
||||
data = self.post(f"{self.host}/api/v1/app/screen/screenType", headers=self.headers).json()
|
||||
result = {}
|
||||
cate = {
|
||||
"类型": "type",
|
||||
"地区": "area",
|
||||
"年份": "year"
|
||||
}
|
||||
sort = {
|
||||
'key': 'sort',
|
||||
'name': '排序',
|
||||
'value': [{'n': '最新', 'v': 'NEWEST'}, {'n': '热门', 'v': 'HOT'}, {'n': '收藏', 'v': 'COLLECT'}]
|
||||
}
|
||||
classes = []
|
||||
filters = {}
|
||||
for k in data.get('data', []):
|
||||
classes.append({
|
||||
'type_name': k['name'],
|
||||
'type_id': str(k['id'])
|
||||
})
|
||||
filters[str(k['id'])] = []
|
||||
for v in k.get('children', []):
|
||||
if v['name'] in cate:
|
||||
filters[str(k['id'])].append({
|
||||
'name': v['name'],
|
||||
'key': cate[v['name']],
|
||||
'value': [{'n': i['name'], 'v': i['name']} for i in v.get('children', [])]
|
||||
})
|
||||
filters[str(k['id'])].append(sort)
|
||||
result['class'] = classes
|
||||
result['filters'] = filters
|
||||
return result
|
||||
|
||||
def homeVideoContent(self):
|
||||
jdata = {
|
||||
"condition": {
|
||||
"sreecnTypeEnum": "NEWEST"
|
||||
},
|
||||
"pageNum": 1,
|
||||
"pageSize": 40
|
||||
}
|
||||
data = self.post(f"{self.host}/api/v1/app/screen/screenMovie", headers=self.headers, json=jdata).json()
|
||||
return {'list': self.getlist(data.get('data', {}).get('records', []))}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
# 保持最纯粹的条件字段,移除任何空字符串占位
|
||||
condition = {
|
||||
'sreecnTypeEnum': 'NEWEST',
|
||||
'typeId': int(tid) if str(tid).isdigit() else tid
|
||||
}
|
||||
|
||||
if extend:
|
||||
if 'sort' in extend:
|
||||
condition['sreecnTypeEnum'] = extend.pop('sort')
|
||||
condition.update(extend)
|
||||
|
||||
jdata = {
|
||||
'condition': condition,
|
||||
'pageNum': int(pg),
|
||||
'pageSize': 40,
|
||||
}
|
||||
|
||||
try:
|
||||
data = self.post(f"{self.host}/api/v1/app/screen/screenMovie", headers=self.headers, json=jdata).json()
|
||||
result = {}
|
||||
if data and data.get('data') and 'records' in data['data']:
|
||||
result['list'] = self.getlist(data['data']['records'])
|
||||
else:
|
||||
result['list'] = []
|
||||
result['page'] = pg
|
||||
result['pagecount'] = 9999
|
||||
result['limit'] = 40
|
||||
result['total'] = 999999
|
||||
return result
|
||||
except Exception as e:
|
||||
print(f"分类获取错误: {e}")
|
||||
return {'list': [], 'page': pg}
|
||||
|
||||
def detailContent(self, ids):
|
||||
ids = ids[0].split('@@')
|
||||
jdata = {"id": int(ids[0]), "typeId": ids[-1]}
|
||||
v = self.post(f"{self.host}/api/v1/app/play/movieDesc", headers=self.headers, json=jdata).json()
|
||||
v = v.get('data', {})
|
||||
vod = {
|
||||
'type_name': v.get('typeId', ''),
|
||||
'vod_year': v.get('year', ''),
|
||||
'vod_area': v.get('area', ''),
|
||||
'vod_actor': v.get('star', ''),
|
||||
'vod_director': v.get('director', ''),
|
||||
'vod_content': v.get('introduce', ''),
|
||||
'vod_play_from': '',
|
||||
'vod_play_url': ''
|
||||
}
|
||||
|
||||
play_params = {
|
||||
"id": int(ids[0]),
|
||||
"source": 0,
|
||||
"typeId": ids[-1]
|
||||
}
|
||||
encrypt_payload = {"key": self.rsa_encrypt(json.dumps(play_params))}
|
||||
|
||||
c_res = self.post(f"{self.host}/api/v1/app/play/movieDetails", headers=self.headers, json=encrypt_payload).json()
|
||||
decrypted_play_str = self.rsa_decrypt(c_res.get('data', ''))
|
||||
if not decrypted_play_str:
|
||||
return {'list': [vod]}
|
||||
|
||||
decrypted_play_data = json.loads(decrypted_play_str)
|
||||
l = decrypted_play_data.get('moviePlayerList', [])
|
||||
if not l:
|
||||
return {'list': [vod]}
|
||||
|
||||
n = {str(i['id']): i['moviePlayerName'] for i in l}
|
||||
|
||||
m = play_params.copy()
|
||||
m.update({'playerId': l[0]['id']})
|
||||
|
||||
first_source_payload = {"key": self.rsa_encrypt(json.dumps(m))}
|
||||
first_res = self.post(f"{self.host}/api/v1/app/play/movieDetails", headers=self.headers, json=first_source_payload).json()
|
||||
|
||||
decrypted_first_str = self.rsa_decrypt(first_res.get('data', ''))
|
||||
if decrypted_first_str:
|
||||
decrypted_first_episode = json.loads(decrypted_first_str)
|
||||
pd = self.getv(m, decrypted_first_episode.get('episodeList', []))
|
||||
else:
|
||||
pd = {}
|
||||
|
||||
if len(l) > 1:
|
||||
with ThreadPoolExecutor(max_workers=len(l)-1) as executor:
|
||||
future_to_player = {executor.submit(self.getd, play_params, player): player for player in l[1:]}
|
||||
for future in future_to_player:
|
||||
try:
|
||||
o, p = future.result()
|
||||
if p:
|
||||
pd.update(self.getv(o, p))
|
||||
except Exception as e:
|
||||
print(f"多线路请求失败: {e}")
|
||||
w, e = [], []
|
||||
for i, x in pd.items():
|
||||
if x:
|
||||
w.append(n.get(i, '未知线路'))
|
||||
e.append(x)
|
||||
vod['vod_play_from'] = '$$$'.join(w)
|
||||
vod['vod_play_url'] = '$$$'.join(e)
|
||||
return {'list': [vod]}
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
jdata = {
|
||||
"condition": {
|
||||
"value": str(key)
|
||||
},
|
||||
"pageNum": int(pg),
|
||||
"pageSize": 40
|
||||
}
|
||||
try:
|
||||
data = self.post(f"{self.host}/api/v1/app/search/searchMovie", headers=self.headers, json=jdata).json()
|
||||
return {'list': self.getlist(data.get('data', {}).get('records', [])), 'page': pg}
|
||||
except Exception as e:
|
||||
print(f"搜索请求失败: {e}")
|
||||
return {'list': [], 'page': pg}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
raw_id_str = self.d64(id)
|
||||
if not raw_id_str:
|
||||
return {'parse': 0, 'url': ''}
|
||||
jdata = json.loads(raw_id_str)
|
||||
encrypt_payload = {"key": self.rsa_encrypt(json.dumps(jdata))}
|
||||
data = self.post(f"{self.host}/api/v1/app/play/movieDetails", headers=self.headers, json=encrypt_payload).json()
|
||||
|
||||
try:
|
||||
decrypted_url_data = json.loads(self.rsa_decrypt(data.get('data', '')))
|
||||
playerUrl = decrypted_url_data.get('url', '')
|
||||
if not playerUrl:
|
||||
return {'parse': 0, 'url': ''}
|
||||
|
||||
params = {'playerUrl': playerUrl, 'playerId': jdata['playerId']}
|
||||
pd = self.fetch(f"{self.host}/api/v1/app/play/analysisMovieUrl", headers=self.headers, params=params).json()
|
||||
url, p = pd.get('data', ''), 0
|
||||
except Exception as e:
|
||||
print(f"解析流媒体直链失败: {e}")
|
||||
url, p = "", 0
|
||||
return {'parse': p, 'url': url, 'header': {'User-Agent': 'okhttp/4.12.0'}, 'danmaku': 'http://127.0.0.1:9978/proxy?do=diydanmu'}
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
|
||||
def liveContent(self, url):
|
||||
pass
|
||||
|
||||
def gettk(self):
|
||||
self.headers.update({'deviceId': self.getdid()})
|
||||
try:
|
||||
data = self.fetch(f"{self.host}/api/v1/app/user/visitorInfo", headers=self.headers).json()
|
||||
return data.get('data', {}).get('token', '')
|
||||
except:
|
||||
return ""
|
||||
|
||||
def getdid(self):
|
||||
did = self.getCache('ldid')
|
||||
if not did:
|
||||
hex_chars = '0123456789abcdef'
|
||||
did = ''.join(random.choice(hex_chars) for _ in range(16))
|
||||
self.setCache('ldid', did)
|
||||
return did
|
||||
|
||||
def getd(self, jdata, player):
|
||||
x = jdata.copy()
|
||||
x.update({'playerId': player['id']})
|
||||
encrypt_payload = {"key": self.rsa_encrypt(json.dumps(x))}
|
||||
response = self.post(f"{self.host}/api/v1/app/play/movieDetails", headers=self.headers, json=encrypt_payload).json()
|
||||
decrypted_str = self.rsa_decrypt(response.get('data', ''))
|
||||
if decrypted_str:
|
||||
decrypted_episode = json.loads(decrypted_str)
|
||||
return x, decrypted_episode.get('episodeList', [])
|
||||
return x, []
|
||||
|
||||
def getv(self, d, c):
|
||||
f = {str(d['playerId']): ''}
|
||||
g = []
|
||||
for i in c:
|
||||
j = d.copy()
|
||||
j.update({'episodeId': i['id']})
|
||||
g.append(f"{i['episode']}${self.e64(json.dumps(j))}")
|
||||
f[str(d['playerId'])] = '#'.join(g)
|
||||
return f
|
||||
|
||||
def getlist(self, data):
|
||||
videos = []
|
||||
for i in data:
|
||||
if not i.get('id'):
|
||||
continue
|
||||
videos.append({
|
||||
'vod_id': f"{i['id']}@@{i.get('typeId', '')}",
|
||||
'vod_name': i.get('name', ''),
|
||||
'vod_pic': i.get('cover', ''),
|
||||
'vod_year': i.get('year', ''),
|
||||
'vod_remarks': i.get('totalEpisode', '')
|
||||
})
|
||||
return videos
|
||||
|
||||
def e64(self, text):
|
||||
try:
|
||||
return b64encode(text.encode('utf-8')).decode('utf-8')
|
||||
except:
|
||||
return ""
|
||||
|
||||
def d64(self, encoded_text):
|
||||
try:
|
||||
return b64decode(encoded_text.encode('utf-8')).decode('utf-8')
|
||||
except:
|
||||
return ""
|
||||
+14472
File diff suppressed because it is too large
Load Diff
+388
@@ -0,0 +1,388 @@
|
||||
# coding=utf-8
|
||||
"""
|
||||
目标站: 4kvm 首页: https://www.4kvm.net
|
||||
动态筛选、精准分集、去重列表
|
||||
"""
|
||||
import re
|
||||
import sys
|
||||
import json
|
||||
import urllib.parse
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
sys.path.append('..')
|
||||
from base.spider import Spider
|
||||
|
||||
class Spider(Spider):
|
||||
|
||||
def init(self, extend=""):
|
||||
self.site_url = "https://www.4kvm.top"
|
||||
self.headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Referer': self.site_url,
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8'
|
||||
}
|
||||
self.categories = [
|
||||
{"type_id": "1", "type_name": "电影"},
|
||||
{"type_id": "2", "type_name": "电视剧"},
|
||||
{"type_id": "3", "type_name": "动漫"}
|
||||
]
|
||||
self._filters_cache = None
|
||||
|
||||
# ================= 动态筛选解析 =================
|
||||
def _fetch_filters_for_classify(self, tid):
|
||||
"""请求 /filter?classify=tid,解析页面筛选区域,返回该分类的筛选列表"""
|
||||
url = f"{self.site_url}/filter?classify={tid}"
|
||||
resp = self.fetch(url, headers=self.headers)
|
||||
if not resp:
|
||||
return []
|
||||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||||
filter_groups = []
|
||||
containers = soup.select('main div.flex.flex-wrap.items-center.gap-3')
|
||||
for container in containers:
|
||||
links = container.select('a[href]')
|
||||
if len(links) < 2:
|
||||
continue
|
||||
first_text = links[0].get_text(strip=True)
|
||||
if not first_text.startswith('全部'):
|
||||
continue
|
||||
group_name = first_text.replace('全部', '', 1).strip()
|
||||
# 从非全部的链接中提取参数键
|
||||
param_key = None
|
||||
for a in links[1:]:
|
||||
href = a.get('href', '')
|
||||
parsed = urllib.parse.urlparse(href)
|
||||
qs = urllib.parse.parse_qs(parsed.query)
|
||||
for k in qs:
|
||||
if k not in ('classify', 'page'):
|
||||
param_key = k
|
||||
break
|
||||
if param_key:
|
||||
break
|
||||
if not param_key:
|
||||
continue
|
||||
if param_key in ('sort_by', 'order'):
|
||||
continue
|
||||
options = []
|
||||
for a in links:
|
||||
text = a.get_text(strip=True)
|
||||
href = a.get('href', '')
|
||||
parsed = urllib.parse.urlparse(href)
|
||||
qs = urllib.parse.parse_qs(parsed.query)
|
||||
val = ''
|
||||
if param_key in qs:
|
||||
val = qs[param_key][0] if qs[param_key] else ''
|
||||
if text.startswith('全部'):
|
||||
val = ''
|
||||
options.append({"n": text, "v": val})
|
||||
if options:
|
||||
filter_groups.append({
|
||||
"key": param_key,
|
||||
"name": group_name,
|
||||
"value": options
|
||||
})
|
||||
return filter_groups
|
||||
|
||||
def _get_all_filters(self):
|
||||
if self._filters_cache is not None:
|
||||
return self._filters_cache
|
||||
filters = {}
|
||||
for cat in self.categories:
|
||||
tid = cat["type_id"]
|
||||
groups = self._fetch_filters_for_classify(tid)
|
||||
if groups:
|
||||
filters[tid] = groups
|
||||
# 为没有筛选的分类复用电影分类的筛选
|
||||
if "1" in filters:
|
||||
if "3" not in filters:
|
||||
filters["3"] = filters["1"]
|
||||
if "4" not in filters:
|
||||
filters["4"] = filters["1"]
|
||||
self._filters_cache = filters
|
||||
return filters
|
||||
|
||||
# ================= 核心业务方法 =================
|
||||
def homeContent(self, filter):
|
||||
url = self.site_url + "/"
|
||||
resp = self.fetch(url, headers=self.headers)
|
||||
video_list = []
|
||||
if resp:
|
||||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||||
# 使用唯一卡片容器
|
||||
cards = soup.select('div[data-vod-id]')
|
||||
for card in cards[:20]:
|
||||
a = card.select_one('a.block[href^="/play/"]')
|
||||
if not a:
|
||||
continue
|
||||
vod_id = card.get('data-vod-id', '').strip()
|
||||
if not vod_id:
|
||||
href = a.get('href', '')
|
||||
vod_id = href.replace('/play/', '').strip()
|
||||
if not vod_id:
|
||||
continue
|
||||
title_tag = card.select_one('h3.text-white') or card.select_one('h3')
|
||||
vod_name = title_tag.get_text(strip=True) if title_tag else ''
|
||||
if not vod_name:
|
||||
continue
|
||||
img = card.select_one('img[data-src]')
|
||||
vod_pic = ''
|
||||
if img:
|
||||
src = img.get('data-src', '')
|
||||
if src and not src.startswith('data:'):
|
||||
vod_pic = src if src.startswith('http') else 'https:' + src
|
||||
remark_tag = card.select_one('.text-green-500, .text-yellow-400, span[class*="px-1.5"]')
|
||||
vod_remarks = remark_tag.get_text(strip=True) if remark_tag else ''
|
||||
video_list.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": vod_remarks
|
||||
})
|
||||
return {"class": self.categories, "list": video_list, "filters": self._get_all_filters()}
|
||||
|
||||
def homeVideoContent(self):
|
||||
return self.homeContent(False)
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
page = int(pg) if pg else 1
|
||||
params = {"classify": tid}
|
||||
if extend:
|
||||
for k, v in extend.items():
|
||||
if v and k != 'classify':
|
||||
params[k] = v
|
||||
if page > 1:
|
||||
params['page'] = page
|
||||
query = urllib.parse.urlencode(params)
|
||||
url = f"{self.site_url}/filter?{query}"
|
||||
|
||||
resp = self.fetch(url, headers=self.headers)
|
||||
if not resp:
|
||||
return {"list": [], "page": page, "pagecount": 1, "limit": 24, "total": 0}
|
||||
|
||||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||||
video_list = []
|
||||
cards = soup.select('div[data-vod-id]')
|
||||
for card in cards:
|
||||
a = card.select_one('a.block[href^="/play/"]')
|
||||
if not a:
|
||||
continue
|
||||
vod_id = card.get('data-vod-id', '').strip()
|
||||
if not vod_id:
|
||||
href = a.get('href', '')
|
||||
vod_id = href.replace('/play/', '').strip()
|
||||
if not vod_id:
|
||||
continue
|
||||
title_tag = card.select_one('h3.text-white') or card.select_one('h3')
|
||||
vod_name = title_tag.get_text(strip=True) if title_tag else ''
|
||||
if not vod_name:
|
||||
continue
|
||||
img = card.select_one('img[data-src]')
|
||||
vod_pic = ''
|
||||
if img:
|
||||
src = img.get('data-src', '')
|
||||
if src and not src.startswith('data:'):
|
||||
vod_pic = src if src.startswith('http') else 'https:' + src
|
||||
remark_tag = card.select_one('.text-green-500, .text-yellow-400, span[class*="px-1.5"]')
|
||||
vod_remarks = remark_tag.get_text(strip=True) if remark_tag else ''
|
||||
video_list.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": vod_remarks
|
||||
})
|
||||
|
||||
# 分页处理
|
||||
pagecount = page
|
||||
page_text = soup.find(string=re.compile(r'共\s*\d+\s*页'))
|
||||
if page_text:
|
||||
nums = re.findall(r'\d+', page_text)
|
||||
if nums:
|
||||
pagecount = int(nums[-1])
|
||||
else:
|
||||
page_block = soup.select_one('.flex.justify-center')
|
||||
if page_block:
|
||||
page_links = page_block.select('a[href*="page="]')
|
||||
for a in page_links:
|
||||
text = a.get_text(strip=True)
|
||||
if text.isdigit():
|
||||
pagecount = max(pagecount, int(text))
|
||||
|
||||
return {
|
||||
"list": video_list,
|
||||
"page": page,
|
||||
"pagecount": pagecount,
|
||||
"limit": 24,
|
||||
"total": len(video_list) * pagecount
|
||||
}
|
||||
|
||||
def detailContent(self, ids):
|
||||
if not ids:
|
||||
return {"list": []}
|
||||
vod_id = ids[0]
|
||||
url = f"{self.site_url}/play/{vod_id}"
|
||||
resp = self.fetch(url, headers=self.headers)
|
||||
if not resp or resp.status_code != 200:
|
||||
return {"list": []}
|
||||
|
||||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||||
|
||||
# 标题
|
||||
title_elem = soup.select_one('h1.text-xl') or soup.select_one('h1') or soup.select_one('h2')
|
||||
vod_name = title_elem.get_text(strip=True) if title_elem else vod_id
|
||||
|
||||
# 图片
|
||||
vod_pic = ''
|
||||
img_elem = soup.select_one('img.w-full') or soup.select_one('img[src]')
|
||||
if img_elem:
|
||||
src = img_elem.get('src', '') or img_elem.get('data-src', '')
|
||||
if src and not src.startswith('data:'):
|
||||
vod_pic = src if src.startswith('http') else 'https:' + src
|
||||
|
||||
# 导演、主演、简介
|
||||
vod_director = ''
|
||||
vod_actor = ''
|
||||
vod_content = ''
|
||||
info_block = soup.select_one('.rounded-lg div.grid') or soup.select_one('div.grid')
|
||||
if info_block:
|
||||
text = info_block.get_text(' ', strip=True)
|
||||
dir_match = re.search(r'导演\s*([^主\n]+)', text)
|
||||
if dir_match:
|
||||
vod_director = dir_match.group(1).strip()
|
||||
act_match = re.search(r'主演\s*([^剧\n]+)', text)
|
||||
if act_match:
|
||||
vod_actor = act_match.group(1).strip()
|
||||
desc_match = re.search(r'剧情简介\s*(.+)', text, re.DOTALL)
|
||||
if desc_match:
|
||||
vod_content = desc_match.group(1).strip()
|
||||
elif re.search(r'简介\s*(.+)', text, re.DOTALL):
|
||||
vod_content = re.search(r'简介\s*(.+)', text, re.DOTALL).group(1).strip()
|
||||
|
||||
# ================= 分集解析 (基于 episodeManager) =================
|
||||
play_from_list = []
|
||||
play_url_list = []
|
||||
|
||||
episode_manager = soup.select_one('[x-data*="episodeManager"]')
|
||||
if episode_manager:
|
||||
xdata = episode_manager.get('x-data', '')
|
||||
lines_raw = re.findall(r'\{[^}]*lineName\s*:\s*\'([^\']+)\'[^}]*episodeCount\s*:\s*(\d+)[^}]*\}', xdata)
|
||||
lines_info = [{'lineName': name, 'episodeCount': int(count)} for name, count in lines_raw]
|
||||
|
||||
episode_links = episode_manager.select('a[data-episode]')
|
||||
lines_eps = {}
|
||||
for a in episode_links:
|
||||
line = a.get('data-line', '1')
|
||||
ep = a.get('data-episode', '')
|
||||
href = a.get('href', '')
|
||||
if not href or not ep:
|
||||
continue
|
||||
full_url = href if href.startswith('http') else self.site_url + href
|
||||
lines_eps.setdefault(line, []).append((int(ep), full_url))
|
||||
|
||||
for line_key in sorted(lines_eps.keys()):
|
||||
eps = sorted(lines_eps[line_key], key=lambda x: x[0])
|
||||
line_name = f'线路{line_key}'
|
||||
for info in lines_info:
|
||||
line_name = info['lineName']
|
||||
break # 目前只用第一个线路名
|
||||
if not eps:
|
||||
continue
|
||||
episode_strs = [f"第{ep[0]}集${ep[1]}" for ep in eps]
|
||||
play_from_list.append(line_name)
|
||||
play_url_list.append('#'.join(episode_strs))
|
||||
|
||||
# 回退:无分集则直接播放当前页
|
||||
if not play_url_list:
|
||||
play_from_list.append('播放')
|
||||
play_url_list.append(f"播放${vod_id}")
|
||||
|
||||
vod_play_from = '$$$'.join(play_from_list)
|
||||
vod_play_url = '$$$'.join(play_url_list)
|
||||
|
||||
result = [{
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_content": vod_content,
|
||||
"vod_actor": vod_actor,
|
||||
"vod_director": vod_director,
|
||||
"vod_area": "",
|
||||
"vod_year": "",
|
||||
"vod_play_from": vod_play_from,
|
||||
"vod_play_url": vod_play_url
|
||||
}]
|
||||
return {"list": result}
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
page = int(pg) if pg else 1
|
||||
params = {"q": key}
|
||||
if page > 1:
|
||||
params['page'] = page
|
||||
query = urllib.parse.urlencode(params)
|
||||
url = f"{self.site_url}/search?{query}"
|
||||
resp = self.fetch(url, headers=self.headers)
|
||||
if not resp:
|
||||
return {"list": [], "page": page, "pagecount": 1}
|
||||
|
||||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||||
video_list = []
|
||||
cards = soup.select('div[data-vod-id]')
|
||||
if not cards:
|
||||
# 搜索页可能没有 data-vod-id,降级处理
|
||||
for a in soup.select('a.block[href^="/play/"]'):
|
||||
href = a.get('href', '')
|
||||
vod_id = href.replace('/play/', '').strip()
|
||||
if not vod_id:
|
||||
continue
|
||||
h3 = a.select_one('h3')
|
||||
vod_name = h3.get_text(strip=True) if h3 else href
|
||||
if not vod_name:
|
||||
continue
|
||||
img = a.select_one('img[data-src]')
|
||||
vod_pic = ''
|
||||
if img:
|
||||
src = img.get('data-src', '')
|
||||
if src and not src.startswith('data:'):
|
||||
vod_pic = src if src.startswith('http') else 'https:' + src
|
||||
video_list.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": ''
|
||||
})
|
||||
else:
|
||||
for card in cards[:30]:
|
||||
a = card.select_one('a.block[href^="/play/"]')
|
||||
if not a:
|
||||
continue
|
||||
vod_id = card.get('data-vod-id', '').strip()
|
||||
if not vod_id:
|
||||
href = a.get('href', '')
|
||||
vod_id = href.replace('/play/', '').strip()
|
||||
if not vod_id:
|
||||
continue
|
||||
title_tag = card.select_one('h3.text-white') or card.select_one('h3')
|
||||
vod_name = title_tag.get_text(strip=True) if title_tag else ''
|
||||
if not vod_name:
|
||||
continue
|
||||
img = card.select_one('img[data-src]')
|
||||
vod_pic = ''
|
||||
if img:
|
||||
src = img.get('data-src', '')
|
||||
if src and not src.startswith('data:'):
|
||||
vod_pic = src if src.startswith('http') else 'https:' + src
|
||||
remark_tag = card.select_one('.text-green-500, .text-yellow-400, span[class*="px-1.5"]')
|
||||
vod_remarks = remark_tag.get_text(strip=True) if remark_tag else ''
|
||||
video_list.append({
|
||||
"vod_id": vod_id,
|
||||
"vod_name": vod_name,
|
||||
"vod_pic": vod_pic,
|
||||
"vod_remarks": vod_remarks
|
||||
})
|
||||
return {"list": video_list, "page": page, "pagecount": 1}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
if not id.startswith('http'):
|
||||
url = f"{self.site_url}/play/{id}"
|
||||
else:
|
||||
url = id
|
||||
return {"parse": 1, "url": url, "header": self.headers}
|
||||
+404
@@ -0,0 +1,404 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# 🌈 Love
|
||||
import json
|
||||
import random
|
||||
import re
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from base64 import b64decode, b64encode
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
from Crypto.Cipher import AES
|
||||
from Crypto.Util.Padding import unpad
|
||||
from pyquery import PyQuery as pq
|
||||
sys.path.append('..')
|
||||
from base.spider import Spider
|
||||
|
||||
|
||||
class Spider(Spider):
|
||||
|
||||
def init(self, extend=""):
|
||||
try:self.proxies = json.loads(extend)
|
||||
except:self.proxies = {}
|
||||
self.headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9',
|
||||
'Connection': 'keep-alive',
|
||||
'Cache-Control': 'no-cache',
|
||||
}
|
||||
# Use working dynamic URLs directly
|
||||
self.host = self.get_working_host()
|
||||
self.headers.update({'Origin': self.host, 'Referer': f"{self.host}/"})
|
||||
self.log(f"使用站点: {self.host}")
|
||||
print(f"使用站点: {self.host}")
|
||||
pass
|
||||
|
||||
def getName(self):
|
||||
return "通用吸瓜"
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
# Treat direct media formats as playable without parsing
|
||||
return any(ext in (url or '') for ext in ['.m3u8', '.mp4', '.ts'])
|
||||
|
||||
def manualVideoCheck(self):
|
||||
return False
|
||||
|
||||
def destroy(self):
|
||||
pass
|
||||
|
||||
def homeContent(self, filter):
|
||||
try:
|
||||
response = requests.get(self.host, headers=self.headers, proxies=self.proxies, timeout=15)
|
||||
if response.status_code != 200:
|
||||
return {'class': [], 'list': []}
|
||||
|
||||
data = self.getpq(response.text)
|
||||
result = {}
|
||||
classes = []
|
||||
|
||||
# Try to get categories from different possible locations
|
||||
category_selectors = [
|
||||
'.category-list ul li',
|
||||
'.nav-menu li',
|
||||
'.menu li',
|
||||
'nav ul li'
|
||||
]
|
||||
|
||||
for selector in category_selectors:
|
||||
for k in data(selector).items():
|
||||
link = k('a')
|
||||
href = (link.attr('href') or '').strip()
|
||||
name = (link.text() or '').strip()
|
||||
# Skip placeholder or invalid entries
|
||||
if not href or href == '#' or not name:
|
||||
continue
|
||||
classes.append({
|
||||
'type_name': name,
|
||||
'type_id': href
|
||||
})
|
||||
if classes:
|
||||
break
|
||||
|
||||
# If no categories found, create some default ones
|
||||
if not classes:
|
||||
classes = [
|
||||
{'type_name': '首页', 'type_id': '/'},
|
||||
{'type_name': '最新', 'type_id': '/latest/'},
|
||||
{'type_name': '热门', 'type_id': '/hot/'}
|
||||
]
|
||||
|
||||
result['class'] = classes
|
||||
result['list'] = self.getlist(data('#index article a'))
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"homeContent error: {e}")
|
||||
return {'class': [], 'list': []}
|
||||
|
||||
def homeVideoContent(self):
|
||||
try:
|
||||
response = requests.get(self.host, headers=self.headers, proxies=self.proxies, timeout=15)
|
||||
if response.status_code != 200:
|
||||
return {'list': []}
|
||||
data = self.getpq(response.text)
|
||||
return {'list': self.getlist(data('#index article a, #archive article a'))}
|
||||
except Exception as e:
|
||||
print(f"homeVideoContent error: {e}")
|
||||
return {'list': []}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
try:
|
||||
if '@folder' in tid:
|
||||
id = tid.replace('@folder', '')
|
||||
videos = self.getfod(id)
|
||||
else:
|
||||
# Build URL properly
|
||||
if tid.startswith('/'):
|
||||
if pg and pg != '1':
|
||||
url = f"{self.host}{tid}page/{pg}/"
|
||||
else:
|
||||
url = f"{self.host}{tid}"
|
||||
else:
|
||||
url = f"{self.host}/{tid}"
|
||||
|
||||
response = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=15)
|
||||
if response.status_code != 200:
|
||||
return {'list': [], 'page': pg, 'pagecount': 1, 'limit': 90, 'total': 0}
|
||||
|
||||
data = self.getpq(response.text)
|
||||
videos = self.getlist(data('#archive article a, #index article a'), tid)
|
||||
|
||||
result = {}
|
||||
result['list'] = videos
|
||||
result['page'] = pg
|
||||
result['pagecount'] = 1 if '@folder' in tid else 99999
|
||||
result['limit'] = 90
|
||||
result['total'] = 999999
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"categoryContent error: {e}")
|
||||
return {'list': [], 'page': pg, 'pagecount': 1, 'limit': 90, 'total': 0}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
url = f"{self.host}{ids[0]}" if not ids[0].startswith('http') else ids[0]
|
||||
response = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=15)
|
||||
|
||||
if response.status_code != 200:
|
||||
return {'list': [{'vod_play_from': '通用吸瓜', 'vod_play_url': f'页面加载失败${url}'}]}
|
||||
|
||||
data = self.getpq(response.text)
|
||||
vod = {'vod_play_from': '通用吸瓜'}
|
||||
|
||||
# Get content/description
|
||||
try:
|
||||
clist = []
|
||||
if data('.tags .keywords a'):
|
||||
for k in data('.tags .keywords a').items():
|
||||
title = k.text()
|
||||
href = k.attr('href')
|
||||
if title and href:
|
||||
clist.append('[a=cr:' + json.dumps({'id': href, 'name': title}) + '/]' + title + '[/a]')
|
||||
vod['vod_content'] = ' '.join(clist) if clist else data('.post-title').text()
|
||||
except:
|
||||
vod['vod_content'] = data('.post-title').text() or '通用吸瓜视频'
|
||||
|
||||
# Get video URLs (build episode list when multiple players exist)
|
||||
try:
|
||||
plist = []
|
||||
used_names = set()
|
||||
if data('.dplayer'):
|
||||
for c, k in enumerate(data('.dplayer').items(), start=1):
|
||||
config_attr = k.attr('data-config')
|
||||
if config_attr:
|
||||
try:
|
||||
config = json.loads(config_attr)
|
||||
video_url = config.get('video', {}).get('url', '')
|
||||
# Determine a readable episode name from nearby headings if present
|
||||
ep_name = ''
|
||||
try:
|
||||
parent = k.parents().eq(0)
|
||||
# search up to a few ancestors for a heading text
|
||||
for _ in range(3):
|
||||
if not parent: break
|
||||
heading = parent.find('h2, h3, h4').eq(0).text() or ''
|
||||
heading = heading.strip()
|
||||
if heading:
|
||||
ep_name = heading
|
||||
break
|
||||
parent = parent.parents().eq(0)
|
||||
except Exception:
|
||||
ep_name = ''
|
||||
base_name = ep_name if ep_name else f"视频{c}"
|
||||
name = base_name
|
||||
count = 2
|
||||
# Ensure the name is unique
|
||||
while name in used_names:
|
||||
name = f"{base_name} {count}"
|
||||
count += 1
|
||||
used_names.add(name)
|
||||
if video_url:
|
||||
self.log(f"解析到视频: {name} -> {video_url}")
|
||||
print(f"解析到视频: {name} -> {video_url}")
|
||||
plist.append(f"{name}${video_url}")
|
||||
except:
|
||||
continue
|
||||
|
||||
if plist:
|
||||
self.log(f"拼装播放列表,共{len(plist)}个")
|
||||
print(f"拼装播放列表,共{len(plist)}个")
|
||||
vod['vod_play_url'] = '#'.join(plist)
|
||||
else:
|
||||
vod['vod_play_url'] = f"未找到视频源${url}"
|
||||
|
||||
except Exception as e:
|
||||
vod['vod_play_url'] = f"视频解析失败${url}"
|
||||
|
||||
return {'list': [vod]}
|
||||
|
||||
except Exception as e:
|
||||
print(f"detailContent error: {e}")
|
||||
return {'list': [{'vod_play_from': '通用吸瓜', 'vod_play_url': f'详情页加载失败${ids[0] if ids else ""}'}]}
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
try:
|
||||
url = f"{self.host}/search/{key}/{pg}" if pg != "1" else f"{self.host}/search/{key}/"
|
||||
response = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=15)
|
||||
|
||||
if response.status_code != 200:
|
||||
return {'list': [], 'page': pg}
|
||||
|
||||
data = self.getpq(response.text)
|
||||
videos = self.getlist(data('#archive article a, #index article a'))
|
||||
return {'list': videos, 'page': pg}
|
||||
|
||||
except Exception as e:
|
||||
print(f"searchContent error: {e}")
|
||||
return {'list': [], 'page': pg}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
url = id
|
||||
p = 1
|
||||
if self.isVideoFormat(url):
|
||||
# m3u8/mp4 direct play; when using proxy setting, wrap to proxy for m3u8
|
||||
if '.m3u8' in url:
|
||||
url = self.proxy(url)
|
||||
p = 0
|
||||
self.log(f"播放请求: parse={p}, url={url}")
|
||||
print(f"播放请求: parse={p}, url={url}")
|
||||
return {'parse': p, 'url': url, 'header': self.headers}
|
||||
|
||||
def localProxy(self, param):
|
||||
if param.get('type') == 'img':
|
||||
res=requests.get(param['url'], headers=self.headers, proxies=self.proxies, timeout=10)
|
||||
return [200,res.headers.get('Content-Type'),self.aesimg(res.content)]
|
||||
elif param.get('type') == 'm3u8':return self.m3Proxy(param['url'])
|
||||
else:return self.tsProxy(param['url'])
|
||||
|
||||
def proxy(self, data, type='m3u8'):
|
||||
if data and len(self.proxies):return f"{self.getProxyUrl()}&url={self.e64(data)}&type={type}"
|
||||
else:return data
|
||||
|
||||
def m3Proxy(self, url):
|
||||
url=self.d64(url)
|
||||
ydata = requests.get(url, headers=self.headers, proxies=self.proxies, allow_redirects=False)
|
||||
data = ydata.content.decode('utf-8')
|
||||
if ydata.headers.get('Location'):
|
||||
url = ydata.headers['Location']
|
||||
data = requests.get(url, headers=self.headers, proxies=self.proxies).content.decode('utf-8')
|
||||
lines = data.strip().split('\n')
|
||||
last_r = url[:url.rfind('/')]
|
||||
parsed_url = urlparse(url)
|
||||
durl = parsed_url.scheme + "://" + parsed_url.netloc
|
||||
iskey=True
|
||||
for index, string in enumerate(lines):
|
||||
if iskey and 'URI' in string:
|
||||
pattern = r'URI="([^"]*)"'
|
||||
match = re.search(pattern, string)
|
||||
if match:
|
||||
lines[index] = re.sub(pattern, f'URI="{self.proxy(match.group(1), "mkey")}"', string)
|
||||
iskey=False
|
||||
continue
|
||||
if '#EXT' not in string:
|
||||
if 'http' not in string:
|
||||
domain = last_r if string.count('/') < 2 else durl
|
||||
string = domain + ('' if string.startswith('/') else '/') + string
|
||||
lines[index] = self.proxy(string, string.split('.')[-1].split('?')[0])
|
||||
data = '\n'.join(lines)
|
||||
return [200, "application/vnd.apple.mpegur", data]
|
||||
|
||||
def tsProxy(self, url):
|
||||
url = self.d64(url)
|
||||
data = requests.get(url, headers=self.headers, proxies=self.proxies, stream=True)
|
||||
return [200, data.headers['Content-Type'], data.content]
|
||||
|
||||
def e64(self, text):
|
||||
try:
|
||||
text_bytes = text.encode('utf-8')
|
||||
encoded_bytes = b64encode(text_bytes)
|
||||
return encoded_bytes.decode('utf-8')
|
||||
except Exception as e:
|
||||
print(f"Base64编码错误: {str(e)}")
|
||||
return ""
|
||||
|
||||
def d64(self, encoded_text):
|
||||
try:
|
||||
encoded_bytes = encoded_text.encode('utf-8')
|
||||
decoded_bytes = b64decode(encoded_bytes)
|
||||
return decoded_bytes.decode('utf-8')
|
||||
except Exception as e:
|
||||
print(f"Base64解码错误: {str(e)}")
|
||||
return ""
|
||||
|
||||
def get_working_host(self):
|
||||
"""Get working host from known dynamic URLs"""
|
||||
# Known working URLs from the dynamic gateway
|
||||
dynamic_urls = [
|
||||
'https://advise.nlwkmsv.cc/',
|
||||
'advise.nlwkmsv.cc',
|
||||
# 'https://am.vgwtswi.xyz'
|
||||
]
|
||||
|
||||
# Test each URL to find a working one
|
||||
for url in dynamic_urls:
|
||||
try:
|
||||
response = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=10)
|
||||
if response.status_code == 200:
|
||||
# Verify it has the expected content structure
|
||||
data = self.getpq(response.text)
|
||||
articles = data('#index article a')
|
||||
if len(articles) > 0:
|
||||
self.log(f"选用可用站点: {url}")
|
||||
print(f"选用可用站点: {url}")
|
||||
return url
|
||||
except Exception as e:
|
||||
continue
|
||||
|
||||
# Fallback to first URL if none work (better than crashing)
|
||||
self.log(f"未检测到可用站点,回退: {dynamic_urls[0]}")
|
||||
print(f"未检测到可用站点,回退: {dynamic_urls[0]}")
|
||||
return dynamic_urls[0]
|
||||
|
||||
|
||||
def getlist(self, data, tid=''):
|
||||
videos = []
|
||||
l = '/mrdg' in tid
|
||||
for k in data.items():
|
||||
a = k.attr('href')
|
||||
b = k('h2').text()
|
||||
# Some pages might not include datePublished; use a fallback
|
||||
c = k('span[itemprop="datePublished"]').text() or k('.post-meta, .entry-meta, time').text()
|
||||
if a and b:
|
||||
videos.append({
|
||||
'vod_id': f"{a}{'@folder' if l else ''}",
|
||||
'vod_name': b.replace('\n', ' '),
|
||||
'vod_pic': self.getimg(k('script').text()),
|
||||
'vod_remarks': c or '',
|
||||
'vod_tag': 'folder' if l else '',
|
||||
'style': {"type": "rect", "ratio": 1.33}
|
||||
})
|
||||
return videos
|
||||
|
||||
def getfod(self, id):
|
||||
url = f"{self.host}{id}"
|
||||
data = self.getpq(requests.get(url, headers=self.headers, proxies=self.proxies).text)
|
||||
vdata=data('.post-content[itemprop="articleBody"]')
|
||||
r=['.txt-apps','.line','blockquote','.tags','.content-tabs']
|
||||
for i in r:vdata.remove(i)
|
||||
p=vdata('p')
|
||||
videos=[]
|
||||
for i,x in enumerate(vdata('h2').items()):
|
||||
c=i*2
|
||||
videos.append({
|
||||
'vod_id': p.eq(c)('a').attr('href'),
|
||||
'vod_name': p.eq(c).text(),
|
||||
'vod_pic': f"{self.getProxyUrl()}&url={p.eq(c+1)('img').attr('data-xkrkllgl')}&type=img",
|
||||
'vod_remarks':x.text()
|
||||
})
|
||||
return videos
|
||||
|
||||
def getimg(self, text):
|
||||
match = re.search(r"loadBannerDirect\('([^']+)'", text)
|
||||
if match:
|
||||
url = match.group(1)
|
||||
return f"{self.getProxyUrl()}&url={url}&type=img"
|
||||
else:
|
||||
return ''
|
||||
|
||||
def aesimg(self, word):
|
||||
key = b'f5d965df75336270'
|
||||
iv = b'97b60394abc2fbe1'
|
||||
cipher = AES.new(key, AES.MODE_CBC, iv)
|
||||
decrypted = unpad(cipher.decrypt(word), AES.block_size)
|
||||
return decrypted
|
||||
|
||||
def getpq(self, data):
|
||||
try:
|
||||
return pq(data)
|
||||
except Exception as e:
|
||||
print(f"{str(e)}")
|
||||
return pq(data.encode('utf-8'))
|
||||
@@ -0,0 +1,306 @@
|
||||
# coding=utf-8
|
||||
#!/usr/bin/python
|
||||
import re
|
||||
import sys
|
||||
from html import unescape
|
||||
from urllib.parse import urljoin, quote
|
||||
|
||||
sys.path.append('..')
|
||||
from base.spider import Spider
|
||||
|
||||
|
||||
class Spider(Spider):
|
||||
host = 'https://www.xb6v.org'
|
||||
headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Linux; Android 13) AppleWebKit/537.36 Chrome/120 Mobile Safari/537.36',
|
||||
'Referer': 'https://www.xb6v.org/',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9',
|
||||
}
|
||||
classes = [
|
||||
{'type_name': '首页', 'type_id': '/'},
|
||||
{'type_name': '喜剧片', 'type_id': '/xijupian/'},
|
||||
{'type_name': '动作片', 'type_id': '/dongzuopian/'},
|
||||
{'type_name': '爱情片', 'type_id': '/aiqingpian/'},
|
||||
{'type_name': '科幻片', 'type_id': '/kehuanpian/'},
|
||||
{'type_name': '恐怖片', 'type_id': '/kongbupian/'},
|
||||
{'type_name': '剧情片', 'type_id': '/juqingpian/'},
|
||||
{'type_name': '战争片', 'type_id': '/zhanzhengpian/'},
|
||||
{'type_name': '纪录片', 'type_id': '/jilupian/'},
|
||||
{'type_name': '动画片', 'type_id': '/donghuapian/'},
|
||||
{'type_name': '电视剧', 'type_id': '/dianshiju/'},
|
||||
{'type_name': '综艺', 'type_id': '/ZongYi/'},
|
||||
]
|
||||
|
||||
def getName(self):
|
||||
return '6v影视'
|
||||
|
||||
def init(self, extend=''):
|
||||
pass
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
pass
|
||||
|
||||
def manualVideoCheck(self):
|
||||
pass
|
||||
|
||||
def destroy(self):
|
||||
pass
|
||||
|
||||
def homeContent(self, filter):
|
||||
return {'class': self.classes}
|
||||
|
||||
def homeVideoContent(self):
|
||||
return {'list': self._parse_list(self._html(self.host + '/'))}
|
||||
|
||||
# 兼容部分 Python Spider 壳的命名
|
||||
def homeVod(self):
|
||||
return self.homeVideoContent()
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
page = int(pg or 1)
|
||||
path = tid or '/'
|
||||
if page > 1:
|
||||
if path.endswith('/'):
|
||||
path = path + 'index_%d.html' % page
|
||||
else:
|
||||
path = path.rstrip('/') + '/index_%d.html' % page
|
||||
html = self._html(self._abs(path))
|
||||
videos = self._parse_list(html)
|
||||
return {'list': videos, 'page': page, 'pagecount': page + 1 if videos else page, 'limit': 18, 'total': 999999 if videos else 0}
|
||||
|
||||
def detailContent(self, ids):
|
||||
url = ids[0] if isinstance(ids, list) else ids
|
||||
url = self._abs(url)
|
||||
html = self._html(url)
|
||||
title = self._first([r'<title>\s*([^<]+?)(?:-|_|\|).*?</title>', r'<h1[^>]*>(.*?)</h1>'], html, '未知影片')
|
||||
pic = self._first([r'<div[^>]+class=["\'][^"\']*thumbnail[^"\']*["\'][\s\S]*?<img[^>]+(?:src|data-original|data-src)=["\']([^"\']+)["\']', r'<img[^>]+(?:src|data-original|data-src)=["\']([^"\']+\.(?:jpg|jpeg|png|webp)[^"\']*)["\']'], html, '')
|
||||
desc = self._first([r'◎简\s*介\s*([\s\S]*?)(?:◎|<h3|</article>|$)', r'<meta[^>]+name=["\']description["\'][^>]+content=["\']([^"\']+)["\']'], html, '暂无简介')
|
||||
lines = self._parse_detail_lines(html)
|
||||
play_from = '$$$'.join([x['name'] for x in lines]) or '详情页'
|
||||
play_url = '$$$'.join(['#'.join(['%s$%s' % (ep['title'], ep['url']) for ep in x['episodes']]) for x in lines]) or ('打开详情页$' + url)
|
||||
vod = {
|
||||
'vod_id': url,
|
||||
'vod_name': self._clean(title),
|
||||
'vod_pic': self._abs(pic),
|
||||
'type_name': '6v影视',
|
||||
'vod_year': self._year(title + ' ' + html),
|
||||
'vod_area': '',
|
||||
'vod_remarks': '',
|
||||
'vod_actor': '',
|
||||
'vod_director': '',
|
||||
'vod_content': self._clean(desc),
|
||||
'vod_play_from': play_from,
|
||||
'vod_play_url': play_url,
|
||||
}
|
||||
return {'list': [vod]}
|
||||
|
||||
def searchContent(self, key, quick, pg='1'):
|
||||
url = self.host + '/e/search/11index.php'
|
||||
body = 'keyboard=%s&show=title&tempid=1&tbname=article&mid=1&dopost=search&submit=' % quote(key or '')
|
||||
html = self._html(url, method='post', data=body, extra_headers={'Content-Type': 'application/x-www-form-urlencoded'})
|
||||
return {'list': self._parse_list(html)}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
url = self._abs(id)
|
||||
low = url.lower()
|
||||
if low.startswith('magnet:') or any(x in low for x in ['pan.quark.cn', 'pan.baidu.com', 'xunlei.com', 'aliyundrive.com', 'alipan.com', 'cloud.189.cn', 'yun.139.com', '123pan']):
|
||||
return {'parse': 0, 'jx': 0, 'url': url, 'header': self.headers}
|
||||
real = self._resolve_play_url(url)
|
||||
return {'parse': 0, 'jx': 0, 'url': real or url, 'header': {'User-Agent': self.headers['User-Agent'], 'Referer': self.host + '/'}}
|
||||
|
||||
def localProxy(self, param):
|
||||
return None
|
||||
|
||||
def _html(self, url, method='get', data=None, extra_headers=None):
|
||||
headers = dict(self.headers)
|
||||
if extra_headers:
|
||||
headers.update(extra_headers)
|
||||
try:
|
||||
if method.lower() == 'post':
|
||||
res = self.fetch(url, headers=headers, data=data, method='post', timeout=20)
|
||||
else:
|
||||
res = self.fetch(url, headers=headers, timeout=20)
|
||||
return self._res_text(res)
|
||||
except Exception:
|
||||
# 某些壳不支持 method 参数,POST 搜索失败时返回空;分类/首页不受影响
|
||||
try:
|
||||
if method.lower() == 'post':
|
||||
res = self.fetch(url, headers=headers, postData=data, timeout=20)
|
||||
return self._res_text(res)
|
||||
except Exception:
|
||||
pass
|
||||
return ''
|
||||
|
||||
def _res_text(self, res):
|
||||
if res is None:
|
||||
return ''
|
||||
if isinstance(res, str):
|
||||
return res
|
||||
if isinstance(res, bytes):
|
||||
return self._decode(res)
|
||||
if isinstance(res, dict):
|
||||
val = res.get('content') or res.get('body') or res.get('data') or res.get('text') or ''
|
||||
if isinstance(val, bytes):
|
||||
return self._decode(val)
|
||||
return str(val or '')
|
||||
if hasattr(res, 'content'):
|
||||
val = getattr(res, 'content')
|
||||
if isinstance(val, bytes):
|
||||
return self._decode(val)
|
||||
return str(val or '')
|
||||
if hasattr(res, 'text'):
|
||||
return str(getattr(res, 'text') or '')
|
||||
return str(res)
|
||||
|
||||
def _decode(self, data):
|
||||
for enc in ('utf-8', 'gbk', 'gb18030'):
|
||||
try:
|
||||
txt = data.decode(enc)
|
||||
if '锟斤拷' not in txt and '\ufffd' not in txt:
|
||||
return txt
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
return data.decode('utf-8', errors='ignore')
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
def _abs(self, url):
|
||||
if not url:
|
||||
return ''
|
||||
if str(url).startswith('magnet:'):
|
||||
return url
|
||||
return urljoin(self.host + '/', str(url).replace('&', '&'))
|
||||
|
||||
def _clean(self, text):
|
||||
text = re.sub(r'<script[\s\S]*?</script>|<style[\s\S]*?</style>', '', str(text or ''), flags=re.I)
|
||||
text = re.sub(r'<[^>]+>', ' ', text)
|
||||
text = unescape(text)
|
||||
text = re.sub(r'"?>\s*', ' ', text)
|
||||
text = re.sub(r'[\ue000-\uf8ff]', '', text)
|
||||
return re.sub(r'\s+', ' ', text).strip()
|
||||
|
||||
def _first(self, patterns, html, default=''):
|
||||
for pat in patterns:
|
||||
m = re.search(pat, html or '', flags=re.I)
|
||||
if m:
|
||||
return self._clean(m.group(1))
|
||||
return default
|
||||
|
||||
def _year(self, text):
|
||||
m = re.search(r'\b((?:19|20)\d{2})\b', text or '')
|
||||
return m.group(1) if m else ''
|
||||
|
||||
def _parse_list(self, html):
|
||||
html = html or ''
|
||||
videos = []
|
||||
seen = set()
|
||||
# 6v 实测列表:<li class="post box row fixed-hight"> ... thumbnail ... h2 ... </li>
|
||||
blocks = re.findall(r'<li[^>]+class=["\'][^"\']*\bpost\b[^"\']*["\'][^>]*>[\s\S]*?</li>', html, flags=re.I)
|
||||
for block in blocks:
|
||||
href = self._first_raw([r'<h2>[\s\S]*?<a[^>]+href=["\']([^"\']+\.html)["\']', r'<a[^>]+href=["\']([^"\']+\.html)["\'][^>]*class=["\'][^"\']*zoom'], block)
|
||||
title = self._first_raw([r'<h2>[\s\S]*?<a[^>]*>([\s\S]*?)</a>', r'<h2>[\s\S]*?<a[^>]*title=["\']([^"\']+)["\']', r'<a[^>]*title=["\']([^"\']+)["\']'], block)
|
||||
pic = self._first_raw([r'<img[^>]+(?:src|data-original|data-src)=["\']([^"\']+)["\']'], block)
|
||||
item = self._item(href, title, pic)
|
||||
if item and item['vod_id'] not in seen:
|
||||
seen.add(item['vod_id'])
|
||||
videos.append(item)
|
||||
if videos:
|
||||
return videos
|
||||
# 兜底:全页 anchor,只收真正详情页,避免导航链接
|
||||
for href, inner in re.findall(r'<a\b[^>]*href=["\']([^"\']+\.html)["\'][^>]*>([\s\S]*?)</a>', html, flags=re.I):
|
||||
if '/e/' in href or href in ('/', '/index.html'):
|
||||
continue
|
||||
if not re.search(r'/[A-Za-z0-9_-]+/\d+\.html|/\d+\.html', href):
|
||||
continue
|
||||
title = self._clean(inner)
|
||||
pic = self._first_raw([r'<img[^>]+(?:src|data-original|data-src)=["\']([^"\']+)["\']'], inner)
|
||||
item = self._item(href, title, pic)
|
||||
if item and item['vod_id'] not in seen:
|
||||
seen.add(item['vod_id'])
|
||||
videos.append(item)
|
||||
return videos
|
||||
|
||||
def _first_raw(self, patterns, text):
|
||||
for pat in patterns:
|
||||
m = re.search(pat, text or '', flags=re.I)
|
||||
if m:
|
||||
return m.group(1)
|
||||
return ''
|
||||
|
||||
def _item(self, href, title, pic=''):
|
||||
title = self._clean(title)
|
||||
if not href or len(title) < 2:
|
||||
return None
|
||||
url = self._abs(href)
|
||||
return {'vod_id': url, 'vod_name': title, 'vod_pic': self._abs(pic), 'vod_remarks': self._year(title) or '点击查看'}
|
||||
|
||||
def _parse_detail_lines(self, html):
|
||||
lines = []
|
||||
h3s = list(re.finditer(r'<h3[^>]*>([^<]*播放地址[^<]*)</h3>', html or '', flags=re.I))
|
||||
for i, h3 in enumerate(h3s):
|
||||
section = html[h3.start():(h3s[i + 1].start() if i + 1 < len(h3s) else len(html))]
|
||||
eps = self._parse_eps(section, play=True)
|
||||
if eps:
|
||||
lines.append({'name': self._clean(h3.group(1)) or '在线播放', 'episodes': eps})
|
||||
if not lines:
|
||||
eps = self._parse_eps(html, play=True)
|
||||
if eps:
|
||||
lines.append({'name': '在线播放', 'episodes': eps})
|
||||
idx = (html or '').find('【下载地址】')
|
||||
if idx >= 0:
|
||||
sec = html[idx:]
|
||||
cut_points = [x for x in [sec.find('<h3', 6), sec.find('<div class="widget', 6)] if x > 0]
|
||||
if cut_points:
|
||||
sec = sec[:min(cut_points)]
|
||||
eps = self._parse_eps(sec, play=False)
|
||||
if eps:
|
||||
lines.append({'name': '下载地址', 'episodes': eps})
|
||||
return lines
|
||||
|
||||
def _parse_eps(self, html, play=True):
|
||||
eps, seen = [], set()
|
||||
if play:
|
||||
pat = r'<a\s+(?:[^>]*?\s+)?href\s*=\s*["\']([^"\']*/e/DownSys/play/[^"\']+)["\'][^>]*>(.*?)</a>'
|
||||
else:
|
||||
pat = r'<a\s+(?:[^>]*?\s+)?href\s*=\s*["\']([^"\']+)["\'][^>]*>(.*?)</a>'
|
||||
for href, title in re.findall(pat, html or '', flags=re.I):
|
||||
title = self._clean(title)
|
||||
if href in seen or not title:
|
||||
continue
|
||||
if not play and ('#respond' in href or 'category' in href):
|
||||
continue
|
||||
seen.add(href)
|
||||
if not play and (title == '链接' or len(title) < 2):
|
||||
low = href.lower()
|
||||
title = '夸克网盘' if 'quark' in low else ('迅雷网盘' if 'xunlei' in low else ('磁力链接' if low.startswith('magnet:') else '下载'))
|
||||
eps.append({'title': title, 'url': self._abs(href)})
|
||||
return eps
|
||||
|
||||
def _resolve_play_url(self, play_url):
|
||||
html = self._html(play_url)
|
||||
media = self._find_media(html, play_url)
|
||||
if media:
|
||||
return media
|
||||
iframe = re.search(r'<iframe[^>]+src\s*=\s*["\']([^"\']+)["\']', html or '', flags=re.I)
|
||||
if iframe:
|
||||
iframe_url = urljoin(play_url, iframe.group(1))
|
||||
media = self._find_media(self._html(iframe_url), iframe_url)
|
||||
if media:
|
||||
return media
|
||||
return None
|
||||
|
||||
def _find_media(self, html, base_url):
|
||||
patterns = [
|
||||
r'https?://[^\s"\'<>]+\.(?:m3u8|mp4)[^\s"\'<>]*',
|
||||
r'const\s+url\s*=\s*["\']([^"\']+\.(?:m3u8|mp4)[^"\']*)["\']',
|
||||
r'url\s*[:=]\s*["\']([^"\']+\.(?:m3u8|mp4)[^"\']*)["\']',
|
||||
r'["\']url["\']\s*:\s*["\']([^"\']+\.(?:m3u8|mp4)[^"\']*)["\']',
|
||||
]
|
||||
for pat in patterns:
|
||||
m = re.search(pat, html or '', flags=re.I)
|
||||
if m:
|
||||
val = m.group(1) if m.lastindex else m.group(0)
|
||||
return urljoin(base_url, val.replace('\\/', '/'))
|
||||
return None
|
||||
Reference in New Issue
Block a user