Sync all projects

This commit is contained in:
github-actions[bot]
2026-07-30 15:35:26 +00:00
parent d98caa5150
commit 3f366cb15e
30 changed files with 77528 additions and 26558 deletions
+25 -6
View File
@@ -134,6 +134,18 @@
"api":"csp_yt",
"homePage":"https://fgblh.github.io/uhuj.github.io/youtube.html"
},
{
"key": "视觉影院",
"name": "🐬视觉影院.py[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "ddvm",
"name": "🐬嘀嗒影视.py(关梯)[追剧]",
@@ -487,12 +499,6 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/小鸭子看看.py"
},
{
"key": "fY",
"name": "🐬枫叶影院.py(关梯)[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/枫叶影院.py"
},
{
"key":"6v",
"name":"🐬6V影视[追剧][html]",
@@ -647,6 +653,12 @@
"成长逆袭"
]
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py[短剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py[短剧]",
@@ -1128,6 +1140,13 @@
"type":3,
"api":"csp_wbbb",
"homePage":"https://fgblh.github.io/uhuj.github.io/javdb.html"
},
{
"key":"WhoStv.",
"name":"🐬WhoStv|🔞[成人][html]",
"type":3,
"api":"csp_wbbb",
"homePage":"https://fgblh.github.io/uhuj.github.io/WhoStv.html"
},
{
"key": "采集聚合成人",
+18 -6
View File
@@ -65,6 +65,18 @@
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/油管.py"
},
{
"key": "视觉影院",
"name": "🐬视觉影院.py[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "ppx",
"name": "🐬皮皮虾.py[追剧]",
"type": 3,
@@ -389,18 +401,18 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/采集聚合.py"
},
{
"key": "fY",
"name": "🐬枫叶影院.py(关梯)[追剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/枫叶影院.py"
},
{
"key": "MiFun",
"name": "🐬MiFun动漫.py[动漫]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/MiFun动漫.py"
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py[短剧]",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py[短剧]",
+18 -189
View File
@@ -127,6 +127,18 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/油管直播.py"
},
{
"key": "视觉影院",
"name": "🐬视觉影院.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "ppx",
"name": "🐬皮皮虾.py",
@@ -468,12 +480,6 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/采集聚合.py"
},
{
"key": "fY",
"name": "🐬枫叶影院.py(关梯子使用)",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/枫叶影院.py"
},
{
"key":"FY",
"name":"🐬枫叶影院(关梯子使用)",
@@ -543,189 +549,6 @@
"type":3,
"api":"csp_gy",
"homePage":"https://fgblh.github.io/uhuj.github.io/观影.html"
},
{
"key": "非凡",
"name": "🐬非凡影视",
"type": 1,
"api": "http://cj.ffzyapi.com/api.php/provide/vod/",
"searchable": 1,
"changeable": 1,
"categories": [
"国产动漫",
"日韩动漫",
"国产剧",
"韩国剧",
"日本剧",
"电影片",
"连续剧",
"综艺片",
"动漫片",
"动作片",
"喜剧片",
"爱情片",
"科幻片",
"恐怖片",
"剧情片",
"战争片",
"香港剧",
"欧美剧",
"记录片",
"台湾剧",
"海外剧",
"泰国剧",
"大陆综艺",
"港台综艺",
"日韩综艺",
"欧美综艺",
"欧美动漫",
"港台动漫",
"海外动漫"
]
},
{
"key": "闪电影视",
"name": "🐬闪电影视|追剧",
"type": 1,
"api": "http://sdzyapi.com/api.php/provide/vod/",
"searchable": 1,
"quickSearch": 1,
"categories": [
"国产剧",
"短剧",
"韩国剧",
"香港剧",
"台湾剧",
"欧美剧",
"动作片",
"科幻片",
"战争片",
"奇幻片",
"喜剧片",
"爱情片",
"恐怖片",
"犯罪片",
"悬疑片",
"惊悚片",
"剧情片",
"冒险片",
"记录片",
"动作片",
"日本剧",
"泰剧",
"国产综艺",
"港台综艺",
"欧美综艺",
"日韩综艺",
"国产动漫",
"港台动漫",
"日韩动漫"
]
},
{
"key": "牛牛影视",
"name": "🐬牛牛影视|追剧",
"type": 0,
"api": "https://api.niuniuzy.me/api.php/provide/vod/at/xml",
"searchable": 1,
"quickSearch": 1,
"categories": [
"动作片",
"喜剧片",
"爱情片",
"科幻片",
"恐怖片",
"剧情片",
"战争片",
"动画片",
"纪录片",
"预告片",
"邵氏电影",
"4K电影",
"国产剧",
"欧美剧",
"韩剧",
"日剧",
"港剧",
"台剧",
"泰剧",
"海外剧",
"国产动漫",
"日韩动漫",
"欧美动漫",
"港台动漫",
"海外动漫",
"影视解说",
"爽文短剧",
"有声动漫",
"女频恋爱",
"反转爽剧",
"古装仙侠",
"年代穿越",
"脑洞悬疑",
"现代都市",
"演唱会",
"大陆综艺",
"日韩综艺",
"港台综艺",
"欧美综艺",
"篮球",
"足球"
]
},
{
"key": "雲飛影视",
"name": "🐬雲飛影视|追剧",
"type": 1,
"api": "http://cj.lziapi.com/api.php/provide/vod/",
"searchable": 1,
"quickSearch": 1,
"filterable": 0,
"playurl": "json:https://lziplayer.com/?url=",
"categories": [
"国产剧",
"国产动漫",
"泰国剧",
"小湾剧",
"香港剧",
"欧美剧",
"韩国剧",
"日本剧",
"动漫",
"体育",
"加片",
"动作片",
"爱情片",
"喜剧片"
]
},
{"key": "火狐","name": "🐬火狐影视🦊|追剧","type": 1,"api": "https://hhzyapi.com/api.php/provide/vod/","searchable": 1,"quickSearch": 0,"filterable": 1,"categories": [ "内地剧", "动作片", "科幻片", "战争片", "喜剧片", "爱情片", "恐怖片", "犯罪片", "剧情片", "冒险片", "记录片", "韩剧", "香港剧", "台湾剧", "欧美剧", "日剧", "马泰剧", "体育赛事", "综艺", "动画片", "中国动漫", "日本动漫", "欧美动漫"]
},
{
"key": "量子2k",
"name": "🐬量子影视|追剧",
"type": 1,
"api": "http://cj.lziapi.com/api.php/provide/vod/",
"searchable": 1,
"quickSearch": 1,
"filterable": 0,
"playurl": "json:https://lziplayer.com/?url=",
"categories": [
"国产剧",
"国产动漫",
"泰国剧",
"台湾剧",
"香港剧",
"欧美剧",
"韩国剧",
"日本剧",
"动漫",
"体育",
"剧情片",
"动作片",
"爱情片",
"喜剧片"
]
},
{
"key": "API_如意",
@@ -798,6 +621,12 @@
"成长逆袭"
]
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py",
+18
View File
@@ -87,6 +87,18 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/油管.py"
},
{
"key": "视觉影院",
"name": "🐬视觉影院.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "4K影视",
"name": "🐬4K影视.py",
@@ -575,6 +587,12 @@
"成长逆袭"
]
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py",
+15 -3
View File
@@ -46,10 +46,16 @@
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/油管.py"
},
{
"key": "fY",
"name": "🐬枫叶影院.py(关梯子使用)",
"key": "视觉影院",
"name": "🐬视觉影院.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/枫叶影院.py"
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "ddvm",
@@ -418,6 +424,12 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/MiFun动漫.py"
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py",
+18 -6
View File
@@ -87,6 +87,18 @@
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/油管.py"
},
{
"key": "视觉影院",
"name": "🐬视觉影院.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/视觉影院.py"
},
{
"key": "袋鼠影视",
"name": "🐬袋鼠影视.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/袋鼠影视.py"
},
{
"key": "4K影视",
"name": "🐬4K影视.py",
@@ -224,12 +236,6 @@
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/jar_js/布布追剧.js",
"changeable": 0
},
{
"key": "fY",
"name": "🐬枫叶影院(关梯子使用)",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/枫叶影院.py"
},
{
"key": "hxq",
"name": "🐬韩小圈.py(关梯)",
@@ -732,6 +738,12 @@
"成长逆袭"
]
},
{
"key": "喜福网",
"name": "🐬喜福短剧.py",
"type": 3,
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/喜福短剧.py"
},
{
"key": "smdj",
"name": "🐬星芽短剧.py",
+233
View File
@@ -0,0 +1,233 @@
# coding=utf-8
# !/usr/bin/python
from Crypto.Util.Padding import unpad, pad
from Crypto.Cipher import ARC4, AES
from urllib.parse import unquote, quote
from base.spider import Spider
from datetime import datetime
from bs4 import BeautifulSoup
from base64 import b64decode
import urllib.request
import urllib.parse
import binascii
import requests
import hashlib
import base64
import uuid
import hmac
import json
import time
import sys
import re
import os
sys.path.append('..')
xurl = "https://minidrama-api.contentchina.com"
headerx = {
'User-Agent': 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.87 Safari/537.36'
}
headers = {
'Accept': '*/*',
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8,en-GB;q=0.7,en-US;q=0.6',
'Cache-Control': 'no-cache',
'Connection': 'keep-alive',
'Origin': 'https://minidrama.contentchina.com',
'Pragma': 'no-cache',
'Referer': 'https://minidrama.contentchina.com/',
'Sec-Fetch-Dest': 'empty',
'Sec-Fetch-Mode': 'cors',
'Sec-Fetch-Site': 'cross-site',
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/143.0.0.0 Safari/537.36 Edg/143.0.0.0',
'sec-ch-ua': '"Microsoft Edge";v="143", "Chromium";v="143", "Not A(Brand";v="24"',
'sec-ch-ua-mobile': '?0',
'sec-ch-ua-platform': '"Windows"',
}
class Spider(Spider):
def getName(self):
return "🌷"
def init(self, extend):
pass
def isVideoFormat(self, url):
pass
def manualVideoCheck(self):
pass
def homeContent(self, filter):
result = {"class": []}
def get_category_data():
url = f"{xurl}/web/v1/home/categoryList?isLeft=1"
response = requests.get(url=url, headers=headerx)
response.encoding = "utf-8"
return response.json()
def process_categories(data):
categories = []
for vod in data['data']['categories']:
category_info = {
"type_id": vod['id'],
"type_name": f"🌷{vod['name']}"
}
categories.append(category_info)
return categories
def build_result(categories):
return {"class": categories}
data = get_category_data()
categories = process_categories(data)
result = build_result(categories)
return result
def homeVideoContent(self):
pass
def categoryContent(self, cid, pg, filter, ext):
def get_page_number():
return int(pg) if pg else 1
def fetch_category_data(page_number):
url = f'{xurl}/web/v1/drama/list?pageSize=24&currentPage={str(page_number)}&filterCategories[]={cid}'
response = requests.get(url=url, headers=headerx)
response.encoding = "utf-8"
return response.json()
def parse_videos(data):
video_list = []
for vod in data['data']['data']:
video_info = {
"vod_id": f"{vod['albumId']}@{vod['total']}",
"vod_name": vod['title'],
"vod_pic": vod['coverUrl'],
"vod_remarks": f"🌷共{vod.get('total', '暂无备注')}"
}
video_list.append(video_info)
return video_list
def build_result(video_list, page_number):
return {
'list': video_list,
'page': page_number,
'pagecount': 9999,
'limit': 90,
'total': 999999
}
current_page = get_page_number()
response_data = fetch_category_data(current_page)
parsed_videos = parse_videos(response_data)
result = build_result(parsed_videos, current_page)
return result
def detailContent(self, ids):
def parse_ids():
did = ids[0]
parts = did.split("@")
return parts[0], int(parts[1])
def generate_play_items(base_id, total_count):
return [f"{i}${base_id}@{i}" for i in range(1, total_count + 1)]
def build_play_url(items):
return "#".join(items)
def create_video_info(did, play_url):
return {
"vod_id": did,
"vod_play_from": "喜福专线",
"vod_play_url": play_url
}
def build_result(video_info):
return {'list': [video_info]}
base_id, total_count = parse_ids()
play_items = generate_play_items(base_id, total_count)
play_url = build_play_url(play_items)
video_info = create_video_info(ids[0], play_url)
result = build_result(video_info)
return result
def playerContent(self, flag, id, vipFlags):
def parse_video_id():
parts = id.split("@")
return parts[0], parts[1]
def get_play_auth(album_id, seq):
url = f'{xurl}/web/v1/drama/play_auth?albumId={album_id}&seq={seq}'
response = requests.get(url=url, headers=headerx)
response.encoding = "utf-8"
return response.json()
def extract_credentials(data):
video_id = data['data']['vid']
play_auth_b64 = data['data']['playAuth']
play_auth_json = base64.b64decode(play_auth_b64).decode('utf-8')
return json.loads(play_auth_json), video_id
def build_request_params(credentials, video_id):
params = {
'Action': 'GetPlayInfo',
'Version': '2017-03-21',
'Format': 'JSON',
'AccessKeyId': credentials['AccessKeyId'],
'SecurityToken': credentials['SecurityToken'],
'VideoId': video_id,
'AuthInfo': credentials.get('AuthInfo', ''),
'Timestamp': datetime.utcnow().strftime('%Y-%m-%dT%H:%M:%SZ'),
'SignatureMethod': 'HMAC-SHA1',
'SignatureVersion': '1.0',
'SignatureNonce': str(uuid.uuid4()),
}
if 'PlayConfig' in credentials:
params['PlayConfig'] = json.dumps(credentials.get('PlayConfig'))
return params
def generate_signature(params, credentials):
sorted_params = sorted(params.items())
canonicalized_query_string = '&'.join(
[f"{urllib.parse.quote(k, safe='')}={urllib.parse.quote(str(v), safe='')}" for k, v in sorted_params])
string_to_sign = f"GET&%2F&{urllib.parse.quote(canonicalized_query_string, safe='')}"
key = credentials['AccessKeySecret'] + "&"
signature = hmac.new(key.encode('utf-8'), string_to_sign.encode('utf-8'), hashlib.sha1).digest()
return base64.b64encode(signature).decode('utf-8'), canonicalized_query_string
def get_final_play_url(credentials, canonicalized_query_string, signature_b64):
final_url = f"https://vod.{credentials.get('Region', 'cn-shanghai')}.aliyuncs.com/?{canonicalized_query_string}&Signature={urllib.parse.quote(signature_b64, safe='')}"
response = requests.get(url=final_url, headers=headers)
response.encoding = "utf-8"
data = response.json()
return data['PlayInfoList']['PlayInfo'][0]['PlayURL']
def build_result(play_url):
return {
"parse": 0,
"playUrl": '',
"url": play_url,
"header": headerx
}
album_id, seq = parse_video_id()
auth_data = get_play_auth(album_id, seq)
credentials, video_id = extract_credentials(auth_data)
request_params = build_request_params(credentials, video_id)
signature_b64, canonicalized_query_string = generate_signature(request_params, credentials)
play_url = get_final_play_url(credentials, canonicalized_query_string, signature_b64)
result = build_result(play_url)
return result
def searchContentPage(self, key, quick, pg):
pass
def searchContent(self, key, quick, pg="1"):
return self.searchContentPage(key, quick, '1')
def localProxy(self, params):
if params['type'] == "m3u8":
return self.proxyM3u8(params)
elif params['type'] == "media":
return self.proxyMedia(params)
elif params['type'] == "ts":
return self.proxyTs(params)
return None
-392
View File
@@ -1,392 +0,0 @@
# -*- coding: utf-8 -*-
import re, urllib.parse
import json
from bs4 import BeautifulSoup
import requests
from base.spider import Spider as BaseSpider
class Spider(BaseSpider):
def init(self, extend=""):
self.host = "https://maihaolian.com"
self.headers = {
"User-Agent": "Mozilla/5.0 (iPhone; CPU iPhone OS 16_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.0 Mobile/15E148 Safari/604.1",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9",
}
def getName(self):
return '枫叶影院'
def homeContent(self, filter):
return {"class": [
{'type_id': "/label/qq", 'type_name': "腾讯VIP精选"},
{'type_id': "/label/bli", 'type_name': "B站VIP精选"},
{'type_id': "/label/youku", 'type_name': "优酷VIP精选"},
{"type_id": "5", "type_name": "红果短剧"},
{"type_id": "2", "type_name": "电视剧"},
{"type_id": "1", "type_name": "电影"},
{"type_id": "4", "type_name": "动漫"},
{"type_id": "3", "type_name": "综艺"},
], "filters": self._build_filters()}
def _build_filters(self):
area = [{"n": "全部", "v": ""}, {"n": "大陆", "v": "大陆"}, {"n": "香港", "v": "香港"},
{"n": "台湾", "v": "台湾"}, {"n": "美国", "v": "美国"}, {"n": "韩国", "v": "韩国"},
{"n": "日本", "v": "日本"}, {"n": "泰国", "v": "泰国"}, {"n": "新加坡", "v": "新加坡"},
{"n": "马来西亚", "v": "马来西亚"}, {"n": "印度", "v": "印度"}, {"n": "英国", "v": "英国"},
{"n": "法国", "v": "法国"}, {"n": "加拿大", "v": "加拿大"}, {"n": "西班牙", "v": "西班牙"},
{"n": "俄罗斯", "v": "俄罗斯"}, {"n": "其它", "v": "其它"}]
year = [{"n": "全部", "v": ""}, {"n": "2026", "v": "2026"}, {"n": "2025", "v": "2025"},
{"n": "2024", "v": "2024"}, {"n": "2023", "v": "2023"}, {"n": "2022", "v": "2022"},
{"n": "2021", "v": "2021"}, {"n": "2020", "v": "2020"}, {"n": "2019", "v": "2019"},
{"n": "2018", "v": "2018"}, {"n": "2017", "v": "2017"}, {"n": "2016", "v": "2016"},
{"n": "2015", "v": "2015"}, {"n": "2014", "v": "2014"}, {"n": "2013", "v": "2013"},
{"n": "2012", "v": "2012"}, {"n": "2011", "v": "2011"}, {"n": "2010", "v": "2010"},
{"n": "2009", "v": "2009"}, {"n": "2008", "v": "2008"}, {"n": "2007", "v": "2007"},
{"n": "2006", "v": "2006"}, {"n": "2005", "v": "2005"}, {"n": "2004", "v": "2004"}]
lang = [{"n": "全部", "v": ""}, {"n": "国语", "v": "国语"}, {"n": "英语", "v": "英语"},
{"n": "粤语", "v": "粤语"}, {"n": "闽南语", "v": "闽南语"}, {"n": "韩语", "v": "韩语"},
{"n": "日语", "v": "日语"}, {"n": "法语", "v": "法语"}, {"n": "德语", "v": "德语"},
{"n": "其它", "v": "其它"}]
sort = [{"n": "时间", "v": "time"}, {"n": "人气", "v": "hits"}, {"n": "评分", "v": "score"}]
letter = [{"n": "全部", "v": ""}, {"n": "A", "v": "A"}, {"n": "B", "v": "B"}, {"n": "C", "v": "C"},
{"n": "D", "v": "D"}, {"n": "E", "v": "E"}, {"n": "F", "v": "F"}, {"n": "G", "v": "G"},
{"n": "H", "v": "H"}, {"n": "I", "v": "I"}, {"n": "J", "v": "J"}, {"n": "K", "v": "K"},
{"n": "L", "v": "L"}, {"n": "M", "v": "M"}, {"n": "N", "v": "N"}, {"n": "O", "v": "O"},
{"n": "P", "v": "P"}, {"n": "Q", "v": "Q"}, {"n": "R", "v": "R"}, {"n": "S", "v": "S"},
{"n": "T", "v": "T"}, {"n": "U", "v": "U"}, {"n": "V", "v": "V"}, {"n": "W", "v": "W"},
{"n": "X", "v": "X"}, {"n": "Y", "v": "Y"}, {"n": "Z", "v": "Z"}, {"n": "0-9", "v": "0-9"}]
return {
"2": [
{"key": "class", "name": "类型",
"value": [{"n": "全部", "v": "2"}, {"n": "国产剧", "v": "13"}, {"n": "日韩剧", "v": "15"},
{"n": "海外剧", "v": "16"}]},
{"key": "area", "name": "地区", "value": area},
{"key": "genre", "name": "剧情", "value": [{"n": v[0], "v": v[1]} for v in
[("全部", ""), ("古装", "古装"), ("战争", "战争"),
("青春偶像", "青春偶像"), ("喜剧", "喜剧"),
("家庭", "家庭"), ("犯罪", "犯罪"), ("动作", "动作"),
("奇幻", "奇幻"), ("剧情", "剧情"), ("历史", "历史"),
("经典", "经典"), ("乡村", "乡村"), ("情景", "情景"),
("商战", "商战"), ("网剧", "网剧"), ("其他", "其他")]]},
{"key": "year", "name": "年份", "value": year},
{"key": "lang", "name": "语言", "value": lang},
{"key": "letter", "name": "字母", "value": letter},
{"key": "sort", "name": "排序", "value": sort},
],
"1": [
{"key": "class", "name": "类型",
"value": [{"n": "全部", "v": "1"}, {"n": "动作片", "v": "6"}, {"n": "喜剧片", "v": "7"},
{"n": "恐怖片", "v": "8"}, {"n": "科幻片", "v": "9"}, {"n": "爱情片", "v": "10"},
{"n": "剧情片", "v": "11"}, {"n": "战争片", "v": "12"}, {"n": "纪录片", "v": "20"}]},
{"key": "area", "name": "地区", "value": area},
{"key": "genre", "name": "剧情", "value": [{"n": v[0], "v": v[1]} for v in
[("全部", ""), ("喜剧", "喜剧"), ("爱情", "爱情"),
("恐怖", "恐怖"), ("动作", "动作"), ("科幻", "科幻"),
("剧情", "剧情"), ("战争", "战争"), ("警匪", "警匪"),
("犯罪", "犯罪"), ("动画", "动画"), ("奇幻", "奇幻"),
("武侠", "武侠"), ("冒险", "冒险"), ("枪战", "枪战"),
("悬疑", "悬疑"), ("惊悚", "惊悚"), ("经典", "经典"),
("青春", "青春"), ("文艺", "文艺"), ("微电影", "微电影"),
("古装", "古装"), ("历史", "历史"), ("运动", "运动"),
("农村", "农村"), ("儿童", "儿童"),
("网络电影", "网络电影")]]},
{"key": "year", "name": "年份", "value": year},
{"key": "lang", "name": "语言", "value": lang},
{"key": "letter", "name": "字母", "value": letter},
{"key": "sort", "name": "排序", "value": sort},
],
"4": [
{"key": "class", "name": "类型",
"value": [{"n": "全部", "v": "4"}, {"n": "国产动漫", "v": "25"}, {"n": "日韩动漫", "v": "26"}]},
{"key": "genre", "name": "剧情", "value": [{"n": v[0], "v": v[1]} for v in
[("全部", ""), ("情感", "情感"), ("科幻", "科幻"),
("热血", "热血"), ("推理", "推理"), ("搞笑", "搞笑"),
("冒险", "冒险"), ("奇幻", "奇幻"), ("战斗", "战斗"),
("校园", "校园"), ("萝莉", "萝莉"), ("治愈", "治愈"),
("原创", "原创"), ("亲子", "亲子"), ("益智", "益智"),
("励志", "励志"), ("其他", "其他")]]},
{"key": "area", "name": "地区",
"value": [{"n": "全部", "v": ""}, {"n": "大陆", "v": "大陆"}, {"n": "香港", "v": "香港"},
{"n": "台湾", "v": "台湾"}, {"n": "美国", "v": "美国"}, {"n": "韩国", "v": "韩国"},
{"n": "日本", "v": "日本"}, {"n": "法国", "v": "法国"}, {"n": "英国", "v": "英国"},
{"n": "其它", "v": "其它"}]},
{"key": "year", "name": "年份", "value": year},
{"key": "lang", "name": "语言", "value": lang},
{"key": "letter", "name": "字母", "value": letter},
{"key": "sort", "name": "排序", "value": sort},
],
"3": [
{"key": "class", "name": "类型",
"value": [{"n": "全部", "v": "3"}, {"n": "大陆综艺", "v": "21"}, {"n": "日韩综艺", "v": "22"}]},
{"key": "genre", "name": "剧情", "value": [{"n": v[0], "v": v[1]} for v in
[("全部", ""), ("选秀", "选秀"), ("情感", "情感"),
("访谈", "访谈"), ("播报", "播报"), ("音乐", "音乐"),
("美食", "美食"), ("旅游", "旅游"), ("搞笑", "搞笑"),
("游戏", "游戏"), ("亲子", "亲子"), ("其它", "其它")]]},
{"key": "area", "name": "地区",
"value": [{"n": "全部", "v": ""}, {"n": "大陆", "v": "大陆"}, {"n": "香港", "v": "香港"},
{"n": "台湾", "v": "台湾"}, {"n": "美国", "v": "美国"}, {"n": "韩国", "v": "韩国"},
{"n": "日本", "v": "日本"}, {"n": "英国", "v": "英国"}, {"n": "其它", "v": "其它"}]},
{"key": "year", "name": "年份", "value": year},
{"key": "lang", "name": "语言", "value": lang},
{"key": "letter", "name": "字母", "value": letter},
{"key": "sort", "name": "排序", "value": sort},
],
}
def homeVideoContent(self):
html = self._fetch('/')
return {"list": self._parse_video_list(html)}
def categoryContent(self, tid, pg, filter, extend):
# 构建筛选参数:参照歪比巴卜,直接取extend里的值,fallback到filter
if tid.startswith('/label'):
url = f'{tid}/page/{pg}.html'
html = self._fetch(url)
items = self._parse_video_list(html)
page = int(pg)
page_count = page if len(items) < 24 else page + 2
return {"list": items, "page": page, "pagecount": page_count, "limit": 24, "total": page_count * 24}
args = {}
if extend and isinstance(extend, dict):
for k, v in extend.items():
if v:
args[k] = str(v)
if isinstance(filter, dict):
for k, v in filter.items():
if v and k not in args:
args[k] = str(v)
route_tid = args.get('class', args.get('tid', str(tid)))
area = args.get('area', '')
genre = args.get('genre', '')
year = args.get('year', '')
lang = args.get('lang', '')
letter = args.get('letter', '')
sort = args.get('sort', '')
# 无筛选走正常分页
if not area and not genre and not year and not lang and not letter and not sort:
url = f'/cupfox-list/{route_tid}--------{pg}---.html'
html = self._fetch(url)
items = self._parse_video_list(html)
page = int(pg)
soup = BeautifulSoup(html, 'html.parser')
pagecount = page
for a in soup.select('a.page-link'):
if a.text == '尾页':
m = re.search(r'---(\d+)---', a.get('href', ''))
if m:
pagecount = int(m.group(1))
break
if not items:
pagecount = 0
return {"list": items, "page": page, "pagecount": pagecount, "limit": 36, "total": 9999}
# 有筛选:{tid}-{area}-{sort}-{genre}-{lang}-{letter}------{year}.html
segs = [route_tid, area, sort, genre, lang, letter, '', '', year]
url = '/cupfox-list/' + '-'.join(segs) + '.html'
html = self._fetch(url)
items = self._parse_video_list(html)
return {"list": items, "page": 1, "pagecount": 1, "limit": 36, "total": 9999}
def detailContent(self, ids):
result = {"list": []}
vid = ids[0].split(',')[0].strip()
try:
html = self._fetch(f'/detail/{vid}.html')
if not html: return result
soup = BeautifulSoup(html, 'html.parser')
vod_name = soup.select_one('h3.slide-info-title')
vod_name = vod_name.text.strip() if vod_name else ''
vod_pic = soup.select_one('img.lazy')
vod_pic = self._fix_pic(vod_pic.get('data-src', '')) if vod_pic else ''
vod_director = ''
vod_actor = ''
for el in soup.select('.slide-info'):
text = el.get_text(' ').strip()
if text.startswith('导演:'):
vod_director = text.replace('导演:', '').strip()
elif text.startswith('演员:'):
vod_actor = text.replace('演员:', '').strip()
vod_content = soup.select_one('#height_limit')
vod_content = vod_content.get_text(' ', strip=True) if vod_content else ''
play_from, play_url = [], []
for tab in soup.select('.anthology-tab a.swiper-slide'):
src_name = re.sub(r'<[^>]+>', '', str(tab)).strip() or tab.get_text(' ', strip=True).strip()
if src_name:
play_from.append(src_name)
tab_blocks = soup.select('.anthology-list-box')
for i, block in enumerate(tab_blocks):
ep_list = []
for a in block.select('li a'):
href = a.get('href', '')
m = re.search(r'/play/(.*?)\.html', href)
if m:
ep_list.append(f'{a.text.strip()}${vid}-{m.group(1)}')
ep_list.reverse()
if ep_list and i < len(play_from):
play_url.append('#'.join(ep_list))
valid_from = [pf for i, pf in enumerate(play_from) if i < len(play_url)]
result["list"].append({
"vod_id": vid, "vod_name": vod_name, "vod_pic": vod_pic,
"vod_director": vod_director, "vod_actor": vod_actor,
"vod_content": vod_content,
"vod_play_from": "$$$".join(valid_from),
"vod_play_url": "$$$".join(play_url),
})
except:
pass
return result
def searchContent(self, key, quick, pg="1"):
try:
decoded = urllib.parse.unquote(key)
except:
decoded = key
html = self._fetch(f'/cupfox-search/{urllib.parse.quote(decoded)}----------{pg}---.html')
items = self._parse_search_list(html)
return {"list": items, "page": int(pg), "pagecount": 1, "limit": 36, "total": len(items)}
def playerContent(self, flag, id, vipFlags):
url = ''
try:
url = id if id.startswith('http') else f'{self.host}/play/{id}.html'
html = self._fetch(url)
if html:
m = re.search(r'player_aaaa=(.*?)</script>', html, re.S)
if m:
try:
pd = json.loads(m.group(1))
except Exception as e:
print(e)
pd = {}
# print('pd:', pd)
play_url = pd.get('url')
play_id = pd.get('from')
api_map = {
'YYNB': 'https://zzrs.mfdyvip.com/player/mplayer.php',
'JD4K': 'https://fgsrg.hzqingshan.com/player/mplayer.php',
}
if not play_url:
return {"parse": 0, "url": 'https://php.doube.eu.org/error.m3u8',
"header": {'User-Agent': 'Mozilla/5.0'}}
if play_url.startswith('http') and (play_url.endswith('.m3u8') or play_url.endswith('.mp4')):
return {"parse": 0, "url": play_url, "header": {'User-Agent': 'Mozilla/5.0'}}
else:
headers = {
'User-Agent': "Mozilla/5.0 (Linux; Android 10; K) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/137.0.0.0 Safari/537.36",
'Accept': "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
'accept-language': "zh-CN,zh;q=0.9",
'cache-control': "no-cache",
'pragma': "no-cache",
'priority': "u=0, i",
'referer': "https://www.ht10010.com/",
'Content-Type': 'application/x-www-form-urlencoded',
}
response = requests.get(f"https://fgsrg.hzqingshan.com/player/?url={play_url}", headers=headers)
token = re.search(r'data-te="(.*?)"', response.text)
if token:
token = token.group(1)
payload = {
'url': play_url,
'token': token
}
# print('payload', payload)
try:
response = self.post(api_map[play_id], data=payload, headers=headers)
response.raise_for_status()
result = response.json()
# print('result:', result)
if result['code'] == 200 and 'url' in result:
play_url = result['url']
return {"parse": 0, "url": play_url, "header": {
'User-Agent': 'Mozilla/5.0 (iPhone; CPU iPhone OS 16_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.0 Mobile/15E148 Safari/604.1'}}
except Exception as e:
print(e)
except Exception as e:
print(e)
return {"parse": 1, "url": url}
def localProxy(self, param=''):
return {}
def isVideoFormat(self, url):
return False
def manualVideoCheck(self):
return False
def _fetch(self, url):
try:
if not url.startswith('http'):
url = self.host + url
rsp = self.fetch(url, headers=self.headers)
return rsp.text if rsp else ''
except:
return ''
def _fix_pic(self, u):
if not u: return ''
if u.startswith('//'): return 'https:' + u
return u.replace('&amp;', '&')
def _parse_video_list(self, html):
videos, seen = [], set()
soup = BeautifulSoup(html, 'html.parser')
cards = soup.select('a.public-list-exp')
for a in cards:
href = a.get('href', '')
m = re.search(r'/detail/(\d+)\.html', href)
if not m: continue
vod_id = m.group(1)
if vod_id in seen: continue
seen.add(vod_id)
span = ','.join([span.text for span in a.select('span.public-prt')])
# print('span', span)
vod_name = a.get('title', '') or (a.select_one('img') and a.select_one('img').get('alt', '')) or ''
pic_el = a.select_one('img')
vod_pic = self._fix_pic(pic_el.get('data-src', '')) if pic_el else ''
remark_el = a.select_one('.ft2') or a.select_one('.public-list-prb')
vod_remarks = remark_el.text.strip() if remark_el else ''
videos.append(
{"vod_id": vod_id, "vod_name": vod_name.strip(), "vod_pic": vod_pic, "vod_remarks": vod_remarks, "vod_year": span})
return videos
def _parse_search_list(self, html):
videos, seen = [], set()
soup = BeautifulSoup(html, 'html.parser')
cards = soup.select('a.public-list-exp')
for a in cards:
href = a.get('href', '')
m = re.search(r'/detail/(\d+)\.html', href)
if not m: continue
vod_id = m.group(1)
if vod_id in seen: continue
seen.add(vod_id)
pic_el = a.select_one('img')
vod_pic = self._fix_pic(pic_el.get('data-src', '')) if pic_el else ''
title_el = soup.select_one(f'a.thumb-txt[href="/detail/{vod_id}.html"]')
if title_el:
vod_name = title_el.text.strip()
else:
vod_name = a.select_one('img') and a.select_one('img').get('alt', '') or ''
remark_el = a.select_one('.public-list-prb') or a.select_one('.ft2')
vod_remarks = remark_el.text.strip() if remark_el else ''
videos.append(
{"vod_id": vod_id, "vod_name": vod_name.strip(), "vod_pic": vod_pic, "vod_remarks": vod_remarks})
return videos
if __name__ == '__main__':
sp = Spider()
sp.init()
# 20067-5-189
print(sp.categoryContent('/label/qq','1',True, {}))
# print(sp.playerContent('', '20067-6-189', []))
# print(sp.playerContent('', '20067-5-189', []))
pass
+208
View File
@@ -0,0 +1,208 @@
# -*- coding: utf-8 -*-
import sys
import re
import requests
from urllib.parse import quote, unquote
sys.path.append('..')
from base.spider import Spider
class Spider(Spider):
def init(self, extend=""):
self.host = "https://dsystv.com"
self.headers = {
"User-Agent": "Mozilla/5.0 (Linux; Android 11; SAMSUNG SM-G973U) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/87.0.4280.141 Mobile Safari/537.36",
"Referer": self.host + "/",
"Origin": self.host
}
def getName(self):
return "袋鼠影视"
def isVideoFormat(self, url):
return bool(re.search(r'\.(m3u8|mp4|flv|avi|mkv|mov|ts)(\?|$)', url or "", re.I))
def manualVideoCheck(self):
return False
def homeContent(self, filter):
return {
"class": [
{"type_id": "1", "type_name": "电影"},
{"type_id": "2", "type_name": "电视剧"},
{"type_id": "3", "type_name": "综艺"},
{"type_id": "4", "type_name": "动漫"},
{"type_id": "44", "type_name": "短剧"}
]
}
def homeVideoContent(self):
return {"list": self.parseList(self.get(self.host + "/"))}
def categoryContent(self, tid, pg, filter, extend):
url = self.host + ("/frim/index" + str(tid) + ".html" if str(pg) == "1" else "/search.php?searchtype=5&tid=" + str(tid) + "&page=" + str(pg))
html = self.get(url)
return {
"page": int(pg),
"pagecount": 999,
"limit": 24,
"total": 999999,
"list": self.parseList(html)
}
def detailContent(self, ids):
vid = ids[0]
html = self.get(self.host + "/movie/index" + vid + ".html")
name = self.clean(self.match(html, r'<h1[^>]*>(.*?)</h1>') or self.match(html, r'<meta property="og:title" content="(.*?)"'))
name = name.replace("全集在线观看 - 国产剧 | 袋鼠影视", "").replace("全集在线观看 - 袋鼠影视", "").replace("", "").replace("", "").strip()
pic = self.fix(self.match(html, r'<meta property="og:image" content="(.*?)"') or self.match(html, r'<a[^>]+class="[^"]*videopic[^"]*"[\s\S]*?<img[^>]+(?:data-original|data-src)=["\']([^"\']+)') or self.match(html, r'<a[^>]+class="[^"]*videopic[^"]*"[\s\S]*?<img[^>]+src=["\']([^"\']+)'))
desc = self.clean(self.match(html, r'<div class="plot"[^>]*>\s*<p>(.*?)</p>') or self.match(html, r'<meta property="og:description" content="(.*?)"'))
actor = self.match(html, r'<li[^>]+data-video-meta=["\']([^"\']*)["\'][^>]*><span class="text-muted">主演:')
director = self.match(html, r'<li[^>]+data-video-meta=["\']([^"\']*)["\'][^>]*><span class="text-muted">导演:')
year = self.clean(self.match(html, r'年份:</span>([^<]+)'))
area = self.clean(self.match(html, r'地区:</span>([^<]+)'))
lang = self.clean(self.match(html, r'语言:</span>([^<]+)'))
cate = self.clean(self.match(html, r'类型:</span><a[^>]*>(.*?)</a>'))
remarks = self.clean(self.match(html, r'<span class="note textbg">(.*?)</span>'))
tabs = re.findall(r'<a class="option"[\s\S]*?title=["\']([^"\']+)["\'][\s\S]*?</a>', html)
panels = re.findall(r'<div[^>]+class=["\']playlist[^"\']*["\'][^>]*>\s*<ul[^>]*>([\s\S]*?)</ul>', html, re.S)
play_from = []
play_url = []
for i, p in enumerate(panels):
eps = []
for m in re.finditer(r'<a[^>]+title=["\']([^"\']+)["\'][^>]+href=["\']([^"\']*?/play/[^"\']+)["\']', p):
t = self.clean(m.group(1))
u = self.fix(m.group(2))
if t and u:
eps.append(t + "$" + u)
if not eps:
for m in re.finditer(r'<a[^>]+href=["\']([^"\']*?/play/[^"\']+)["\'][^>]*>(.*?)</a>', p):
t = self.clean(m.group(2))
u = self.fix(m.group(1))
if t and u:
eps.append(t + "$" + u)
if eps:
key = self.clean(tabs[i]) if i < len(tabs) else "线路" + str(i + 1)
if key not in play_from:
play_from.append(key)
play_url.append("#".join(eps))
if not play_url:
eps = []
for m in re.finditer(r'<a[^>]+href=["\']([^"\']*?/play/' + vid + r'-[^"\']+)["\'][^>]*>(.*?)</a>', html):
t = self.clean(m.group(2)) or "播放"
u = self.fix(m.group(1))
if t and u:
eps.append(t + "$" + u)
if eps:
play_from.append("默认")
play_url.append("#".join(eps))
return {
"list": [{
"vod_id": vid,
"vod_name": name,
"vod_pic": pic,
"vod_remarks": remarks,
"type_name": cate,
"vod_year": year,
"vod_area": area,
"vod_lang": lang,
"vod_actor": actor,
"vod_director": director,
"vod_content": desc,
"vod_play_from": "$$$".join(play_from),
"vod_play_url": "$$$".join(play_url)
}]
}
def searchContent(self, key, quick, pg="1"):
html = ""
try:
r = requests.post(self.host + "/search.php", headers=self.headers, data={"searchword": key}, timeout=15)
r.encoding = r.apparent_encoding or "utf-8"
html = r.text
except Exception:
html = ""
data = self.parseList(html)
if not data:
html = self.get(self.host + "/search.php?searchword=" + quote(key) + "&page=" + str(pg))
data = self.parseList(html)
return {"list": data, "page": int(pg)}
def playerContent(self, flag, id, vipFlags):
html = self.get(id)
url = self.match(html, r'var\s+now\s*=\s*["\']([^"\']+)["\']')
url = unquote(url) if url else id
return {"parse": 0 if self.isVideoFormat(url) else 1, "playUrl": "", "url": url, "header": self.headers}
def localProxy(self, param):
return [404, "text/plain", "", ""]
def destroy(self):
return "正在Destroy"
def get(self, url):
try:
r = requests.get(url, headers=self.headers, timeout=15, verify=False)
r.encoding = r.apparent_encoding or "utf-8"
return r.text
except Exception:
return ""
def match(self, text, rule):
m = re.search(rule, text or "", re.S)
return m.group(1) if m else ""
def clean(self, text):
return re.sub(r"\s+", " ", re.sub(r"<.*?>", "", text or "").replace("&nbsp;", " ")).strip()
def fix(self, url):
if not url:
return ""
if url.startswith("//"):
return "https:" + url
if url.startswith("/"):
return self.host + url
return url
def parseList(self, html):
res = []
seen = set()
for m in re.finditer(r'<a[^>]+class=["\'][^"\']*videopic[^"\']*["\'][^>]+href=["\']/movie/index(\d+)\.html["\'][^>]*title=["\']([^"\']+)["\']([\s\S]{0,1200}?)</a>', html or "", re.S):
vid = m.group(1)
if vid in seen:
continue
seen.add(vid)
item = m.group(0) + m.group(3)
name = self.clean(m.group(2))
pics = re.findall(r'(?:data-original|data-src)=["\']([^"\']+\.(?:jpg|jpeg|png|webp|gif)[^"\']*)["\']', item, re.I)
if not pics:
pics = re.findall(r'src=["\']([^"\']+\.(?:jpg|jpeg|png|webp|gif)[^"\']*)["\']', item, re.I)
pic = ""
for p in pics:
if "load.gif" not in p and "nopic" not in p and "logo" not in p and "templets" not in p:
pic = self.fix(p)
break
remarks = self.clean(self.match(item, r'<span[^>]+class=["\'][^"\']*note[^"\']*["\'][^>]*>(.*?)</span>') or self.match(item, r'<span[^>]+class=["\'][^"\']*textbg[^"\']*["\'][^>]*>(.*?)</span>'))
if name:
res.append({
"vod_id": vid,
"vod_name": name,
"vod_pic": pic,
"vod_remarks": remarks
})
if not res:
for m in re.finditer(r'href=["\']/movie/index(\d+)\.html["\'][^>]*title=["\']([^"\']+)["\'][\s\S]{0,1200}?<img[^>]+([^>]+)>', html or "", re.S):
vid = m.group(1)
if vid in seen:
continue
seen.add(vid)
img = m.group(3)
pic = self.match(img, r'(?:data-original|data-src)=["\']([^"\']+)["\']') or self.match(img, r'src=["\']([^"\']+)["\']')
if "load.gif" in pic or "templets" in pic:
pic = ""
res.append({
"vod_id": vid,
"vod_name": self.clean(m.group(2)),
"vod_pic": self.fix(pic),
"vod_remarks": ""
})
return res
+311
View File
@@ -0,0 +1,311 @@
# -*- coding: utf-8 -*-
import re, json, requests
from urllib.parse import quote
from lxml import etree
from base.spider import Spider
class Spider(Spider):
def __init__(self):
self.name = "sypfjy"
self.host = "https://www.sypfjy.com"
self.header = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
'Accept-Language': 'zh-CN,zh;q=0.9',
'Referer': self.host
}
def getName(self):
return self.name
def init(self, extend=''):
pass
def _get(self, url, params=None):
r = requests.get(url, headers=self.header, params=params, timeout=20)
r.encoding = 'utf-8'
return r.text
def _post(self, url, data=None):
r = requests.post(url, headers=self.header, data=data, timeout=20)
r.encoding = 'utf-8'
return r.text
def _fix_url(self, url):
if not url:
return ''
if url.startswith('//'):
return 'https:' + url
if url.startswith('/'):
return self.host + url
return url
def _parse_pic(self, elem):
if elem is None:
return ''
if elem.tag == 'img':
pic = elem.get('data-src') or elem.get('src', '')
else:
imgs = elem.xpath('.//img')
pic = (imgs[0].get('data-src') or imgs[0].get('src', '')) if imgs else ''
if pic and pic.startswith('data:image'):
pic = ''
return self._fix_url(pic)
def _parse_text(self, elem):
if elem is None:
return ''
return ''.join(elem.itertext()).strip()
def _build_vodshow(self, tid, area, order, cls, lang, page, year):
"""12 段位 URL: type-area-order-class-lang-_-_-_-page-_-_-year
默认排序 time 不写入 URL(留空)"""
def q(v):
return quote(v) if v else ''
# 默认排序不写
order_v = q(order) if order and order != 'time' else ''
# 页码 1 不写(留空默认)
page_v = str(page) if int(page) > 1 else ''
segs = [str(tid), q(area), order_v, q(cls), q(lang), '', '', '', page_v, '', '', q(year)]
return f"{self.host}/vodshow/{'-'.join(segs)}.html"
def _parse_list_item(self, item):
a = item.xpath('.//div[contains(@class,"video-name")]//a')
if not a:
return None
a = a[0]
href = a.get('href', '')
m = re.search(r'/voddetail/(\d+)\.html', href)
if not m:
return None
vid = m.group(1)
vod_name = a.get('title', '').strip() or (a.text or '').strip()
vod_pic = self._parse_pic(item)
caption = item.xpath('.//div[contains(@class,"module-item-caption")]//text()')
parts = [s.strip() for s in caption if s.strip()]
vod_remarks = ' / '.join(parts[-2:]) if len(parts) >= 2 else (parts[-1] if parts else '')
return {"vod_id": vid, "vod_name": vod_name, "vod_pic": vod_pic, "vod_remarks": vod_remarks}
def homeContent(self, filter):
classes = [
{"type_name": "电视剧", "type_id": "2"},
{"type_name": "电影", "type_id": "dianying"},
{"type_name": "动漫", "type_id": "dongman"},
{"type_name": "短剧", "type_id": "remenduanju"},
{"type_name": "综艺", "type_id": "zongyi"},
{"type_name": "体育", "type_id": "tiyusaishi"}
]
filters = {}
order_vals = [{"n": "时间排序", "v": "time"}, {"n": "人气排序", "v": "hits"}, {"n": "评分排序", "v": "score"}]
area_vals = [{"n": "全部", "v": ""}, {"n": "内地", "v": "内地"}, {"n": "中国", "v": "中国"}, {"n": "香港", "v": "香港"}, {"n": "台湾", "v": "台湾"}, {"n": "韩国", "v": "韩国"}, {"n": "日本", "v": "日本"}, {"n": "美国", "v": "美国"}, {"n": "泰国", "v": "泰国"}, {"n": "英国", "v": "英国"}, {"n": "新加坡", "v": "新加坡"}, {"n": "其他", "v": "其他"}]
class_vals = [{"n": "全部", "v": ""}, {"n": "古装", "v": "古装"}, {"n": "战争", "v": "战争"}, {"n": "青春偶像", "v": "青春偶像"}, {"n": "喜剧", "v": "喜剧"}, {"n": "家庭", "v": "家庭"}, {"n": "犯罪", "v": "犯罪"}, {"n": "动作", "v": "动作"}, {"n": "奇幻", "v": "奇幻"}, {"n": "剧情", "v": "剧情"}, {"n": "历史", "v": "历史"}, {"n": "经典", "v": "经典"}, {"n": "科幻", "v": "科幻"}, {"n": "悬疑", "v": "悬疑"}, {"n": "爱情", "v": "爱情"}, {"n": "惊悚", "v": "惊悚"}, {"n": "恐怖", "v": "恐怖"}, {"n": "灾难", "v": "灾难"}, {"n": "网络", "v": "网络"}, {"n": "商战", "v": "商战"}, {"n": "乡村", "v": "乡村"}, {"n": "情景", "v": "情景"}, {"n": "武侠", "v": "武侠"}, {"n": "冒险", "v": "冒险"}, {"n": "谍战", "v": "谍战"}, {"n": "其他", "v": "其他"}]
lang_vals = [{"n": "全部", "v": ""}, {"n": "国语", "v": "国语"}, {"n": "英语", "v": "英语"}, {"n": "粤语", "v": "粤语"}, {"n": "闽南语", "v": "闽南语"}, {"n": "韩语", "v": "韩语"}, {"n": "日语", "v": "日语"}, {"n": "其他", "v": "其他"}]
year_vals = [{"n": "全部", "v": ""}, {"n": "2026", "v": "2026"}, {"n": "2025", "v": "2025"}, {"n": "2024", "v": "2024"}, {"n": "2023", "v": "2023"}, {"n": "2022", "v": "2022"}, {"n": "2021", "v": "2021"}, {"n": "2020", "v": "2020"}, {"n": "2019", "v": "2019"}, {"n": "2018", "v": "2018"}, {"n": "2017", "v": "2017"}, {"n": "2016", "v": "2016"}, {"n": "2015", "v": "2015"}]
for c in classes:
filters[c['type_id']] = [
{"key": "area", "name": "地区", "value": area_vals},
{"key": "class", "name": "类型", "value": class_vals},
{"key": "lang", "name": "语言", "value": lang_vals},
{"key": "year", "name": "年份", "value": year_vals},
{"key": "order", "name": "排序", "value": order_vals}
]
return {"class": classes, "filters": filters}
def homeVideoContent(self):
videos = []
try:
html = self._get(self.host)
root = etree.HTML(html)
for item in root.xpath('//div[contains(@class, "module-item")]'):
try:
v = self._parse_list_item(item)
if v:
videos.append(v)
except Exception:
pass
except Exception:
pass
return {"list": videos}
def categoryContent(self, tid, pg, filter, extend):
videos = []
try:
if isinstance(extend, str) and extend:
try: extend = json.loads(extend)
except: extend = {}
elif not extend:
extend = {}
area = extend.get('area', '')
cls = extend.get('class', '')
year = extend.get('year', '')
lang = extend.get('lang', '')
order = extend.get('order', 'time')
if order in ('hits_week', 'hits_month'): order = 'hits'
elif order not in ('time', 'hits', 'score'): order = 'time'
url = self._build_vodshow(tid, area, order, cls, lang, pg, year)
html = self._get(url)
root = etree.HTML(html)
for item in root.xpath('//div[contains(@class, "module-item")]'):
try:
v = self._parse_list_item(item)
if v:
videos.append(v)
except Exception:
pass
# 去重
seen = set()
unique = []
for v in videos:
if v['vod_id'] not in seen:
seen.add(v['vod_id'])
unique.append(v)
# 总页数
pm = 1
for m in re.finditer(r'href="(/vodshow/[^"]*?-(\d+)-[^"]*)"', html):
pm = max(pm, int(m.group(2)))
if pm < 1:
m = re.search(r'第(\d+)页.*尾页', html)
if m: pm = int(m.group(1))
limit = len(unique) if unique else 24
return {'list': unique, 'page': int(pg), 'pagecount': pm, 'limit': limit, 'total': pm * limit}
except Exception:
return {'list': [], 'page': 1, 'pagecount': 0, 'limit': 0, 'total': 0}
def detailContent(self, ids):
try:
vod_id = ids[0]
html = self._get(f"{self.host}/voddetail/{vod_id}.html")
root = etree.HTML(html)
title = root.xpath('//h1[@class="page-title"]/text()')
vod_name = title[0].strip() if title else ''
if not vod_name:
ts = root.xpath('//title/text()')
if ts: vod_name = ts[0].split('-')[0].strip()
pic_a = root.xpath('//div[contains(@class,"video-cover")]//img')
vod_pic = ''
if pic_a:
pic = pic_a[0].get('data-src') or pic_a[0].get('src', '')
if not pic.startswith('data:image'): vod_pic = self._fix_url(pic)
year_a = root.xpath('//a[contains(@href,"/vodsearch/year/")]')
vod_year = ''
if year_a:
m = re.search(r'(\d{4})', year_a[0].text or '')
if m: vod_year = m.group(1)
if not vod_year:
yl = root.xpath('//a[contains(@href,"/vodshow/")]')
for t in yl:
m = re.search(r'^(\d{4})\s*$', (t.text or '').strip())
if m: vod_year = m.group(1); break
area_a = root.xpath('//a[contains(@href,"/vodshow/")]')
area_k = ['中国大陆', '中国香港', '中国台湾', '香港', '台湾', '内地', '韩国', '日本', '美国', '泰国', '英国', '新加坡', '法国']
vod_area = ''
for t in area_a:
txt = (t.text or '').strip()
if re.match(r'^\d{4}$', txt): continue
for k in area_k:
if k in txt:
if k == '香港': vod_area = '中国香港'
elif k == '台湾': vod_area = '中国台湾'
elif k == '内地': vod_area = '中国大陆'
else: vod_area = txt
break
if vod_area: break
def _ef(label):
p = html.find('class="video-info-itemtitle">%s</span>' % label)
if p < 0: return ''
end = html.find('class="video-info-items"', p+1)
if end < 0: end = p + 800
seg = html[p:end]
names = [x.strip() for x in re.findall(r'href="[^"]*/vodsearch/[^"]+"[^>]*>([^<]+)<', seg)]
if not names:
d = re.search(r'class="video-info-item"[^>]*>([^<]*)<', seg)
if d: return d.group(1).strip()
return ' '.join(names)
vod_actor = _ef('主演:')
vod_director = _ef('导演:')
sq = root.xpath('//p[@class="sqjj_a"]')
vod_content = ''
if sq: vod_content = self._parse_text(sq[0])
if not vod_content:
zk = root.xpath('//p[@class="zkjj_a"]')
if zk: vod_content = self._parse_text(zk[0])
vod_content = re.sub(r'\[收起部分\]|\[展开全部\]', '', vod_content)
vod_content = re.sub(r'\s+', '', vod_content).strip()
tabs = re.findall(r'data-dropdown-value="([^"]+)"', html)
sections = re.split(r'\bid="glist-\d+"', html)
froms = []; urls = []
for idx, sec in enumerate(sections):
if idx >= len(tabs): break
name = tabs[idx].strip()
if not name or name == 'http下载': continue
eps = re.findall(r'href="(/vodplay/[^"]+)"[^>]*>(?:<span>)?([^<]*)(?:</span>)?</a>', sec)
if not eps: continue
pl = []
for h, t in eps:
txt = t.strip()
if not txt:
mm = re.search(r'/vodplay/\d+-\d+-(\d+)\.html', h)
txt = '%d' % int(mm.group(1)) if mm else h
pl.append(f"{txt}${self._fix_url(h)}")
if pl:
froms.append(name); urls.append("#".join(pl))
return {'list': [{"vod_id": vod_id, "vod_name": vod_name, "vod_pic": vod_pic,
"vod_year": vod_year, "vod_area": vod_area, "vod_actor": vod_actor,
"vod_director": vod_director, "vod_content": vod_content,
"vod_play_from": "$$$".join(froms) if froms else "默认",
"vod_play_url": "$$$".join(urls) if urls else ""}]}
except Exception:
return {'list': []}
def playerContent(self, flag, id, vipFlags):
try:
html = self._get(id)
m = re.search(r'player_aaaa=({.*?})\s*</script>', html, re.S)
if m:
try:
d = json.loads(m.group(1))
url = d.get('url', '')
if url:
if url.startswith('//'): url = 'https:' + url
return {"parse": 0 if self.isVideoFormat(url) else 1, "playUrl": "", "url": url, "header": json.dumps(self.header)}
except: pass
ifr = re.search(r'<iframe[^>]+src\s*=\s*"([^"]+)"', html)
if ifr:
u = self._fix_url(ifr.group(1))
return {"parse": 0 if self.isVideoFormat(u) else 1, "playUrl": "", "url": u, "header": json.dumps(self.header)}
m3u8 = re.search(r'["\'](https?://[^"\']+\.m3u8[^"\']*)["\']', html)
if m3u8: return {"parse": 0, "playUrl": "", "url": m3u8.group(1), "header": json.dumps(self.header)}
mp4 = re.search(r'["\'](https?://[^"\']+\.(?:mp4|flv|ts))["\']', html)
if mp4: return {"parse": 0, "playUrl": "", "url": mp4.group(1), "header": json.dumps(self.header)}
return {"parse": 0, "playUrl": "", "url": ""}
except Exception:
return {"parse": 0, "playUrl": "", "url": ""}
def searchContent(self, key, quick, pg='1'):
videos = []
try:
html = self._get(f"{self.host}/vodsearch.html", params={"wd": key, "pg": pg})
parts = html.split('<div class="module-search-item">')
for p in parts[1:]:
vm = re.search(r'href="(/voddetail/(\d+)\.html)"', p)
if not vm: continue
nm = re.search(r'<h3><a href="/voddetail/\d+\.html" title="([^"]+)"', p)
pm = re.search(r'data-src="([^"]+)"', p)
rm = re.search(r'video-serial"[^>]*>([^<]+)<', p)
videos.append({"vod_id": vm.group(2),
"vod_name": nm.group(1) if nm else '',
"vod_pic": self._fix_url(pm.group(1)) if pm else '',
"vod_remarks": rm.group(1).strip() if rm else ''})
tm = re.search(r'<strong class="mac_total">(\d+)</strong>', html)
total = int(tm.group(1)) if tm else len(videos)
pc = max(1, (total + (len(videos) or 1) - 1) // (len(videos) or 1))
return {'list': videos, 'page': int(pg), 'pagecount': pc, 'limit': len(videos), 'total': total}
except Exception:
return {'list': [], 'page': 1, 'pagecount': 0, 'limit': 0, 'total': 0}
def isVideoFormat(self, url):
return any(url.lower().endswith(f) for f in ['.m3u8', '.mp4', '.flv', '.ts'])
def manualVideoCheck(self): pass
def localProxy(self, params): return None
def destroy(self): pass