Sync all projects
This commit is contained in:
@@ -246,6 +246,18 @@
|
||||
"name": "🐬麦田影院.py[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
@@ -730,6 +742,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧.py"
|
||||
},
|
||||
{
|
||||
"key": "wddj18",
|
||||
"name": "🐬黄豆短剧(魔改).py|🔞[短剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧(魔改).py"
|
||||
},
|
||||
{
|
||||
"key": "djjh",
|
||||
"name": "🐬短剧聚合.py[短剧]",
|
||||
@@ -921,6 +939,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/123AV.py"
|
||||
},
|
||||
{
|
||||
"key": "18av",
|
||||
"name": "🐬18AV.py|🔞(关梯)[成人]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/18AV.py"
|
||||
},
|
||||
{
|
||||
"key": "Xvideos",
|
||||
"name": "🐬Xvideos.py|🔞[成人]",
|
||||
|
||||
@@ -184,6 +184,18 @@
|
||||
"name": "🐬麦田影院.py[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)[追剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
@@ -329,6 +341,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧.py"
|
||||
},
|
||||
{
|
||||
"key": "wddj18",
|
||||
"name": "🐬黄豆短剧(魔改).py|🔞[短剧]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧(魔改).py"
|
||||
},
|
||||
{
|
||||
"key": "djjh",
|
||||
"name": "🐬短剧聚合.py[短剧]",
|
||||
@@ -509,6 +527,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/123AV.py"
|
||||
},
|
||||
{
|
||||
"key": "18av",
|
||||
"name": "🐬18AV.py|🔞(关梯)[成人]",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/18AV.py"
|
||||
},
|
||||
{
|
||||
"key": "Xvideos",
|
||||
"name": "🐬Xvideos.py|🔞[成人]",
|
||||
|
||||
@@ -233,6 +233,18 @@
|
||||
"name": "🐬麦田影院.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
|
||||
@@ -225,6 +225,18 @@
|
||||
"name": "🐬麦田影院.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
@@ -663,6 +675,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧.py"
|
||||
},
|
||||
{
|
||||
"key": "wddj18",
|
||||
"name": "🐬黄豆短剧(魔改).py|🔞",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧(魔改).py"
|
||||
},
|
||||
{
|
||||
"key": "djjh",
|
||||
"name": "🐬短剧聚合.py",
|
||||
@@ -860,6 +878,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/123AV.py"
|
||||
},
|
||||
{
|
||||
"key": "18av",
|
||||
"name": "🐬18AV.py|🔞(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/18AV.py"
|
||||
},
|
||||
{
|
||||
"key": "Xvideos",
|
||||
"name": "🐬Xvideos.py|🔞",
|
||||
|
||||
@@ -183,6 +183,24 @@
|
||||
"name": "🐬麦田影院.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "jgvy",
|
||||
"name": "🐬追光影视.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/追光影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
@@ -346,6 +364,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧.py"
|
||||
},
|
||||
{
|
||||
"key": "wddj18",
|
||||
"name": "🐬黄豆短剧(魔改).py|🔞",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/黄豆短剧(魔改).py"
|
||||
},
|
||||
{
|
||||
"key": "djjh",
|
||||
"name": "🐬短剧聚合.py",
|
||||
@@ -526,6 +550,12 @@
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/123AV.py"
|
||||
},
|
||||
{
|
||||
"key": "18av",
|
||||
"name": "🐬18AV.py|🔞(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/18AV.py"
|
||||
},
|
||||
{
|
||||
"key": "Xvideos",
|
||||
"name": "🐬Xvideos.py|🔞",
|
||||
|
||||
@@ -219,6 +219,24 @@
|
||||
"name": "🐬麦田影院.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/麦田影院.py"
|
||||
},
|
||||
{
|
||||
"key": "58vy",
|
||||
"name": "🐬58影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/58影视.py"
|
||||
},
|
||||
{
|
||||
"key": "sdvy",
|
||||
"name": "🐬兄弟影视.py(关梯)",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/兄弟影视.py"
|
||||
},
|
||||
{
|
||||
"key": "jgvy",
|
||||
"name": "🐬追光影视.py",
|
||||
"type": 3,
|
||||
"api": "https://ghfast.top/https://raw.githubusercontent.com/FGBLH/HKL/refs/heads/main/py/追光影视.py"
|
||||
},
|
||||
{
|
||||
"key": "xhvm",
|
||||
|
||||
@@ -0,0 +1,724 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
import sys
|
||||
import re
|
||||
import json
|
||||
import base64
|
||||
import threading
|
||||
import requests
|
||||
import urllib3
|
||||
from http.server import HTTPServer, BaseHTTPRequestHandler
|
||||
from socketserver import ThreadingMixIn
|
||||
from urllib.parse import unquote, quote
|
||||
|
||||
urllib3.disable_warnings()
|
||||
sys.path.append('..')
|
||||
from base.spider import Spider
|
||||
|
||||
# ===== 纯 Python AES-128 工具 =====
|
||||
_sbox = bytes([
|
||||
0x63,0x7c,0x77,0x7b,0xf2,0x6b,0x6f,0xc5,0x30,0x01,0x67,0x2b,0xfe,0xd7,0xab,0x76,
|
||||
0xca,0x82,0xc9,0x7d,0xfa,0x59,0x47,0xf0,0xad,0xd4,0xa2,0xaf,0x9c,0xa4,0x72,0xc0,
|
||||
0xb7,0xfd,0x93,0x26,0x36,0x3f,0xf7,0xcc,0x34,0xa5,0xe5,0xf1,0x71,0xd8,0x31,0x15,
|
||||
0x04,0xc7,0x23,0xc3,0x18,0x96,0x05,0x9a,0x07,0x12,0x80,0xe2,0xeb,0x27,0xb2,0x75,
|
||||
0x09,0x83,0x2c,0x1a,0x1b,0x6e,0x5a,0xa0,0x52,0x3b,0xd6,0xb3,0x29,0xe3,0x2f,0x84,
|
||||
0x53,0xd1,0x00,0xed,0x20,0xfc,0xb1,0x5b,0x6a,0xcb,0xbe,0x39,0x4a,0x4c,0x58,0xcf,
|
||||
0xd0,0xef,0xaa,0xfb,0x43,0x4d,0x33,0x85,0x45,0xf9,0x02,0x7f,0x50,0x3c,0x9f,0xa8,
|
||||
0x51,0xa3,0x40,0x8f,0x92,0x9d,0x38,0xf5,0xbc,0xb6,0xda,0x21,0x10,0xff,0xf3,0xd2,
|
||||
0xcd,0x0c,0x13,0xec,0x5f,0x97,0x44,0x17,0xc4,0xa7,0x7e,0x3d,0x64,0x5d,0x19,0x73,
|
||||
0x60,0x81,0x4f,0xdc,0x22,0x2a,0x90,0x88,0x46,0xee,0xb8,0x14,0xde,0x5e,0x0b,0xdb,
|
||||
0xe0,0x32,0x3a,0x0a,0x49,0x06,0x24,0x5c,0xc2,0xd3,0xac,0x62,0x91,0x95,0xe4,0x79,
|
||||
0xe7,0xc8,0x37,0x6d,0x8d,0xd5,0x4e,0xa9,0x6c,0x56,0xf4,0xea,0x65,0x7a,0xae,0x08,
|
||||
0xba,0x78,0x25,0x2e,0x1c,0xa6,0xb4,0xc6,0xe8,0xdd,0x74,0x1f,0x4b,0xbd,0x8b,0x8a,
|
||||
0x70,0x3e,0xb5,0x66,0x48,0x03,0xf6,0x0e,0x61,0x35,0x57,0xb9,0x86,0xc1,0x1d,0x9e,
|
||||
0xe1,0xf8,0x98,0x11,0x69,0xd9,0x8e,0x94,0x9b,0x1e,0x87,0xe9,0xce,0x55,0x28,0xdf,
|
||||
0x8c,0xa1,0x89,0x0d,0xbf,0xe6,0x42,0x68,0x41,0x99,0x2d,0x0f,0xb0,0x54,0xbb,0x16])
|
||||
|
||||
_inv_sbox = bytes([
|
||||
0x52,0x09,0x6a,0xd5,0x30,0x36,0xa5,0x38,0xbf,0x40,0xa3,0x9e,0x81,0xf3,0xd7,0xfb,
|
||||
0x7c,0xe3,0x39,0x82,0x9b,0x2f,0xff,0x87,0x34,0x8e,0x43,0x44,0xc4,0xde,0xe9,0xcb,
|
||||
0x54,0x7b,0x94,0x32,0xa6,0xc2,0x23,0x3d,0xee,0x4c,0x95,0x0b,0x42,0xfa,0xc3,0x4e,
|
||||
0x08,0x2e,0xa1,0x66,0x28,0xd9,0x24,0xb2,0x76,0x5b,0xa2,0x49,0x6d,0x8b,0xd1,0x25,
|
||||
0x72,0xf8,0xf6,0x64,0x86,0x68,0x98,0x16,0xd4,0xa4,0x5c,0xcc,0x5d,0x65,0xb6,0x92,
|
||||
0x6c,0x70,0x48,0x50,0xfd,0xed,0xb9,0xda,0x5e,0x15,0x46,0x57,0xa7,0x8d,0x9d,0x84,
|
||||
0x90,0xd8,0xab,0x00,0x8c,0xbc,0xd3,0x0a,0xf7,0xe4,0x58,0x05,0xb8,0xb3,0x45,0x06,
|
||||
0xd0,0x2c,0x1e,0x8f,0xca,0x3f,0x0f,0x02,0xc1,0xaf,0xbd,0x03,0x01,0x13,0x8a,0x6b,
|
||||
0x3a,0x91,0x11,0x41,0x4f,0x67,0xdc,0xea,0x97,0xf2,0xcf,0xce,0xf0,0xb4,0xe6,0x73,
|
||||
0x96,0xac,0x74,0x22,0xe7,0xad,0x35,0x85,0xe2,0xf9,0x37,0xe8,0x1c,0x75,0xdf,0x6e,
|
||||
0x47,0xf1,0x1a,0x71,0x1d,0x29,0xc5,0x89,0x6f,0xb7,0x62,0x0e,0xaa,0x18,0xbe,0x1b,
|
||||
0xfc,0x56,0x3e,0x4b,0xc6,0xd2,0x79,0x20,0x9a,0xdb,0xc0,0xfe,0x78,0xcd,0x5a,0xf4,
|
||||
0x1f,0xdd,0xa8,0x33,0x88,0x07,0xc7,0x31,0xb1,0x12,0x10,0x59,0x27,0x80,0xec,0x5f,
|
||||
0x60,0x51,0x7f,0xa9,0x19,0xb5,0x4a,0x0d,0x2d,0xe5,0x7a,0x9f,0x93,0xc9,0x9c,0xef,
|
||||
0xa0,0xe0,0x3b,0x4d,0xae,0x2a,0xf5,0xb0,0xc8,0xeb,0xbb,0x3c,0x83,0x53,0x99,0x61,
|
||||
0x17,0x2b,0x04,0x7e,0xba,0x77,0xd6,0x26,0xe1,0x69,0x14,0x63,0x55,0x21,0x0c,0x7d])
|
||||
|
||||
_rcon = [0x01,0x02,0x04,0x08,0x10,0x20,0x40,0x80,0x1b,0x36]
|
||||
|
||||
def _xtime(a):
|
||||
return ((a << 1) ^ 0x1b) & 0xff if a & 0x80 else (a << 1) & 0xff
|
||||
|
||||
def _gf_mul(a, b):
|
||||
r = 0
|
||||
for _ in range(8):
|
||||
if b & 1: r ^= a
|
||||
a = _xtime(a)
|
||||
b >>= 1
|
||||
return r
|
||||
|
||||
_mul_e = bytes(_gf_mul(0x0e, i) for i in range(256))
|
||||
_mul_b = bytes(_gf_mul(0x0b, i) for i in range(256))
|
||||
_mul_d = bytes(_gf_mul(0x0d, i) for i in range(256))
|
||||
_mul_9 = bytes(_gf_mul(0x09, i) for i in range(256))
|
||||
|
||||
_key_schedules = {}
|
||||
|
||||
def _key_schedule(key):
|
||||
k = bytes(key)
|
||||
if k in _key_schedules: return _key_schedules[k]
|
||||
w = []
|
||||
for i in range(4):
|
||||
w.append([key[4*i], key[4*i+1], key[4*i+2], key[4*i+3]])
|
||||
for i in range(4, 44):
|
||||
temp = w[i-1][:]
|
||||
if i % 4 == 0:
|
||||
temp = temp[1:] + temp[:1]
|
||||
temp = [_sbox[b] for b in temp]
|
||||
temp[0] ^= _rcon[i//4 - 1]
|
||||
w.append([w[i-4][j] ^ temp[j] for j in range(4)])
|
||||
_key_schedules[k] = w
|
||||
return w
|
||||
|
||||
def _dec_block(block, w):
|
||||
s0,s1,s2,s3,s4,s5,s6,s7,s8,s9,s10,s11,s12,s13,s14,s15 = block
|
||||
s0 ^= w[40][0]; s1 ^= w[40][1]; s2 ^= w[40][2]; s3 ^= w[40][3]
|
||||
s4 ^= w[41][0]; s5 ^= w[41][1]; s6 ^= w[41][2]; s7 ^= w[41][3]
|
||||
s8 ^= w[42][0]; s9 ^= w[42][1]; s10^= w[42][2]; s11^= w[42][3]
|
||||
s12^= w[43][0]; s13^= w[43][1]; s14^= w[43][2]; s15^= w[43][3]
|
||||
box = _inv_sbox
|
||||
for rnd in range(9, 0, -1):
|
||||
t0=box[s0]; t1=box[s13]; t2=box[s10]; t3=box[s7]
|
||||
t4=box[s4]; t5=box[s1]; t6=box[s14]; t7=box[s11]
|
||||
t8=box[s8]; t9=box[s5]; t10=box[s2]; t11=box[s15]
|
||||
t12=box[s12]; t13=box[s9]; t14=box[s6]; t15=box[s3]
|
||||
rk=w[rnd*4]; t0^=rk[0]; t1^=rk[1]; t2^=rk[2]; t3^=rk[3]
|
||||
rk=w[rnd*4+1]; t4^=rk[0]; t5^=rk[1]; t6^=rk[2]; t7^=rk[3]
|
||||
rk=w[rnd*4+2]; t8^=rk[0]; t9^=rk[1]; t10^=rk[2]; t11^=rk[3]
|
||||
rk=w[rnd*4+3]; t12^=rk[0]; t13^=rk[1]; t14^=rk[2]; t15^=rk[3]
|
||||
s0 =_mul_e[t0]^_mul_b[t1]^_mul_d[t2]^_mul_9[t3]
|
||||
s1 =_mul_9[t0]^_mul_e[t1]^_mul_b[t2]^_mul_d[t3]
|
||||
s2 =_mul_d[t0]^_mul_9[t1]^_mul_e[t2]^_mul_b[t3]
|
||||
s3 =_mul_b[t0]^_mul_d[t1]^_mul_9[t2]^_mul_e[t3]
|
||||
s4 =_mul_e[t4]^_mul_b[t5]^_mul_d[t6]^_mul_9[t7]
|
||||
s5 =_mul_9[t4]^_mul_e[t5]^_mul_b[t6]^_mul_d[t7]
|
||||
s6 =_mul_d[t4]^_mul_9[t5]^_mul_e[t6]^_mul_b[t7]
|
||||
s7 =_mul_b[t4]^_mul_d[t5]^_mul_9[t6]^_mul_e[t7]
|
||||
s8 =_mul_e[t8]^_mul_b[t9]^_mul_d[t10]^_mul_9[t11]
|
||||
s9 =_mul_9[t8]^_mul_e[t9]^_mul_b[t10]^_mul_d[t11]
|
||||
s10=_mul_d[t8]^_mul_9[t9]^_mul_e[t10]^_mul_b[t11]
|
||||
s11=_mul_b[t8]^_mul_d[t9]^_mul_9[t10]^_mul_e[t11]
|
||||
s12=_mul_e[t12]^_mul_b[t13]^_mul_d[t14]^_mul_9[t15]
|
||||
s13=_mul_9[t12]^_mul_e[t13]^_mul_b[t14]^_mul_d[t15]
|
||||
s14=_mul_d[t12]^_mul_9[t13]^_mul_e[t14]^_mul_b[t15]
|
||||
s15=_mul_b[t12]^_mul_d[t13]^_mul_9[t14]^_mul_e[t15]
|
||||
t0=box[s0]; t1=box[s13]; t2=box[s10]; t3=box[s7]
|
||||
t4=box[s4]; t5=box[s1]; t6=box[s14]; t7=box[s11]
|
||||
t8=box[s8]; t9=box[s5]; t10=box[s2]; t11=box[s15]
|
||||
t12=box[s12]; t13=box[s9]; t14=box[s6]; t15=box[s3]
|
||||
rk=w[0]; t0^=rk[0]; t1^=rk[1]; t2^=rk[2]; t3^=rk[3]
|
||||
rk=w[1]; t4^=rk[0]; t5^=rk[1]; t6^=rk[2]; t7^=rk[3]
|
||||
rk=w[2]; t8^=rk[0]; t9^=rk[1]; t10^=rk[2]; t11^=rk[3]
|
||||
rk=w[3]; t12^=rk[0]; t13^=rk[1]; t14^=rk[2]; t15^=rk[3]
|
||||
return bytes([t0,t1,t2,t3,t4,t5,t6,t7,t8,t9,t10,t11,t12,t13,t14,t15])
|
||||
|
||||
def _aes_cbc_decrypt(data, key, iv):
|
||||
if not data or len(data) % 16: return data
|
||||
n = len(data) // 16
|
||||
w = _key_schedule(key)
|
||||
out = bytearray(len(data))
|
||||
prev = iv
|
||||
for i in range(n):
|
||||
block = data[i*16:(i+1)*16]
|
||||
dec = _dec_block(block, w)
|
||||
for j in range(16):
|
||||
out[i*16+j] = dec[j] ^ prev[j]
|
||||
prev = block
|
||||
pad = out[-1]
|
||||
if 1 <= pad <= 16:
|
||||
return bytes(out[:-pad])
|
||||
return bytes(out)
|
||||
|
||||
|
||||
# ===== 全局代理服务(封面图走代理绕过 SSL) =====
|
||||
_proxy_port = 0
|
||||
_proxy_started = False
|
||||
_proxy_session = requests.Session()
|
||||
_proxy_session.verify = False
|
||||
_proxy_headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Referer': 'https://mjv011.com/',
|
||||
}
|
||||
|
||||
class _ThreadedHTTPServer(ThreadingMixIn, HTTPServer):
|
||||
daemon_threads = True
|
||||
|
||||
class _ProxyHandler(BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
try:
|
||||
real_url = unquote(self.path[1:])
|
||||
if not real_url or not real_url.startswith('http'):
|
||||
self.send_response(404); self.end_headers(); return
|
||||
r = _proxy_session.get(real_url, headers=_proxy_headers, timeout=20, verify=False)
|
||||
ct = r.headers.get('Content-Type', 'image/jpeg')
|
||||
self.send_response(200)
|
||||
self.send_header('Content-Type', ct)
|
||||
self.send_header('Content-Length', len(r.content))
|
||||
self.send_header('Access-Control-Allow-Origin', '*')
|
||||
self.end_headers()
|
||||
self.wfile.write(r.content)
|
||||
except BrokenPipeError:
|
||||
pass
|
||||
except Exception:
|
||||
self.send_response(404); self.end_headers()
|
||||
def log_message(self, format, *args): pass
|
||||
|
||||
def _find_free_port():
|
||||
import socket
|
||||
sk = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
sk.bind(('127.0.0.1', 0))
|
||||
port = sk.getsockname()[1]
|
||||
sk.close()
|
||||
return port
|
||||
|
||||
def _start_proxy():
|
||||
global _proxy_port, _proxy_started
|
||||
if _proxy_started: return
|
||||
_proxy_port = _find_free_port()
|
||||
server = _ThreadedHTTPServer(('127.0.0.1', _proxy_port), _ProxyHandler)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
_proxy_started = True
|
||||
|
||||
|
||||
# ===== 内容类型判断 =====
|
||||
def _is_novel(tid_or_vid):
|
||||
if not tid_or_vid: return False
|
||||
return tid_or_vid.startswith('novel')
|
||||
|
||||
def _is_image(tid_or_vid):
|
||||
if not tid_or_vid: return False
|
||||
return any(tid_or_vid.startswith(p) for p in ['18H', 'doujin', 'cg', 'cwp'])
|
||||
|
||||
|
||||
# ===== Spider =====
|
||||
class Spider(Spider):
|
||||
session = requests.Session()
|
||||
host = 'https://mjv011.com'
|
||||
headers = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
|
||||
'Accept-Language': 'zh-CN,zh;q=0.9',
|
||||
'Referer': 'https://mjv011.com/',
|
||||
}
|
||||
|
||||
def getName(self): return "mjv011"
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
if not url: return False
|
||||
return '.m3u8' in url or '.mp4' in url or '.ts' in url
|
||||
|
||||
def manualVideoCheck(self): return False
|
||||
def destroy(self): pass
|
||||
|
||||
def localProxy(self, param):
|
||||
return [404, 'text/plain', '']
|
||||
|
||||
def init(self, extend=""):
|
||||
self.session.verify = False
|
||||
_start_proxy()
|
||||
try:
|
||||
self.session.get(
|
||||
f'{self.host}/zh/chinese_IamOverEighteenYearsOld/19/index.html',
|
||||
headers=self.headers, timeout=15, verify=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _proxy_url(self, url):
|
||||
if not url: return ''
|
||||
if url.startswith('http://127.0.0.1'):
|
||||
return url
|
||||
return f'http://127.0.0.1:{_proxy_port}/{quote(url, safe="")}'
|
||||
|
||||
def _fetch(self, url):
|
||||
try:
|
||||
r = self.session.get(url, headers=self.headers, timeout=20, verify=False)
|
||||
r.encoding = 'utf-8'
|
||||
if r.status_code == 200:
|
||||
text = r.text
|
||||
# 年龄验证页检测:自动绕过
|
||||
if len(text) < 3000 and ('同意(enter)' in text or 'IamOverEighteenYearsOld' in text):
|
||||
self.session.get(
|
||||
f'{self.host}/zh/chinese_IamOverEighteenYearsOld/19/index.html',
|
||||
headers=self.headers, timeout=20, verify=False)
|
||||
r = self.session.get(url, headers=self.headers, timeout=20, verify=False)
|
||||
r.encoding = 'utf-8'
|
||||
if r.status_code == 200:
|
||||
return r.text
|
||||
return ''
|
||||
return text
|
||||
return ''
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
# ===== 列表解析 =====
|
||||
def _parse_text_posts(self, text):
|
||||
"""小说列表(无图)"""
|
||||
items = []
|
||||
for m in re.finditer(r"<div class='post'>\s*<div class='con'>\s*<h3[^>]*><a[^>]*href=\"([^\"]+)\"[^>]*>([^<]+)</a></h3>", text):
|
||||
href, title = m.groups()
|
||||
mm = re.search(r'/([^/]+)_content/(\d+)/([^/]+)\.html', href)
|
||||
if not mm: continue
|
||||
ctype, vid, slug = mm.group(1), mm.group(2), mm.group(3)
|
||||
items.append({
|
||||
'vod_id': f'{ctype}#{vid}#{slug}',
|
||||
'vod_name': title.strip(),
|
||||
'vod_pic': '',
|
||||
'vod_remarks': '',
|
||||
})
|
||||
return items
|
||||
|
||||
def _parse_posts(self, text, tid=''):
|
||||
"""根据分类解析列表:小说单独处理,其他统一按带图 post 处理"""
|
||||
if _is_novel(tid):
|
||||
return self._parse_text_posts(text)
|
||||
|
||||
items = []
|
||||
for m in re.finditer(r"<div class='post'>\s*<a[^>]*href=\"([^\"]+)\"[^>]*><img[^>]*src='([^']+)'[^>]*>\s*</a>\s*<div class='con'>\s*<h3[^>]*><a[^>]*>([^<]+)</a></h3>(?:\s*<div class='meta'>([^<]*)</div>)?", text):
|
||||
href, pic, title, date = m.groups()
|
||||
date = date or ''
|
||||
mm = re.search(r'/([^/]+)_content/(\d+)/([^/]+)\.html', href)
|
||||
if not mm: continue
|
||||
ctype, vid, slug = mm.group(1), mm.group(2), mm.group(3)
|
||||
items.append({
|
||||
'vod_id': f'{ctype}#{vid}#{slug}',
|
||||
'vod_name': title.strip(),
|
||||
'vod_pic': self._proxy_url(pic),
|
||||
'vod_remarks': date.strip(),
|
||||
})
|
||||
return items
|
||||
|
||||
def _build_cat_url(self, tid, page, extend=None):
|
||||
if extend and extend.get('sub'):
|
||||
sub = quote(unquote(extend['sub']), safe='/()')
|
||||
return f'{self.host}/zh/{sub}/{page}.html'
|
||||
if tid.startswith('search_'):
|
||||
kw = tid[7:]
|
||||
return f'{self.host}/zh/chinese_search/all/{kw}/{page}.html'
|
||||
# TVBox 可能截断 type_id,自动补全
|
||||
if tid.endswith('_random'):
|
||||
tid = tid + '/all'
|
||||
elif '/' not in tid:
|
||||
tid = tid + '_random/all'
|
||||
if page == 1:
|
||||
return f'{self.host}/zh/{tid}/index.html'
|
||||
return f'{self.host}/zh/{tid}/index_{page}.html'
|
||||
|
||||
# ===== 接口 =====
|
||||
def homeContent(self, filter):
|
||||
classes = [
|
||||
{'type_id': 'chinese_random/all', 'type_name': '中文字幕'},
|
||||
{'type_id': 'censored_random/all', 'type_name': '有码'},
|
||||
{'type_id': 'uncensored_random/all', 'type_name': '无码'},
|
||||
{'type_id': 'reducing-mosaic_random/all', 'type_name': '无码破解'},
|
||||
{'type_id': 'amateurjav_random/all', 'type_name': '素人'},
|
||||
{'type_id': 'animation_random/all', 'type_name': 'H动画'},
|
||||
{'type_id': 'dt_random/all', 'type_name': '国产自拍'},
|
||||
{'type_id': '18H_random/all', 'type_name': '18H漫画'},
|
||||
{'type_id': 'cg_random/all', 'type_name': '写真图集'},
|
||||
{'type_id': 'cwp_random/all', 'type_name': '国产写真'},
|
||||
{'type_id': 'novel_random/all', 'type_name': '小说'},
|
||||
]
|
||||
|
||||
filter = self._build_filters()
|
||||
return {'class': classes, 'filters': filter, 'type': '影视'}
|
||||
|
||||
def _build_filters(self):
|
||||
"""构造 TVBox filter 格式(网站左侧导航真实子分类)"""
|
||||
filters = {}
|
||||
chinese_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': '随机', 'v': 'chinese_randomall/all'},
|
||||
{'n': '类别清单', 'v': 'chinese_categorylist/list'},
|
||||
]
|
||||
filters['chinese_random/all'] = [{'key': 'sub', 'name': '中文字幕', 'value': chinese_opts}]
|
||||
uncensored_opts: list[dict[str, str]] = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': '一本道(1pondo)', 'v': 'uncensored_makersr/32/一本道(1pondo)'},
|
||||
{'n': 'カリビアンコム(Caribbeancom)', 'v': 'uncensored_makersr/30/カリビアンコム(Caribbeancom)'},
|
||||
{'n': 'カリビアンコムPPV', 'v': 'uncensored_makersr/40/カリビアンコムPPV(Caribbeancompr)'},
|
||||
{'n': '天然むすめ(10musume)', 'v': 'uncensored_makersr/31/天然むすめ(10musume)'},
|
||||
{'n': 'HEYZO', 'v': 'uncensored_makersr/17/HEYZO'},
|
||||
{'n': '東京熱(Tokyo Hot)', 'v': 'uncensored_makersr/29/東京熱(Tokyo Hot)'},
|
||||
{'n': 'ガチん娘!(Gachinco)', 'v': 'uncensored_makersr/35/ガチん娘!(Gachinco)'},
|
||||
{'n': 'パコパコママ(pacopacomama)', 'v': 'uncensored_makersr/36/パコパコママ(pacopacomama)'},
|
||||
{'n': 'エッチな4610', 'v': 'uncensored_makersr/34/エッチな4610'},
|
||||
{'n': '人妻斬り0930', 'v': 'uncensored_makersr/38/人妻斬り0930'},
|
||||
{'n': 'エッチな0930', 'v': 'uncensored_makersr/39/エッチな0930'},
|
||||
{'n': 'トリプルエックス(XXX-AV)', 'v': 'uncensored_makersr/126/トリプルエックス (XXX-AV)'},
|
||||
]
|
||||
filters['uncensored_random/all'] = [{'key': 'sub', 'name': '厂商', 'value': uncensored_opts}]
|
||||
|
||||
animation_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': 'H有码动画', 'v': 'CensoredAnimation_random/all'},
|
||||
{'n': 'H无码动画', 'v': 'UncensoredAnimation_random/all'},
|
||||
{'n': 'H_3D动画', 'v': 'tdAnimation_random/all'},
|
||||
]
|
||||
filters['animation_random/all'] = [{'key': 'sub', 'name': '动漫', 'value': animation_opts}]
|
||||
|
||||
comic_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': '短篇同人', 'v': 'doujin_random/all'},
|
||||
]
|
||||
filters['18H_random/all'] = [{'key': 'sub', 'name': '漫画', 'value': comic_opts}]
|
||||
|
||||
cg_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': 'Bejean On Line', 'v': 'cg_search/all/Bejean On Line'},
|
||||
{'n': 'Bomb.tv', 'v': 'cg_search/all/Bomb.tv'},
|
||||
{'n': 'DGC', 'v': 'cg_search/all/DGC'},
|
||||
{'n': 'Graphis Gals', 'v': 'cg_search/all/Graphis Gals'},
|
||||
{'n': 'Graphis Hatsunugi', 'v': 'cg_search/all/Graphis Hatsunugi'},
|
||||
{'n': 'image.tv', 'v': 'cg_search/all/image.tv'},
|
||||
{'n': 'Sabra.net', 'v': 'cg_search/all/Sabra.net'},
|
||||
{'n': 'S-Cute', 'v': 'cg_search/all/S-Cute'},
|
||||
{'n': 'X-City', 'v': 'cg_search/all/X-City'},
|
||||
{'n': 'YS Web', 'v': 'cg_search/all/YS Web'},
|
||||
]
|
||||
filters['cg_random/all'] = [{'key': 'sub', 'name': '系列', 'value': cg_opts}]
|
||||
|
||||
cwp_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': '3AGirl AAA女郎', 'v': 'cwp_search/all/3AGirl AAA女郎'},
|
||||
{'n': 'ROSI寫真', 'v': 'cwp_search/all/ROSI寫真'},
|
||||
{'n': 'RU1MM 如壹寫真', 'v': 'cwp_search/all/RU1MM 如壹寫真'},
|
||||
{'n': 'DISI第四印象', 'v': 'cwp_search/all/DISI第四印象'},
|
||||
]
|
||||
filters['cwp_random/all'] = [{'key': 'sub', 'name': '系列', 'value': cwp_opts}]
|
||||
|
||||
novel_opts = [
|
||||
{'n': '全部', 'v': ''},
|
||||
{'n': '學生校園', 'v': 'novel_search/all/學生校園'},
|
||||
{'n': '職場激情', 'v': 'novel_search/all/職場激情'},
|
||||
{'n': '經驗故事', 'v': 'novel_search/all/經驗故事'},
|
||||
{'n': '暴力虐待', 'v': 'novel_search/all/暴力虐待'},
|
||||
{'n': '不倫戀情', 'v': 'novel_search/all/不倫戀情'},
|
||||
{'n': '群體換伴', 'v': 'novel_search/all/群體換伴'},
|
||||
{'n': '人妻熟女', 'v': 'novel_search/all/人妻熟女'},
|
||||
{'n': '科學幻想', 'v': 'novel_search/all/科學幻想'},
|
||||
{'n': '其他故事', 'v': 'novel_search/all/其他故事'},
|
||||
{'n': '玄幻仙俠', 'v': 'novel_search/all/玄幻仙俠'},
|
||||
{'n': '動漫修改', 'v': 'novel_search/all/動漫修改'},
|
||||
{'n': '長篇連載', 'v': 'novel_search/all/長篇連載'},
|
||||
]
|
||||
filters['novel_random/all'] = [{'key': 'sub', 'name': '主题', 'value': novel_opts}]
|
||||
return filters
|
||||
|
||||
def homeVideoContent(self):
|
||||
text = self._fetch(f'{self.host}/zh/chinese_IamOverEighteenYearsOld/19/index.html')
|
||||
items = self._parse_posts(text, 'content_news/all')
|
||||
return {'list': items}
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
try:
|
||||
return self._categoryContent_inner(tid, pg, filter, extend)
|
||||
except Exception:
|
||||
return {'list': [], 'page': int(pg) if pg else 1, 'pagecount': 1, 'limit': 0, 'total': 0}
|
||||
|
||||
def _categoryContent_inner(self, tid, pg, filter, extend):
|
||||
if isinstance(extend, str):
|
||||
try:
|
||||
extend = json.loads(extend)
|
||||
except Exception:
|
||||
extend = {}
|
||||
if not extend:
|
||||
extend = {}
|
||||
page = int(pg) if pg else 1
|
||||
# ===== 类别清单(中文字幕的子分类): 文件夹形式(仿 18av) =====
|
||||
if extend.get('sub') == 'chinese_categorylist/list':
|
||||
return self._category_folder(page)
|
||||
if '@' in str(tid):
|
||||
return self._folder_detail(tid, page)
|
||||
url = self._build_cat_url(tid, page, extend)
|
||||
# 子分类筛选时按子分类内容类型解析
|
||||
parse_tid = tid
|
||||
if parse_tid.endswith('_random'):
|
||||
parse_tid = parse_tid + '/all'
|
||||
elif '/' not in parse_tid:
|
||||
parse_tid = parse_tid + '_random/all'
|
||||
if extend and extend.get('sub'):
|
||||
sub = extend['sub']
|
||||
if _is_novel(sub):
|
||||
parse_tid = 'novel_random/all'
|
||||
elif _is_image(sub):
|
||||
parse_tid = sub.split('_')[0] + '_random/all' if '_' in sub else 'cg_random/all'
|
||||
text = self._fetch(url)
|
||||
items = self._parse_posts(text, parse_tid)
|
||||
return {'list': items, 'page': page, 'pagecount': page + 1,
|
||||
'limit': len(items), 'total': page * len(items) + 1}
|
||||
|
||||
# ===== 类别清单(文件夹形式,仿 18av) =====
|
||||
def _category_folder(self, pg):
|
||||
"""类别清单: 抓取子分类索引页,以 folder 形式返回,点文件夹进入视频列表"""
|
||||
html = self._fetch(f'{self.host}/zh/chinese_categorylist/list/index.html')
|
||||
lst = []
|
||||
for m in re.finditer(
|
||||
r"<a[^>]+href=[\"']([^\"']*chinese_category/(\d+)/([^\"'/]+)/[^\"']*)[\"'][^>]*>([^<]+)</a>",
|
||||
html, re.I):
|
||||
category_url = m.group(1).strip()
|
||||
category_name = m.group(4).strip()
|
||||
if category_url.startswith(self.host):
|
||||
category_url = category_url[len(self.host):]
|
||||
lst.append({
|
||||
'vod_id': category_url + '@',
|
||||
'vod_name': category_name,
|
||||
'vod_pic': self.host.rstrip('/') + '/images/1v.jpg',
|
||||
'vod_tag': 'folder',
|
||||
'vod_remarks': '分类',
|
||||
})
|
||||
return {'list': lst, 'page': 1, 'pagecount': 1,
|
||||
'limit': len(lst), 'total': len(lst)}
|
||||
|
||||
def _folder_detail(self, tid, pg):
|
||||
"""@ 文件夹: 去掉 @, 把子分类链接当作真实 URL 拉取视频列表(支持翻页)"""
|
||||
tid = str(tid).replace('@', '')
|
||||
url = self.host.rstrip('/') + tid
|
||||
if re.search(r'/\d+\.html$', url):
|
||||
url = re.sub(r'/\d+\.html$', '/' + str(pg) + '.html', url)
|
||||
else:
|
||||
base = url.rstrip('/')
|
||||
url = base + ('/index.html' if pg <= 1 else f'/index_{pg}.html')
|
||||
html = self._fetch(url)
|
||||
lst = self._parse_posts(html, 'chinese_category')
|
||||
return {'list': lst, 'page': pg, 'pagecount': 9999,
|
||||
'limit': len(lst), 'total': 9999}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
return self._detailContent_inner(ids)
|
||||
except Exception:
|
||||
return {'list': []}
|
||||
|
||||
def _detailContent_inner(self, ids):
|
||||
vid = str(ids[0] if isinstance(ids, list) else ids)
|
||||
ctype, num, slug = vid.split('#', 2)
|
||||
if _is_novel(ctype):
|
||||
return self._novel_detail(vid, ctype, num, slug)
|
||||
elif _is_image(ctype):
|
||||
return self._image_detail(vid, ctype, num, slug)
|
||||
else:
|
||||
return self._video_detail(vid, ctype, num, slug)
|
||||
|
||||
def _video_detail(self, vid, ctype, num, slug):
|
||||
"""视频详情(有播放器解密)"""
|
||||
url = f'{self.host}/zh/{ctype}_content/{num}/{slug}.html'
|
||||
text = self._fetch(url)
|
||||
if not text: return {'list': []}
|
||||
|
||||
title = ''
|
||||
m = re.search(r'<h1[^>]*>(.*?)</h1>', text, re.S)
|
||||
if m: title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
|
||||
if not title:
|
||||
m = re.search(r'<title>([^<]+)</title>', text)
|
||||
if m: title = m.group(1).strip()
|
||||
|
||||
cover = ''
|
||||
m = re.search(r'"thumbnailUrl"\s*:\s*"([^"]+)"', text)
|
||||
if m: cover = m.group(1)
|
||||
if not cover:
|
||||
m = re.search(r"<meta[^>]*property=\"og:image\"[^>]*content=\"([^\"]+)\"", text)
|
||||
if m: cover = m.group(1)
|
||||
if not cover:
|
||||
m = re.search(r"<div class='post'>.*?<img[^>]*src='([^']+)'", text, re.S)
|
||||
if m: cover = m.group(1)
|
||||
|
||||
m = re.search(r'hadeedg252=(\d+)', text)
|
||||
if not m: return {'list': []}
|
||||
hadeedg252 = int(m.group(1))
|
||||
m = re.search(r'hcdeedg252=(\d+)', text)
|
||||
if not m: return {'list': []}
|
||||
hcdeedg252 = int(m.group(1))
|
||||
m = re.search(r"var argdeqweqweqwe = '([^']+)'", text)
|
||||
if not m: return {'list': []}
|
||||
aes_key = m.group(1)
|
||||
m = re.search(r"var hdddedg252 = '([^']+)'", text)
|
||||
if not m: return {'list': []}
|
||||
aes_iv = m.group(1)
|
||||
|
||||
mm = re.search(r"mvarr\['10_1'\]=(\[.*?\]);", text, re.S)
|
||||
if not mm: return {'list': []}
|
||||
mvarr_str = mm.group(1)
|
||||
items = re.findall(r"\['([^']*)','([^']*)','([^']*)','([^']*)','([^']*)','([^']*)'\]", mvarr_str)
|
||||
if not items: return {'list': []}
|
||||
|
||||
urls = []
|
||||
for iframe_id, enc, html, prefix, empty, label in items:
|
||||
if not enc or not prefix: continue
|
||||
pid = self._decrypt_id(enc, hadeedg252, hcdeedg252, aes_key, aes_iv)
|
||||
if not pid: continue
|
||||
for res, label_name in [('1080', '1080P'), ('720', '720P'), ('480', '480P')]:
|
||||
urls.append(f'{label_name}${num}|{slug}|{pid}|{res}')
|
||||
|
||||
if not urls: return {'list': []}
|
||||
|
||||
vod = {
|
||||
'vod_id': vid,
|
||||
'vod_name': title,
|
||||
'vod_pic': self._proxy_url(cover),
|
||||
'vod_content': '',
|
||||
'vod_remarks': '',
|
||||
'vod_play_from': 'mjv011',
|
||||
'vod_play_url': '#'.join(urls),
|
||||
}
|
||||
return {'list': [vod]}
|
||||
|
||||
def _novel_detail(self, vid, ctype, num, slug):
|
||||
"""小说详情"""
|
||||
url = f'{self.host}/zh/{ctype}_content/{num}/{slug}.html'
|
||||
text = self._fetch(url)
|
||||
if not text: return {'list': []}
|
||||
|
||||
title = ''
|
||||
m = re.search(r'<h1[^>]*>(.*?)</h1>', text, re.S)
|
||||
if m: title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
|
||||
|
||||
content = ''
|
||||
m = re.search(r"id=['\"]novel_content_txtsize['\"][^>]*>(.*?)</div>", text, re.S)
|
||||
if m:
|
||||
raw = m.group(1)
|
||||
content = re.sub(r'<[^>]+>', '', raw)
|
||||
content = re.sub(r' ', ' ', content)
|
||||
content = re.sub(r'\s+', ' ', content).strip()
|
||||
if not content:
|
||||
m = re.search(r"<span class='content_18h_wpcg'>([\s\S]*?)</span>\s*<div class='contents'", text)
|
||||
if m:
|
||||
raw = m.group(1)
|
||||
content = re.sub(r'<[^>]+>', '', raw)
|
||||
content = re.sub(r' ', ' ', content)
|
||||
content = re.sub(r'\s+', ' ', content).strip()
|
||||
|
||||
if len(content) > 3000:
|
||||
content = content[:3000] + '...'
|
||||
|
||||
novel_json = json.dumps({'title': title, 'content': content}, ensure_ascii=False)
|
||||
play_url = f'阅读$novel://{novel_json}'
|
||||
|
||||
vod = {
|
||||
'vod_id': vid,
|
||||
'vod_name': title,
|
||||
'vod_pic': '',
|
||||
'vod_content': '',
|
||||
'vod_remarks': '',
|
||||
'vod_play_from': '小说',
|
||||
'vod_play_url': play_url,
|
||||
'vod_tag': 'text',
|
||||
'vod_player': '书',
|
||||
}
|
||||
return {'list': [vod]}
|
||||
|
||||
def _image_detail(self, vid, ctype, num, slug):
|
||||
"""图片详情(写真/漫画)"""
|
||||
url = f'{self.host}/zh/{ctype}_content/{num}/{slug}.html'
|
||||
text = self._fetch(url)
|
||||
if not text: return {'list': []}
|
||||
|
||||
title = ''
|
||||
m = re.search(r'<h1[^>]*>(.*?)</h1>', text, re.S)
|
||||
if m: title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
|
||||
|
||||
cover = ''
|
||||
m = re.search(r'"thumbnailUrl"\s*:\s*"([^"]+)"', text)
|
||||
if m: cover = m.group(1)
|
||||
|
||||
# 提取详情页所有大图(第一页已包含全部)
|
||||
all_imgs = re.findall(r"src=['\"]([^'\"]*eemmhh02\.com/[^'\"]+\.(?:jpg|png|webp))['\"]", text, re.I)
|
||||
big_imgs = []
|
||||
seen = set()
|
||||
for img in all_imgs:
|
||||
if img in seen:
|
||||
continue
|
||||
seen.add(img)
|
||||
big_imgs.append(self._proxy_url(img))
|
||||
|
||||
if not big_imgs:
|
||||
return {'list': []}
|
||||
|
||||
pics = '&&'.join(big_imgs)
|
||||
play_url = f'查看$pics://{pics}'
|
||||
|
||||
vod = {
|
||||
'vod_id': vid,
|
||||
'vod_name': title,
|
||||
'vod_pic': self._proxy_url(cover),
|
||||
'vod_content': f'共 {len(big_imgs)} 张图片',
|
||||
'vod_remarks': str(len(big_imgs)) + 'P',
|
||||
'vod_play_from': '图片',
|
||||
'vod_play_url': play_url,
|
||||
'vod_tag': 'image',
|
||||
'vod_player': '画',
|
||||
}
|
||||
return {'list': [vod]}
|
||||
|
||||
def _decrypt_id(self, enc, xor_key, base, aes_key, aes_iv):
|
||||
try:
|
||||
sep = chr(base + 97)
|
||||
parts = enc.split(sep)
|
||||
s1 = ''.join(chr(int(p, base) ^ xor_key) for p in parts if p)
|
||||
data = base64.b64decode(s1)
|
||||
plain = _aes_cbc_decrypt(data, aes_key.encode(), aes_iv.encode())
|
||||
return plain.decode('utf-8')
|
||||
except Exception:
|
||||
return ''
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
try:
|
||||
return self._searchContent_inner(key, quick, pg)
|
||||
except Exception:
|
||||
return {'list': [], 'page': int(pg) if pg else 1, 'pagecount': 1, 'limit': 0, 'total': 0}
|
||||
|
||||
def _searchContent_inner(self, key, quick, pg="1"):
|
||||
page = int(pg) if pg else 1
|
||||
return self._categoryContent_inner(f'search_{key}', page, False, {})
|
||||
|
||||
def playerContent(self, flag, id, vipFlags=None):
|
||||
try:
|
||||
return self._playerContent_inner(flag, id, vipFlags)
|
||||
except Exception:
|
||||
return {'parse': 0, 'url': '', 'header': {}, 'position': '0'}
|
||||
|
||||
def _playerContent_inner(self, flag, id, vipFlags=None):
|
||||
# 小说
|
||||
if id.startswith('novel://'):
|
||||
return {'parse': 0, 'url': id, 'header': '', 'vod_player': '书'}
|
||||
# 图片
|
||||
if id.startswith('pics://'):
|
||||
return {'parse': 0, 'playUrl': '', 'url': id, 'header': self.headers}
|
||||
# 视频播放
|
||||
num, slug, pid, res = id.split('|', 3)
|
||||
url = f'{self.host}/js/player/play.php?numresolution={res}&lo=on&id={pid}'
|
||||
text = self._fetch(url)
|
||||
m3u8 = ''
|
||||
if text:
|
||||
mm = re.search(r'videoSources\s*=\s*(\[.*?\]);', text, re.S)
|
||||
if mm:
|
||||
arr = mm.group(1)
|
||||
sources = re.findall(r"src:\s*'([^']+)'[^}]*?size:\s*(\d+)", arr, re.S)
|
||||
if sources:
|
||||
target = int(res) if str(res).isdigit() else 0
|
||||
for u, s in sources:
|
||||
if int(s) == target:
|
||||
m3u8 = u
|
||||
break
|
||||
if not m3u8:
|
||||
m3u8 = sources[0][0]
|
||||
if not m3u8:
|
||||
mm = re.search(r'https?://[^\s"<>\']+?\.m3u8', text)
|
||||
if mm: m3u8 = mm.group(0)
|
||||
return {'parse': 0, 'url': m3u8, 'header': {'Referer': self.host + '/'}, 'position': '0'}
|
||||
@@ -0,0 +1,385 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
import sys
|
||||
import re
|
||||
import json
|
||||
from urllib.parse import urljoin, quote, unquote, urlparse
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
|
||||
sys.path.append('..')
|
||||
try:
|
||||
from base.spider import Spider
|
||||
except ImportError:
|
||||
class Spider:
|
||||
def fetch(self, url, headers=None, **kw):
|
||||
import requests as rq
|
||||
kw.pop('timeout', None)
|
||||
r = rq.get(url, headers=headers, timeout=15, **kw)
|
||||
r.encoding = 'utf-8'
|
||||
return r
|
||||
|
||||
HOST = "https://www.58hu.com"
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
|
||||
# 封面URL中这些域名已失效,对应影片会被过滤掉
|
||||
DEAD_IMG_HOSTS = {"image.caiji.cyou", "wim.xrc888.com"}
|
||||
|
||||
CATEGORIES = {
|
||||
"1": "电影", "2": "电视剧", "3": "综艺",
|
||||
"4": "动漫", "21": "体育",
|
||||
}
|
||||
|
||||
# 子分类: {父分类ID: {子分类ID: 名称}}
|
||||
SUB_CATS = {
|
||||
"1": {"6": "动作片", "7": "喜剧片", "8": "爱情片", "9": "科幻片", "10": "恐怖片",
|
||||
"11": "剧情片", "12": "战争片", "22": "纪录片"},
|
||||
"2": {"13": "国产剧", "14": "港台剧", "15": "日韩剧", "16": "欧美剧", "24": "海外剧",
|
||||
"29": "BI番剧", "30": "BI国创", "31": "BI电影"},
|
||||
"3": {"28": "最新综艺"},
|
||||
"4": {"27": "最新动漫"},
|
||||
"21": {"26": "体育赛事"},
|
||||
}
|
||||
|
||||
class Spider(Spider):
|
||||
def init(self, extend=""):
|
||||
global HOST
|
||||
try:
|
||||
r = self.fetch(HOST, headers={"User-Agent": UA}, timeout=15000)
|
||||
if hasattr(r, 'url') and r.url and r.url != HOST.rstrip("/"):
|
||||
HOST = r.url.rstrip("/")
|
||||
except:
|
||||
pass
|
||||
|
||||
def homeContent(self, filter=False):
|
||||
r = {"class": [], "list": []}
|
||||
for k, v in CATEGORIES.items():
|
||||
r["class"].append({"type_id": k, "type_name": v})
|
||||
return r
|
||||
|
||||
def homeVideoContent(self):
|
||||
try:
|
||||
r = self.fetch(HOST, headers={"User-Agent": UA}, timeout=15000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
return {"list": self._items(html)}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def categoryContent(self, tid, pg=1, filter=False, extend=""):
|
||||
pn = 1
|
||||
try: pn = max(int(str(pg)), 1)
|
||||
except: pass
|
||||
cat = str(tid)
|
||||
try:
|
||||
url = self._build_category_url(cat, pn, extend)
|
||||
r = self.fetch(url, headers={"User-Agent": UA}, timeout=30000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
cat_name = CATEGORIES.get(cat, "")
|
||||
items = self._items(html, cat_filter=cat_name)
|
||||
# 过滤掉没有封面或没有可播放线路的影片(并发检查)
|
||||
items = self._filter_items(items)
|
||||
pc = self._pagecount(html, pn)
|
||||
return {"page": pn, "pagecount": pc, "limit": 30, "total": len(items), "list": items}
|
||||
except:
|
||||
return {"page": pn, "pagecount": 1, "limit": 30, "total": 0, "list": []}
|
||||
|
||||
def _build_category_url(self, cat, pn, extend=""):
|
||||
"""构建分类URL,支持筛选参数"""
|
||||
# 解析extend中的筛选条件
|
||||
ext = {}
|
||||
if extend and isinstance(extend, str) and extend.strip():
|
||||
try:
|
||||
ext = json.loads(extend)
|
||||
except:
|
||||
pass
|
||||
|
||||
# 判断是否为子分类(在SUB_CATS中)
|
||||
is_sub = False
|
||||
for parent, subs in SUB_CATS.items():
|
||||
if cat in subs:
|
||||
is_sub = True
|
||||
break
|
||||
|
||||
# 所有分类用 vod/show 模式
|
||||
parts = []
|
||||
# class/类型
|
||||
cls = ext.get("class") or ""
|
||||
if cls and cls != "全部":
|
||||
parts.append(f"class/{quote(cls)}")
|
||||
# area/地区
|
||||
area = ext.get("area") or ""
|
||||
if area and area != "全部":
|
||||
parts.append(f"area/{quote(area)}")
|
||||
# year/年代
|
||||
year = ext.get("year") or ""
|
||||
if year and year != "全部":
|
||||
parts.append(f"year/{quote(year)}")
|
||||
# letter/字母
|
||||
letter = ext.get("letter") or ""
|
||||
if letter and letter != "全部":
|
||||
parts.append(f"letter/{quote(letter)}")
|
||||
# sort/排序
|
||||
sort = ext.get("sort") or ""
|
||||
if sort:
|
||||
parts.append(f"by/{quote(sort)}")
|
||||
# page
|
||||
if pn > 1:
|
||||
parts.append(f"page/{pn}")
|
||||
filters = "/".join(parts)
|
||||
if filters:
|
||||
return f"{HOST}/index.php/vod/show/{filters}/id/{cat}.html"
|
||||
else:
|
||||
return f"{HOST}/index.php/vod/show/id/{cat}.html"
|
||||
|
||||
def detailContent(self, ids):
|
||||
if isinstance(ids, list):
|
||||
vid = ids[0] if ids else ""
|
||||
else:
|
||||
vid = str(ids) if ids else ""
|
||||
m = re.search(r'(\d+)', str(vid))
|
||||
vid = m.group(1) if m else ""
|
||||
if not vid:
|
||||
return {"list": []}
|
||||
try:
|
||||
r = self.fetch(f"{HOST}/index.php/vod/detail/id/{vid}.html", headers={"User-Agent": UA}, timeout=30000)
|
||||
h = r.text if hasattr(r, 'text') else str(r)
|
||||
except:
|
||||
return {"list": []}
|
||||
d = {"vod_id": vid, "vod_name": "", "vod_pic": "", "vod_year": "",
|
||||
"vod_area": "", "vod_class": "", "vod_director": "", "vod_actor": "",
|
||||
"vod_content": "", "vod_remarks": "", "vod_play_from": "", "vod_play_url": ""}
|
||||
# 标题
|
||||
tn = re.search(r'<h1[^>]*>(.*?)</h1>', h)
|
||||
if tn:
|
||||
d["vod_name"] = re.sub(r'<[^>]+>', '', tn.group(1)).strip()
|
||||
if not d["vod_name"]:
|
||||
tn = re.search(r'<title>(.*?)</title>', h)
|
||||
if tn:
|
||||
d["vod_name"] = tn.group(1).split("-")[0].strip()
|
||||
# 封面: 优先 class="pic-img video-pic" 里的img
|
||||
p = re.search(r'class="pic-img video-pic"[\s\S]{0,100}?<img[^>]*src="(https?://[^"]+)"', h, re.I)
|
||||
if not p:
|
||||
p = re.search(r'data-src="(https?://[^"]+)"', h)
|
||||
if not p:
|
||||
p = re.search(r'src="(https?://[^"]+\.(?:jpg|jpeg|png|webp))"', h, re.I)
|
||||
if p:
|
||||
pic_url = p.group(1)
|
||||
if pic_url.startswith("http://"):
|
||||
pic_url = pic_url.replace("http://", "https://", 1)
|
||||
d["vod_pic"] = pic_url
|
||||
# 简介
|
||||
desc_m = re.search(r'简介[\s\S]{0,30}?>([\s\S]*?)</div>', h)
|
||||
if desc_m:
|
||||
d["vod_content"] = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', desc_m.group(1))).strip()[:500]
|
||||
if not d["vod_content"]:
|
||||
desc_m = re.search(r'class="[^"]*desc[^"]*"[^>]*>([\s\S]*?)</div>', h)
|
||||
if desc_m:
|
||||
d["vod_content"] = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', desc_m.group(1))).strip()[:500]
|
||||
# 年份
|
||||
ym = re.search(r'(\d{4})', d.get("vod_name", ""))
|
||||
if ym: d["vod_year"] = ym.group(1)
|
||||
# 地区/分类/导演/主演等
|
||||
info_items = re.findall(r'<li[^>]*class="[^"]*data[^"]*"[^>]*>([\s\S]*?)</li>', h)
|
||||
for item in info_items:
|
||||
clean = re.sub(r'<[^>]+>', '', item).strip()
|
||||
if '导演' in item and not d["vod_director"]:
|
||||
d["vod_director"] = clean.replace('导演', '').strip().rstrip(',').strip()
|
||||
elif '主演' in item and not d["vod_actor"]:
|
||||
d["vod_actor"] = clean.replace('主演', '').strip().rstrip(',').strip()
|
||||
elif '类型' in item and not d["vod_class"]:
|
||||
d["vod_class"] = clean.replace('类型', '').strip()
|
||||
elif '地区' in item and not d["vod_area"]:
|
||||
d["vod_area"] = clean.replace('地区', '').strip()
|
||||
# 备注
|
||||
rm = re.search(r'class="[^"]*remarks?[^"]*"[^>]*>([^<]+)<', h)
|
||||
if rm: d["vod_remarks"] = rm.group(1).strip()
|
||||
# 播放源(只保留m3u8直链线路)
|
||||
try:
|
||||
pf, pu = [], []
|
||||
line_map = {}
|
||||
for sid, name in re.findall(r'id="#con_playlist_(\d+)"[^>]*class="gico[^"]*"[^>]*>([^<]+)</a>', h):
|
||||
line_map[sid] = name.strip()
|
||||
for m in re.finditer(r'<ul[^>]*id="con_playlist_(\d+)"[^>]*>([\s\S]*?)</ul>', h):
|
||||
sid = m.group(1)
|
||||
content = m.group(2)
|
||||
eps = re.findall(r'href="(/index\.php/vod/play/id/\d+/sid/\d+/nid/\d+\.html)"[^>]*>([\s\S]*?)</a>', content)
|
||||
if not eps:
|
||||
continue
|
||||
line_name = line_map.get(sid, f"线路{sid}")
|
||||
# 预检第1集:只保留返回m3u8直链的线路
|
||||
first_url = urljoin(HOST, eps[0][0])
|
||||
try:
|
||||
rp = self.fetch(first_url, headers={"User-Agent": UA}, timeout=10000)
|
||||
hp = rp.text if hasattr(rp, 'text') else str(rp)
|
||||
pd = re.search(r'player_data\s*=\s*(\{[\s\S]*?\})\s*[;<]', hp)
|
||||
if not pd:
|
||||
continue
|
||||
pdata = json.loads(pd.group(1))
|
||||
purl = pdata.get("url", "")
|
||||
if not purl or not purl.startswith("http") or ".m3u8" not in purl:
|
||||
continue
|
||||
except:
|
||||
continue
|
||||
ep_list = []
|
||||
for url, name in eps:
|
||||
clean_name = re.sub(r'<[^>]+>', '', name).strip()
|
||||
ep_list.append(f"{clean_name}${urljoin(HOST, url)}")
|
||||
if ep_list:
|
||||
pf.append(line_name)
|
||||
pu.append("#".join(ep_list))
|
||||
if pf:
|
||||
d["vod_play_from"] = "$$$".join(pf)
|
||||
d["vod_play_url"] = "$$$".join(pu)
|
||||
except:
|
||||
pass
|
||||
return {"list": [d]}
|
||||
|
||||
def searchContent(self, key, quick=False, pg="1"):
|
||||
try:
|
||||
pn = 1
|
||||
try: pn = int(str(pg))
|
||||
except: pass
|
||||
url = f"{HOST}/index.php/vod/search/wd/{quote(key)}"
|
||||
if pn > 1:
|
||||
url += f"/page/{pn}"
|
||||
url += ".html"
|
||||
r = self.fetch(url, headers={"User-Agent": UA}, timeout=30000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
items = self._items(html)
|
||||
return {"list": items, "page": pn}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags=None):
|
||||
url = str(id) if id else str(flag)
|
||||
if url.startswith("http") and ".m3u8" in url:
|
||||
return {"url": url}
|
||||
if url.startswith("http"):
|
||||
full_url = url
|
||||
else:
|
||||
if not url.startswith("/"):
|
||||
url = "/" + url
|
||||
full_url = urljoin(HOST, url)
|
||||
try:
|
||||
r = self.fetch(full_url, headers={"User-Agent": UA}, timeout=30000)
|
||||
h = r.text if hasattr(r, 'text') else str(r)
|
||||
except:
|
||||
return {"url": ""}
|
||||
pd = re.search(r'player_data\s*=\s*(\{[\s\S]*?\})\s*[;<]', h)
|
||||
if pd:
|
||||
try:
|
||||
data = json.loads(pd.group(1))
|
||||
play_url = data.get("url", "")
|
||||
if play_url:
|
||||
return {"url": play_url}
|
||||
except:
|
||||
pass
|
||||
m3u8 = re.search(r'(https?://[^\s"\'<>]+\.m3u8)', h)
|
||||
if m3u8:
|
||||
return {"url": m3u8.group(1)}
|
||||
return {"url": ""}
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
|
||||
def _filter_items(self, items):
|
||||
"""过滤:去掉没封面的 + 并发检查线路,去掉没有LZ/YZ的"""
|
||||
if not items:
|
||||
return items
|
||||
# 先去掉没封面的
|
||||
items = [it for it in items if it.get("vod_pic")]
|
||||
if not items:
|
||||
return items
|
||||
# 并发检查每个影片是否有LZ/YZ线路
|
||||
playable_ids = set()
|
||||
def check(it):
|
||||
return it["vod_id"], self._check_playable(it["vod_id"])
|
||||
with ThreadPoolExecutor(max_workers=10) as executor:
|
||||
futures = {executor.submit(check, it): it for it in items}
|
||||
for future in as_completed(futures):
|
||||
try:
|
||||
vid, ok = future.result()
|
||||
if ok:
|
||||
playable_ids.add(vid)
|
||||
except:
|
||||
pass
|
||||
return [it for it in items if it["vod_id"] in playable_ids]
|
||||
|
||||
def _check_playable(self, vid):
|
||||
"""检查详情页是否有LZ或YZ线路名"""
|
||||
try:
|
||||
r = self.fetch(f"{HOST}/index.php/vod/detail/id/{vid}.html", headers={"User-Agent": UA}, timeout=5000)
|
||||
h = r.text if hasattr(r, 'text') else str(r)
|
||||
lines = re.findall(r'id="#con_playlist_\d+"[^>]*class="gico[^"]*"[^>]*>([^<]+)</a>', h)
|
||||
for name in lines:
|
||||
if "LZ" in name or "YZ" in name:
|
||||
return True
|
||||
return False
|
||||
except:
|
||||
return False
|
||||
|
||||
def _pagecount(self, html, current_page=1):
|
||||
# 匹配分页链接: /index.php/vod/show/.../page/2/...
|
||||
pages = re.findall(r'/page/(\d+)/', html)
|
||||
max_page = current_page
|
||||
for p in pages:
|
||||
try:
|
||||
n = int(p)
|
||||
if n > max_page:
|
||||
max_page = n
|
||||
except:
|
||||
pass
|
||||
# 如果当前页是最大且还有下一页
|
||||
has_next = re.search(r'>下一页<', html)
|
||||
if has_next and max_page <= current_page + 5:
|
||||
max_page = current_page + 5
|
||||
return max_page
|
||||
|
||||
def _items(self, html, cat_filter=""):
|
||||
items, seen = [], set()
|
||||
# 分类过滤关键词映射
|
||||
ANIME_KEYWORDS = {'动漫', '动画', '卡通', '番剧', '番'}
|
||||
TV_KEYWORDS = {'电视剧', '国产剧', '港台剧', '日韩剧', '欧美剧', '海外剧'}
|
||||
MOVIE_KEYWORDS = {'电影', '动作片', '喜剧片', '爱情片', '科幻片', '恐怖片', '剧情片', '战争片', '纪录片'}
|
||||
for m in re.finditer(r'href="(/index\.php/vod/detail/id/(\d+)\.html)"[^>]*title="([^"]*)"', html):
|
||||
vid = m.group(2)
|
||||
if vid in seen:
|
||||
continue
|
||||
name = m.group(3).strip()
|
||||
if not name or len(name) > 100:
|
||||
continue
|
||||
after = html[m.end():m.end()+2000]
|
||||
# 封面: 优先 data-original,再 data-src,再 src。HTTP升级为HTTPS
|
||||
cover = re.search(r'data-original="(https?://[^"]+)"', after)
|
||||
if not cover:
|
||||
cover = re.search(r'data-src="(https?://[^"]+)"', after)
|
||||
if not cover:
|
||||
cover = re.search(r'src="(https?://[^"]+\.(?:jpg|jpeg|png|webp))"', after, re.I)
|
||||
pic_url = cover.group(1) if cover else ""
|
||||
if pic_url.startswith("http://"):
|
||||
pic_url = pic_url.replace("http://", "https://", 1)
|
||||
# 过滤失效图床域名
|
||||
if pic_url:
|
||||
host = urlparse(pic_url).hostname or ""
|
||||
if host in DEAD_IMG_HOSTS:
|
||||
pic_url = ""
|
||||
# 备注
|
||||
remark = re.search(r'class="titles"[^>]*>([^<]+)<', after)
|
||||
if not remark:
|
||||
remark = re.search(r'class="[^"]*prb[^"]*"[^>]*>([^<]+)<', after)
|
||||
# 分类过滤: 从 mcat 提取分类标签(注意span内可能有换行)
|
||||
mcat_raw = re.search(r'class="mcat">([\s\S]*?)</div>', after)
|
||||
if mcat_raw and cat_filter:
|
||||
mcat_text = re.sub(r'<[^>]+>', '', mcat_raw.group(1)).strip()
|
||||
if cat_filter == "电视剧":
|
||||
if any(k in mcat_text for k in ANIME_KEYWORDS):
|
||||
continue
|
||||
elif cat_filter == "动漫":
|
||||
if any(k in mcat_text for k in TV_KEYWORDS):
|
||||
continue
|
||||
seen.add(vid)
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": name[:50],
|
||||
"vod_pic": pic_url,
|
||||
"vod_remarks": remark.group(1).strip() if remark else "",
|
||||
})
|
||||
return items
|
||||
@@ -0,0 +1,640 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
import sys
|
||||
import re
|
||||
import json
|
||||
from urllib.parse import urljoin, quote, unquote, urlparse
|
||||
from html import unescape as html_unescape
|
||||
sys.path.append('..')
|
||||
try:
|
||||
from base.spider import Spider
|
||||
except ImportError:
|
||||
class Spider:
|
||||
def fetch(self, url, headers=None, **kw):
|
||||
import requests as rq
|
||||
kw.pop('timeout', None)
|
||||
r = rq.get(url, headers=headers, timeout=15, **kw)
|
||||
r.encoding = 'utf-8'
|
||||
return r
|
||||
|
||||
HOST = "https://www.brovod.com"
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
|
||||
# 封面URL中这些域名已失效,对应影片会被过滤掉
|
||||
DEAD_IMG_HOSTS = {"image.caiji.cyou", "wim.xrc888.com"}
|
||||
# 分类映射: tid -> 中文名
|
||||
CATEGORIES = {
|
||||
"Movies": "电影",
|
||||
"TV": "剧集",
|
||||
"Anime": "动漫",
|
||||
"Documentaries": "纪录片",
|
||||
"Snaps": "短剧",
|
||||
"Shows": "综艺",
|
||||
}
|
||||
|
||||
# weserv.nl 代理前缀,需要去掉以获取原始URL
|
||||
WESERV_PREFIX = "https://images.weserv.nl/?url="
|
||||
|
||||
# from值 -> 线路显示名 (来自 playerconfig.js)
|
||||
# 蓝光①(xdxl/xdyy)和蓝光④(xdjp)已禁用(无法播放)
|
||||
FROM_DISPLAY = {
|
||||
"xdrs": "蓝光②", "xdac": "蓝光③", "xd5": "蓝光⑤",
|
||||
"bfzym3u8": "极速①", "1080zyk": "极速②", "jsm3u8": "极速③",
|
||||
}
|
||||
# 禁用的from值,对应的线路会被过滤掉
|
||||
BLOCKED_FROMS = {"xdxl", "xdyy", "xdjp"}
|
||||
|
||||
|
||||
class Spider(Spider):
|
||||
def init(self, extend=""):
|
||||
global HOST
|
||||
try:
|
||||
r = self.fetch(HOST, headers={"User-Agent": UA}, timeout=15000)
|
||||
if hasattr(r, 'url') and r.url and r.url != HOST.rstrip("/"):
|
||||
HOST = r.url.rstrip("/")
|
||||
except:
|
||||
pass
|
||||
|
||||
def homeContent(self, filter=False):
|
||||
r = {"class": []}
|
||||
for k, v in CATEGORIES.items():
|
||||
r["class"].append({"type_id": k, "type_name": v})
|
||||
return r
|
||||
|
||||
def homeVideoContent(self):
|
||||
try:
|
||||
r = self.fetch(HOST, headers={"User-Agent": UA}, timeout=15000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
return {"list": [it for it in self._items(html) if it.get("vod_pic")]}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def categoryContent(self, tid, pg=1, filter=False, extend=""):
|
||||
pn = 1
|
||||
try:
|
||||
pn = max(int(str(pg)), 1)
|
||||
except:
|
||||
pass
|
||||
cat = str(tid)
|
||||
if cat not in CATEGORIES:
|
||||
return {"page": pn, "pagecount": 1, "limit": 36, "total": 0, "list": []}
|
||||
try:
|
||||
url = self._build_category_url(cat, pn, extend)
|
||||
r = self.fetch(url, headers={"User-Agent": UA}, timeout=30000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
items = [it for it in self._items(html) if it.get("vod_pic")]
|
||||
pc = self._pagecount(html, pn)
|
||||
return {"page": pn, "pagecount": pc, "limit": 36, "total": len(items), "list": items}
|
||||
except:
|
||||
return {"page": pn, "pagecount": 1, "limit": 36, "total": 0, "list": []}
|
||||
|
||||
def _build_category_url(self, cat, pn, extend=""):
|
||||
"""构建分类URL
|
||||
网站URL实际格式为12段(以-分隔):
|
||||
/show/{类名}-{地区}-{排序}-{类型}-{语言}-{字母}-----{页码}---{年份}/
|
||||
段位说明(0-indexed):
|
||||
0: 类名 (Movies/TV/Anime/Documentaries/Snaps/Shows)
|
||||
1: 地区 (大陆/香港/美国/...)
|
||||
2: 排序 (time/hits/score)
|
||||
3: 类型 (喜剧/爱情/动作/...)
|
||||
4: 语言 (国语/英语/粤语/...)
|
||||
5: 字母 (A/B/C/.../0-9)
|
||||
6-7: 空
|
||||
8: 页码 (1/2/3/...)
|
||||
9-10: 空
|
||||
11: 年份 (2026/2025/...)
|
||||
示例: /show/Movies-----------/ (无筛选,首页)
|
||||
示例: /show/Movies--------1---/ (第1页) 注意首页无筛选时不需要页码
|
||||
示例: /show/Movies--------2---/ (第2页)
|
||||
示例: /show/Movies-%E5%A4%A7%E9%99%86----------/ (地区=大陆)
|
||||
示例: /show/Movies---%E5%96%9C%E5%89%A7--------/ (类型=喜剧)
|
||||
示例: /show/Movies-----------2026/ (年份=2026)
|
||||
示例: /show/Movies--time---------/ (按时间排序)
|
||||
"""
|
||||
ext = {}
|
||||
if extend and isinstance(extend, str) and extend.strip():
|
||||
try:
|
||||
ext = json.loads(extend)
|
||||
except:
|
||||
pass
|
||||
|
||||
area = ext.get("area") or ""
|
||||
sort = ext.get("sort") or ""
|
||||
genre = ext.get("class") or ""
|
||||
lang = ext.get("lang") or ""
|
||||
letter = ext.get("letter") or ""
|
||||
year = ext.get("year") or ""
|
||||
|
||||
# 12段: [类名, 地区, 排序, 类型, 语言, 字母, "", "", 页码, "", "", 年份]
|
||||
parts = [cat, "", "", "", "", "", "", "", "", "", "", ""]
|
||||
if area and area != "全部":
|
||||
parts[1] = quote(area)
|
||||
if sort:
|
||||
parts[2] = sort
|
||||
if genre and genre != "全部":
|
||||
parts[3] = quote(genre)
|
||||
if lang and lang != "全部":
|
||||
parts[4] = quote(lang)
|
||||
if letter and letter != "全部":
|
||||
parts[5] = letter
|
||||
if pn > 1:
|
||||
parts[8] = str(pn)
|
||||
if year and year != "全部":
|
||||
parts[11] = str(year)
|
||||
|
||||
url = "/show/" + "-".join(parts) + "/"
|
||||
return HOST + url
|
||||
|
||||
def detailContent(self, ids):
|
||||
if isinstance(ids, list):
|
||||
vid = ids[0] if ids else ""
|
||||
else:
|
||||
vid = str(ids) if ids else ""
|
||||
if not vid:
|
||||
return {"list": []}
|
||||
detail_id = vid
|
||||
|
||||
# 构建详情页URL
|
||||
if "/" in vid:
|
||||
detail_url = vid if vid.startswith("http") else urljoin(HOST, vid)
|
||||
else:
|
||||
detail_url = f"{HOST}/detail/{vid}/"
|
||||
|
||||
try:
|
||||
r = self.fetch(detail_url, headers={"User-Agent": UA}, timeout=30000)
|
||||
h = r.text if hasattr(r, 'text') else str(r)
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
d = {
|
||||
"vod_id": detail_id,
|
||||
"vod_name": "",
|
||||
"vod_pic": "",
|
||||
"vod_year": "",
|
||||
"vod_area": "",
|
||||
"vod_class": "",
|
||||
"vod_director": "",
|
||||
"vod_actor": "",
|
||||
"vod_content": "",
|
||||
"vod_remarks": "",
|
||||
"vod_play_from": "",
|
||||
"vod_play_url": "",
|
||||
}
|
||||
|
||||
# === 标题: <h3 class="slide-info-title hide">名称</h3> ===
|
||||
tn = re.search(r'class="slide-info-title[^"]*"[^>]*>(.*?)</(?:h1|h2|h3)>', h, re.S)
|
||||
if tn:
|
||||
d["vod_name"] = re.sub(r'<[^>]+>', '', tn.group(1)).strip()
|
||||
if not d["vod_name"]:
|
||||
tn = re.search(r'<title>(.*?)</title>', h)
|
||||
if tn:
|
||||
d["vod_name"] = tn.group(1).split("-")[0].strip()
|
||||
|
||||
# === 封面: 详情页中 data-src ===
|
||||
p = re.search(r'data-src="(https?://[^"]+)"', h)
|
||||
if not p:
|
||||
p = re.search(r'src="(https?://[^"]+\.(?:jpg|jpeg|png|webp))"', h, re.I)
|
||||
if p:
|
||||
d["vod_pic"] = self._clean_pic(p.group(1))
|
||||
|
||||
# === slide-info 区域提取年份/地区/类型/导演/演员/备注/更新时间 ===
|
||||
# 结构:
|
||||
# <div class="slide-info">
|
||||
# <span class="slide-info-remarks"><a>2025</a></span>
|
||||
# <span class="slide-info-remarks"><a>大陆</a></span>
|
||||
# <span class="slide-info-remarks"><a>动作</a></span>...
|
||||
# </div>
|
||||
# <div class="slide-info"><strong>备注 :</strong>蓝光</div>
|
||||
# <div class="slide-info"><strong>导演 :</strong><a>巨兴茂</a>/</div>
|
||||
# <div class="slide-info"><strong>演员 :</strong><a>杨旭文</a>/...</div>
|
||||
# <div class="slide-info"><strong>更新 :</strong>2026-02-21 18:28:06</div>
|
||||
|
||||
# 提取slide-info区域中的标签(年份、地区、类型)
|
||||
slide_info_tags = re.findall(
|
||||
r'<div class="slide-info[^"]*">\s*'
|
||||
r'((?:<span class="slide-info-remarks">.*?</span>\s*)+)\s*</div>',
|
||||
h, re.S
|
||||
)
|
||||
if slide_info_tags:
|
||||
tag_text = re.sub(r'<[^>]+>', ' ', slide_info_tags[0])
|
||||
tag_text = re.sub(r'\s+', ' ', tag_text).strip()
|
||||
parts = [t.strip() for t in tag_text.split() if t.strip()]
|
||||
# 第一个通常是年份
|
||||
for p_text in parts:
|
||||
if re.match(r'^\d{4}$', p_text) and not d["vod_year"]:
|
||||
d["vod_year"] = p_text
|
||||
# 地区关键词
|
||||
elif p_text in ("大陆", "香港", "台湾", "美国", "韩国", "日本", "英国",
|
||||
"法国", "德国", "泰国", "印度", "加拿大", "其他",
|
||||
"意大利", "西班牙", "未知"):
|
||||
if not d["vod_area"]:
|
||||
d["vod_area"] = p_text
|
||||
elif not d["vod_area"] and re.match(r'^[\u4e00-\u9fff]{2,3}$', p_text):
|
||||
d["vod_area"] = p_text
|
||||
# 剩余的是类型
|
||||
type_parts = []
|
||||
year_done = False
|
||||
area_done = False
|
||||
for p_text in parts:
|
||||
if re.match(r'^\d{4}$', p_text):
|
||||
year_done = True
|
||||
continue
|
||||
if p_text in ("大陆", "香港", "台湾", "美国", "韩国", "日本", "英国",
|
||||
"法国", "德国", "泰国", "印度", "加拿大", "其他",
|
||||
"意大利", "西班牙", "未知"):
|
||||
area_done = True
|
||||
continue
|
||||
if year_done and area_done:
|
||||
type_parts.append(p_text)
|
||||
if type_parts:
|
||||
d["vod_class"] = " ".join(type_parts)
|
||||
|
||||
# 备注: <strong>备注 :</strong>蓝光
|
||||
rm = re.search(r'备注\s*[::]\s*</strong>\s*([^<\s]+)', h)
|
||||
if rm:
|
||||
d["vod_remarks"] = rm.group(1).strip()
|
||||
|
||||
# 导演: <strong>导演 :</strong><a>巨兴茂</a>
|
||||
dm = re.search(r'导演\s*[::]\s*</strong>([\s\S]*?)(?:</div>)', h)
|
||||
if dm:
|
||||
d["vod_director"] = re.sub(r'<[^>]+>', '', dm.group(1)).replace("/", ",").strip().rstrip(",").strip()
|
||||
|
||||
# 演员: <strong>演员 :</strong><a>杨旭文</a>/<a>杨志刚</a>/...
|
||||
am = re.search(r'演员\s*[::]\s*</strong>([\s\S]*?)(?:</div>)', h)
|
||||
if am:
|
||||
d["vod_actor"] = re.sub(r'<[^>]+>', '', am.group(1)).replace("/", ",").strip().rstrip(",").strip()
|
||||
|
||||
# === 简介 ===
|
||||
# 优先从slide-info-content区域提取
|
||||
desc_m = re.search(r'class="slide-info-content"[^>]*>([\s\S]*?)</div>', h)
|
||||
if desc_m:
|
||||
d["vod_content"] = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', desc_m.group(1))).strip()[:500]
|
||||
if not d["vod_content"]:
|
||||
# 尝试从影视简介区域提取
|
||||
desc_m = re.search(r'影视简介[\s\S]{0,10}?>([\s\S]*?)</(?:div|p)>', h)
|
||||
if desc_m:
|
||||
d["vod_content"] = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', desc_m.group(1))).strip()[:500]
|
||||
if not d["vod_content"]:
|
||||
# 尝试card-text区域(搜索结果页)
|
||||
desc_m = re.search(r'class="card-text">([\s\S]*?)</div>', h)
|
||||
if desc_m:
|
||||
d["vod_content"] = re.sub(r'\s+', ' ', re.sub(r'<[^>]+>', '', desc_m.group(1))).strip()[:500]
|
||||
|
||||
# === 播放列表解析 ===
|
||||
try:
|
||||
pf_list, pu_list = [], []
|
||||
|
||||
# 1. 从anthology-tab提取线路名(按顺序)
|
||||
# HTML结构: <div class="anthology-tab">...<a class="swiper-slide"><i>...</i>蓝光①<span class="badge">40</span></a>...</div>
|
||||
tab_names = []
|
||||
tab_section = re.search(
|
||||
r'<div class="anthology-tab[^"]*">(.*?)</div>\s*</div>\s*<div class="anthology-list',
|
||||
h, re.S
|
||||
)
|
||||
if not tab_section:
|
||||
tab_section = re.search(
|
||||
r'<div class="anthology-tab[^"]*">(.*?)</div>\s*</div>',
|
||||
h, re.S
|
||||
)
|
||||
if tab_section:
|
||||
for tab_m in re.finditer(r'<a[^>]*class="swiper-slide"[^>]*>([\s\S]*?)</a>', tab_section.group(1)):
|
||||
tab_text = html_unescape(re.sub(r'<[^>]+>', '', tab_m.group(1)).strip())
|
||||
tab_text = tab_text.replace('\xa0', ' ').strip()
|
||||
if tab_text:
|
||||
tab_names.append(tab_text)
|
||||
|
||||
# 2. 从anthology-list-box提取各线路的集数链接
|
||||
# HTML结构: <div class="anthology-list"><div class="anthology-list-box">...<ul class="anthology-list-play">...<a href="/play/xxx-1-1/">集名</a>...</div>...</div>
|
||||
# 每个anthology-list-box对应一个线路
|
||||
|
||||
list_section = re.search(
|
||||
r'<div class="anthology-list[^"]*">(.*?)</div>\s*</div>\s*</div>',
|
||||
h, re.S
|
||||
)
|
||||
if not list_section:
|
||||
# 更宽松匹配
|
||||
list_section = re.search(
|
||||
r'class="anthology-list[^"]*"(.*?)$',
|
||||
h, re.S
|
||||
)
|
||||
|
||||
if list_section:
|
||||
boxes = re.findall(
|
||||
r'<div class="anthology-list-box[^"]*">(.*?)</ul>\s*</div>',
|
||||
list_section.group(1), re.S
|
||||
)
|
||||
if not boxes:
|
||||
boxes = re.findall(
|
||||
r'<div class="anthology-list-box[^"]*">(.*?)(?=<div class="anthology-list-box|</div>\s*</div>)',
|
||||
list_section.group(1), re.S
|
||||
)
|
||||
|
||||
for idx, box_html in enumerate(boxes):
|
||||
# 提取集数链接
|
||||
eps = re.findall(
|
||||
r'href="(/play/[^"]+)"[^>]*>(.*?)</a>',
|
||||
box_html, re.S
|
||||
)
|
||||
if not eps:
|
||||
continue
|
||||
|
||||
ep_list = []
|
||||
for ep_url, ep_name in eps:
|
||||
ep_name = re.sub(r'<[^>]+>', '', ep_name).strip()
|
||||
if not ep_name:
|
||||
ep_name = f"第{len(ep_list) + 1}集"
|
||||
full_url = urljoin(HOST, ep_url)
|
||||
ep_list.append(f"{ep_name}${full_url}")
|
||||
|
||||
if ep_list:
|
||||
# 检查该线路的from值,过滤禁用线路
|
||||
skip = False
|
||||
try:
|
||||
first_url = urljoin(HOST, eps[0][0]) if eps else ""
|
||||
rp = self.fetch(first_url, headers={"User-Agent": UA}, timeout=5000)
|
||||
hp = rp.text if hasattr(rp, 'text') else str(rp)
|
||||
fm = re.search(r'"from"\s*:\s*"([^"]+)"', hp)
|
||||
if fm and fm.group(1) in BLOCKED_FROMS:
|
||||
skip = True
|
||||
except:
|
||||
pass
|
||||
if skip:
|
||||
continue
|
||||
# 线路名: 优先使用anthology-tab中的名称
|
||||
if idx < len(tab_names):
|
||||
line_name = tab_names[idx]
|
||||
else:
|
||||
line_name = f"线路{idx + 1}"
|
||||
pf_list.append(line_name)
|
||||
pu_list.append("#".join(ep_list))
|
||||
|
||||
# 如果上述方式没找到,备用方案: 直接匹配所有play链接并按线路序号分组
|
||||
if not pf_list:
|
||||
play_links = re.findall(
|
||||
r'href="(/play/[^"]+-\d+-\d+/)"[^>]*>([\s\S]*?)</a>',
|
||||
h, re.S
|
||||
)
|
||||
line_episodes = {}
|
||||
line_froms = {}
|
||||
for link_url, ep_name in play_links:
|
||||
ep_name = re.sub(r'<[^>]+>', '', ep_name).strip()
|
||||
m2 = re.search(r'/play/.+?-(\d+)-(\d+)/', link_url)
|
||||
if m2:
|
||||
line_idx = m2.group(1)
|
||||
if line_idx not in line_episodes:
|
||||
line_episodes[line_idx] = []
|
||||
full_url = urljoin(HOST, link_url)
|
||||
line_episodes[line_idx].append((ep_name, full_url))
|
||||
# 获取from值(只需每条线路检查一次)
|
||||
if line_idx not in line_froms:
|
||||
try:
|
||||
rp = self.fetch(full_url, headers={"User-Agent": UA}, timeout=5000)
|
||||
hp = rp.text if hasattr(rp, 'text') else str(rp)
|
||||
pm = re.search(r'"from"\s*:\s*"([^"]+)"', hp)
|
||||
if pm:
|
||||
line_froms[line_idx] = pm.group(1)
|
||||
except:
|
||||
pass
|
||||
|
||||
for line_idx in sorted(line_episodes.keys(), key=lambda x: int(x)):
|
||||
eps = line_episodes[line_idx]
|
||||
from_val = line_froms.get(line_idx, "")
|
||||
if from_val in BLOCKED_FROMS:
|
||||
continue
|
||||
line_name = FROM_DISPLAY.get(from_val, f"线路{line_idx}")
|
||||
ep_list = []
|
||||
for ep_name, ep_url in eps:
|
||||
if not ep_name:
|
||||
ep_name = f"第{len(ep_list) + 1}集"
|
||||
ep_list.append(f"{ep_name}${ep_url}")
|
||||
if ep_list:
|
||||
pf_list.append(line_name)
|
||||
pu_list.append("#".join(ep_list))
|
||||
|
||||
if pf_list:
|
||||
d["vod_play_from"] = "$$$".join(pf_list)
|
||||
d["vod_play_url"] = "$$$".join(pu_list)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return {"list": [d]}
|
||||
|
||||
def searchContent(self, key, quick=False, pg="1"):
|
||||
try:
|
||||
pn = 1
|
||||
try:
|
||||
pn = int(str(pg))
|
||||
except:
|
||||
pass
|
||||
# 搜索URL: /ss/-------------/?wd=关键词
|
||||
url = f"{HOST}/ss/-------------/?wd={quote(key)}"
|
||||
r = self.fetch(url, headers={"User-Agent": UA}, timeout=30000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
items = self._search_items(html)
|
||||
return {"list": items, "page": pn}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags=None):
|
||||
url = str(id) if id else str(flag)
|
||||
# 如果已经是m3u8直链,直接返回
|
||||
if url.startswith("http") and ".m3u8" in url:
|
||||
return {"url": url}
|
||||
# 构建完整URL
|
||||
if url.startswith("http"):
|
||||
full_url = url
|
||||
else:
|
||||
if not url.startswith("/"):
|
||||
url = "/" + url
|
||||
full_url = urljoin(HOST, url)
|
||||
|
||||
try:
|
||||
r = self.fetch(full_url, headers={"User-Agent": UA}, timeout=30000)
|
||||
h = r.text if hasattr(r, 'text') else str(r)
|
||||
except:
|
||||
return {"url": ""}
|
||||
|
||||
# 尝试提取 var player_aaaa 中的数据
|
||||
player_m = re.search(r'player_aaaa\s*=\s*(\{.*?\})\s*</script>', h, re.S)
|
||||
if player_m:
|
||||
try:
|
||||
data = json.loads(player_m.group(1))
|
||||
play_url = data.get("url", "")
|
||||
# 如果url是明文的m3u8链接,直接返回
|
||||
if play_url and play_url.startswith("http") and ".m3u8" in play_url:
|
||||
return {"url": play_url}
|
||||
# emoji加密的url,构建解析地址返回
|
||||
if play_url:
|
||||
from urllib.parse import quote
|
||||
parse_url = f"https://play.brovod.com/?url={quote(play_url)}"
|
||||
# parse=1 告诉APP用WebView加载解析页面嗅探m3u8
|
||||
return {"url": parse_url, "parse": 1, "header": {"User-Agent": UA}}
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 尝试匹配页面中的m3u8直链
|
||||
m3u8 = re.search(r'(https?://[^\s"\'<>]+\.m3u8)', h)
|
||||
if m3u8:
|
||||
return {"url": m3u8.group(1)}
|
||||
|
||||
# 兜底:返回播放页URL,让APP的WebView去嗅探
|
||||
return {"url": full_url, "parse": 1, "header": {"User-Agent": UA}}
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
|
||||
def _clean_pic(self, pic_url):
|
||||
"""清理封面URL: 去掉weserv代理前缀, HTTP升级HTTPS, 过滤失效图床"""
|
||||
if not pic_url:
|
||||
return ""
|
||||
# 去掉weserv.nl代理前缀
|
||||
if pic_url.startswith(WESERV_PREFIX):
|
||||
pic_url = pic_url[len(WESERV_PREFIX):]
|
||||
try:
|
||||
pic_url = unquote(pic_url)
|
||||
except Exception:
|
||||
pass
|
||||
# HTTP升级为HTTPS
|
||||
if pic_url.startswith("http://"):
|
||||
pic_url = pic_url.replace("http://", "https://", 1)
|
||||
# 过滤失效图床域名
|
||||
host = urlparse(pic_url).hostname or ""
|
||||
if host in DEAD_IMG_HOSTS:
|
||||
return ""
|
||||
return pic_url
|
||||
|
||||
def _items(self, html, cat_filter=""):
|
||||
"""从分类/首页HTML中提取列表项
|
||||
HTML结构:
|
||||
<a class="public-list-exp" href="/detail/xxx-123/" title="名称">
|
||||
<img data-src="封面URL" />
|
||||
<span class="public-list-prb hide ft2">备注</span>
|
||||
</a>
|
||||
"""
|
||||
items, seen = [], set()
|
||||
# 匹配 public-list-exp 中的详情链接
|
||||
for m in re.finditer(
|
||||
r'class="public-list-exp"[^>]*href="(/detail/[^"]+)"[^>]*title="([^"]*)"',
|
||||
html
|
||||
):
|
||||
detail_url = m.group(1)
|
||||
name = m.group(2).strip()
|
||||
if not name or len(name) > 100:
|
||||
continue
|
||||
# 提取数字ID去重
|
||||
vid_m = re.search(r'-(\d+)/?$', detail_url)
|
||||
vid = vid_m.group(1) if vid_m else detail_url
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
|
||||
# 在匹配位置之后查找封面和备注(向前搜索3000字符)
|
||||
after = html[m.end():m.end() + 3000]
|
||||
|
||||
# 封面: data-src
|
||||
cover = re.search(r'data-src="(https?://[^"]+)"', after)
|
||||
if not cover:
|
||||
cover = re.search(r'src="(https?://[^"]+\.(?:jpg|jpeg|png|webp))"', after, re.I)
|
||||
pic_url = self._clean_pic(cover.group(1)) if cover else ""
|
||||
|
||||
# 备注: class="public-list-prb"
|
||||
remark = re.search(r'class="public-list-prb[^"]*"[^>]*>([^<]+)<', after)
|
||||
if not remark:
|
||||
remark = re.search(r'class="[^"]*prb[^"]*"[^>]*>([^<]+)<', after)
|
||||
|
||||
items.append({
|
||||
"vod_id": detail_url,
|
||||
"vod_name": name[:50],
|
||||
"vod_pic": pic_url,
|
||||
"vod_remarks": remark.group(1).strip() if remark else "",
|
||||
})
|
||||
return items
|
||||
|
||||
def _search_items(self, html):
|
||||
"""从搜索结果HTML中提取列表项
|
||||
搜索结果HTML结构(search-box):
|
||||
<div class="public-list-box search-box flex rel">
|
||||
<div class="left public-list-bj">
|
||||
<a href="/detail/xxx-123/"><img data-src="封面URL" />
|
||||
<span class="public-list-prb hide ft2">备注</span></a>
|
||||
</div>
|
||||
<div class="right">
|
||||
<div class="thumb-txt"><a href="/detail/xxx-123/">名称</a></div>
|
||||
...
|
||||
</div>
|
||||
</div>
|
||||
"""
|
||||
items, seen = [], set()
|
||||
# 按 public-list-box search-box 块分割
|
||||
boxes = re.split(r'<div class="public-list-box[^"]*search-box', html)
|
||||
for box_html in boxes[1:]: # 跳过第一个(分割前的内容)
|
||||
# 提取详情链接
|
||||
detail_m = re.search(r'href="(/detail/[^"]+)"', box_html)
|
||||
if not detail_m:
|
||||
continue
|
||||
detail_url = detail_m.group(1)
|
||||
vid_m = re.search(r'-(\d+)/?$', detail_url)
|
||||
vid = vid_m.group(1) if vid_m else detail_url
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
|
||||
# 名称: thumb-txt中的链接文本
|
||||
name_m = re.search(r'class="thumb-txt[^"]*"[^>]*><a[^>]*href="[^"]*"[^>]*>(.*?)</a>', box_html, re.S)
|
||||
if name_m:
|
||||
name = html_unescape(re.sub(r'<[^>]+>', '', name_m.group(1)).strip())
|
||||
else:
|
||||
name_m = re.search(r'title="([^"]+)"', box_html)
|
||||
name = name_m.group(1).strip() if name_m else ""
|
||||
if not name or len(name) > 100:
|
||||
continue
|
||||
|
||||
# 封面: data-src
|
||||
cover = re.search(r'data-src="(https?://[^"]+)"', box_html)
|
||||
if not cover:
|
||||
cover = re.search(r'src="(https?://[^"]+\.(?:jpg|jpeg|png|webp))"', box_html, re.I)
|
||||
pic_url = self._clean_pic(cover.group(1)) if cover else ""
|
||||
|
||||
# 备注: public-list-prb
|
||||
remark = re.search(r'class="public-list-prb[^"]*"[^>]*>([^<]+)<', box_html)
|
||||
|
||||
items.append({
|
||||
"vod_id": detail_url,
|
||||
"vod_name": name[:50],
|
||||
"vod_pic": pic_url,
|
||||
"vod_remarks": remark.group(1).strip() if remark else "",
|
||||
})
|
||||
return items
|
||||
|
||||
def _pagecount(self, html, current_page=1):
|
||||
"""从HTML中提取总页数
|
||||
分页结构: ...尾页</a>...</div>...
|
||||
尾页链接: href="/show/Movies--------1643---/"
|
||||
"""
|
||||
max_page = current_page
|
||||
|
||||
# 匹配 "尾页" 链接中的页码
|
||||
tail_m = re.search(r'尾页[^>]*href="[^"]*?-(\d+)---/"', html)
|
||||
if tail_m:
|
||||
try:
|
||||
max_page = int(tail_m.group(1))
|
||||
return max_page
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 从分页数字链接中推断最大页码
|
||||
# URL格式: /show/Movies--------{N}---/
|
||||
pages = re.findall(r'/show/\w+-{4,}(\d+)---/', html)
|
||||
for p in pages:
|
||||
try:
|
||||
n = int(p)
|
||||
if n > max_page:
|
||||
max_page = n
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 如果有下一页标记且当前页接近最大值,多给几页
|
||||
has_next = re.search(r'>下一页<', html)
|
||||
if has_next and max_page <= current_page + 5:
|
||||
max_page = current_page + 5
|
||||
|
||||
return max_page if max_page >= 1 else 1
|
||||
@@ -0,0 +1,440 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
top3.zgtv.online (追光影视) 爬虫
|
||||
- MX Pro CMS模板
|
||||
- 分类: /vodshow/{catId}---{area}--{letter}----{page}---.html
|
||||
- 详情: /voddetail/{id}.html
|
||||
- 播放: /vodplay/{id}-{sid}-{nid}.html -> player_aaaa JS变量
|
||||
- 搜索: /vodsearch/{keyword}-------------.html
|
||||
- 播放链接是视频站原始URL(芒果/爱奇艺等), 需要解析线路
|
||||
"""
|
||||
import sys
|
||||
import re
|
||||
import json
|
||||
from collections import defaultdict
|
||||
import requests as rq
|
||||
from urllib.parse import quote
|
||||
|
||||
sys.path.append('..')
|
||||
try:
|
||||
from base.spider import Spider
|
||||
except ImportError:
|
||||
class Spider:
|
||||
def fetch(self, url, headers=None, **kw):
|
||||
kw.pop('timeout', None)
|
||||
r = rq.get(url, headers=headers, timeout=15, **kw)
|
||||
r.encoding = 'utf-8'
|
||||
return r
|
||||
|
||||
def _(x): return x
|
||||
|
||||
HOST = "https://top3.zgtv.online"
|
||||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
|
||||
|
||||
# 播放源 -> 解析接口映射 (from playerconfig.js) - 仅供参考, 当前不使用
|
||||
_PARSE_MAP = {
|
||||
"wsym3u8": "https://wsyzy.vip/m3u8/?url=",
|
||||
"bfzym3u8": "https://free.maccms.xyz/?url=",
|
||||
"360zy": "https://free.maccms.xyz/?url=",
|
||||
"mtm3u8": "https://free.maccms.xyz/?url=",
|
||||
"qq": "https://z01.zgtv.online/player/?url=",
|
||||
"qiyi": "https://z01.zgtv.online/player/?url=",
|
||||
"youku": "https://z01.zgtv.online/player/?url=",
|
||||
"mgtv": "https://z01.zgtv.online/player/?url=",
|
||||
"bilibili": "https://z01.zgtv.online/player/?url=",
|
||||
"mjzy": "https://svip.qlplayer.cyou/?url=",
|
||||
}
|
||||
|
||||
# 备用解析接口 (仅保留供参考)
|
||||
_BACKUP_PARSERS = [
|
||||
("极速", "https://jx.2s0.cn/player/?url="),
|
||||
("super", "https://super.playr.top/?url="),
|
||||
("fongmi", "https://json.fongmi.cc/web?url="),
|
||||
("Jn1", "https://yparse.jn1.cc/index.php?url="),
|
||||
("PlayerJY", "https://jx.playerjy.com/?url="),
|
||||
("冰豆", "https://bd.jx.cn/?url="),
|
||||
("剖元", "https://www.pouyun.com/?url="),
|
||||
("七哥", "https://jx.nnxv.cn/tv.php?url="),
|
||||
("夜幕", "https://www.yemu.xyz/?url="),
|
||||
("Yparse", "https://jx.yparse.com/index.php?url="),
|
||||
("ik9", "https://yparse.ik9.cc/index.php?url="),
|
||||
("花旗", "https://www.huaqi.live/?url="),
|
||||
("网站z01", "https://z01.zgtv.online/player/?url="),
|
||||
("svip2", "https://svip.qlplayer.cyou/?url="),
|
||||
("maccms", "https://free.maccms.xyz/?url="),
|
||||
("wsyzy", "https://wsyzy.vip/m3u8/?url="),
|
||||
]
|
||||
|
||||
# from字段 -> 中文显示名
|
||||
FROM_NAMES = {
|
||||
"qq": "腾讯",
|
||||
"qiyi": "奇艺",
|
||||
"youku": "优酷",
|
||||
"mgtv": "芒果",
|
||||
"bilibili": "B站",
|
||||
"wsym3u8": "自营线路",
|
||||
"bfzym3u8": "暴风资源",
|
||||
"360zy": "无广告",
|
||||
"mtm3u8": "芒果资源",
|
||||
"mjzy": "自营蓝光",
|
||||
"dplayer": "DPlayer",
|
||||
"videojs": "VideoJS",
|
||||
}
|
||||
|
||||
# 需要解析的from类型 (视频站链接, 需要通过第三方解析接口)
|
||||
NEED_PARSE_FROM = {"qq", "qiyi", "youku", "mgtv", "bilibili"}
|
||||
|
||||
# 主分类
|
||||
CLASS_MAP = {
|
||||
"20": "电影",
|
||||
"37": "连续剧",
|
||||
"43": "动漫",
|
||||
"45": "综艺",
|
||||
}
|
||||
|
||||
|
||||
class Spider(Spider):
|
||||
|
||||
def init(self, extend=""):
|
||||
self._session = rq.Session()
|
||||
self._session.headers.update({"User-Agent": UA})
|
||||
try:
|
||||
self._session.get(HOST, timeout=10)
|
||||
except:
|
||||
pass
|
||||
|
||||
def getName(self):
|
||||
return "zgtv"
|
||||
|
||||
def isVideoFormat(self, url):
|
||||
return ".m3u8" in url or ".mp4" in url
|
||||
|
||||
def manualVideoCheck(self):
|
||||
return False
|
||||
|
||||
def homeContent(self, filter=False):
|
||||
classes = []
|
||||
for tid, name in CLASS_MAP.items():
|
||||
classes.append({"type_id": tid, "type_name": name})
|
||||
return {"class": classes}
|
||||
|
||||
def homeVideoContent(self):
|
||||
try:
|
||||
items = self._fetch_list(f"{HOST}/vodtype/20.html")
|
||||
return {"list": items}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def categoryContent(self, tid, pg=1, filter=False, extend=None):
|
||||
try:
|
||||
pn = max(int(str(pg)), 1)
|
||||
# /vodshow/{catId}--------{page}---.html
|
||||
url = f"{HOST}/vodshow/{tid}--------{pn}---.html"
|
||||
items = self._fetch_list(url)
|
||||
return {"list": items, "page": pn, "pagecount": pn + 10, "limit": 24, "total": 0}
|
||||
except:
|
||||
return {"list": [], "page": pg, "pagecount": 1, "limit": 24, "total": 0}
|
||||
|
||||
def detailContent(self, ids):
|
||||
try:
|
||||
vid = str(ids[0]) if ids else ""
|
||||
if not vid:
|
||||
return {"list": []}
|
||||
|
||||
url = f"{HOST}/voddetail/{vid}.html"
|
||||
r = self._get(url, timeout=15000)
|
||||
html = r.text
|
||||
|
||||
# 标题
|
||||
title = self._extract(html, r'<h1[^>]*class="page-title"[^>]*>([^<]+)</h1>') or ""
|
||||
if not title:
|
||||
title = self._extract(html, r'<h1[^>]*>([^<]+)</h1>') or ""
|
||||
|
||||
# 封面
|
||||
cover = self._extract(html, r'data-original="([^"]+\.(?:jpg|png|webp))"') or ""
|
||||
|
||||
if not title:
|
||||
return {"list": []}
|
||||
|
||||
# 播放列表 - 只匹配剧集链接(module-play-list-link类)
|
||||
all_links = re.findall(
|
||||
r'class="module-play-list-link"[^>]+href="(/vodplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>[\s\S]*?<span>([^<]+)</span>',
|
||||
html
|
||||
)
|
||||
# Fallback: 宽松匹配
|
||||
if not all_links:
|
||||
all_links = re.findall(
|
||||
r'href="(/vodplay/(\d+)-(\d+)-(\d+)\.html)"[^>]*>[\s\S]*?<span>([^<]+)</span>',
|
||||
html
|
||||
)
|
||||
|
||||
# 按sid分组 (同一播放源)
|
||||
by_sid = defaultdict(list)
|
||||
for match in all_links:
|
||||
if len(match) == 5:
|
||||
href, v, sid, nid, name = match
|
||||
if v == vid:
|
||||
by_sid[sid].append((int(nid), name.strip(), href))
|
||||
|
||||
# 播放源标签
|
||||
heading = re.search(
|
||||
r'id="y-playList"[^>]*>([\s\S]*?)</div>\s*</div>\s*</div>', html
|
||||
)
|
||||
source_names = []
|
||||
if heading:
|
||||
source_names = re.findall(r'data-dropdown-value="([^"]+)">\s*<span>([^<]+)</span>', heading.group(1))
|
||||
|
||||
pf_list = []
|
||||
pu_list = []
|
||||
src_idx = 0
|
||||
|
||||
for sid in sorted(by_sid.keys()):
|
||||
eps = sorted(by_sid[sid])
|
||||
src_idx += 1
|
||||
|
||||
# 获取该源的from字段(只请求第一集)
|
||||
sid_from = self._get_from_source(vid, sid)
|
||||
|
||||
# 跳过需要解析的视频站源 (qq/qiyi/youku/mgtv/bilibili)
|
||||
if sid_from in NEED_PARSE_FROM:
|
||||
continue
|
||||
|
||||
# 直链源: 优先用网站的source_names, 否则用from映射名
|
||||
if src_idx - 1 < len(source_names):
|
||||
friendly_name = source_names[src_idx - 1][1]
|
||||
else:
|
||||
friendly_name = FROM_NAMES.get(sid_from, f"线路{src_idx}")
|
||||
|
||||
# 直链源
|
||||
ep_list = []
|
||||
for nid, name, href in eps:
|
||||
ep_list.append(f"{name}${href}|{sid_from}")
|
||||
if ep_list:
|
||||
pf_list.append(friendly_name)
|
||||
pu_list.append("#".join(ep_list))
|
||||
|
||||
if not pf_list:
|
||||
return {"list": []}
|
||||
|
||||
vod = {
|
||||
"vod_id": vid,
|
||||
"vod_name": title,
|
||||
"vod_pic": cover,
|
||||
"vod_play_from": "$$$".join(pf_list),
|
||||
"vod_play_url": "$$$".join(pu_list),
|
||||
}
|
||||
return {"list": [vod]}
|
||||
except:
|
||||
return {"list": []}
|
||||
|
||||
def searchContent(self, key, quick=False, pg=1):
|
||||
try:
|
||||
pn = max(int(str(pg)), 1)
|
||||
url = f"{HOST}/vodsearch/{quote(key)}-------------.html"
|
||||
items = self._fetch_list(url)
|
||||
return {"list": items, "page": pn}
|
||||
except:
|
||||
return {"list": [], "page": 1}
|
||||
|
||||
def playerContent(self, flag, id, vipFlags=None):
|
||||
url = str(id) if id else str(flag)
|
||||
if not url:
|
||||
return {"url": ""}
|
||||
if "$" in url:
|
||||
parts = url.split("$", 1)
|
||||
url = parts[1]
|
||||
|
||||
# m3u8/mp4直链直接播
|
||||
if url.startswith("http") and (".m3u8" in url or ".mp4" in url):
|
||||
return {"url": url}
|
||||
|
||||
# 播放页路径 /vodplay/{vid}-{sid}-{nid}.html|from
|
||||
from_src = ""
|
||||
parts = url.split("|")
|
||||
if len(parts) == 2:
|
||||
url = parts[0]
|
||||
from_src = parts[1]
|
||||
|
||||
if url.startswith("/vodplay/"):
|
||||
play_data = self._extract_player_data(url)
|
||||
if play_data:
|
||||
raw_url = play_data.get("url", "")
|
||||
|
||||
# 如果已经是m3u8/mp4直链, 直接播放
|
||||
if raw_url and (".m3u8" in raw_url or ".mp4" in raw_url):
|
||||
return {"url": raw_url}
|
||||
|
||||
# 如果是视频站链接(qq/youku等), 返回原始URL让壳子解析
|
||||
if raw_url and raw_url.startswith("http"):
|
||||
return {"url": raw_url, "parse": 1, "header": {"User-Agent": UA}}
|
||||
return {"url": ""}
|
||||
|
||||
return {"url": url}
|
||||
|
||||
def _get(self, url, timeout=20):
|
||||
r = self._session.get(url, timeout=timeout)
|
||||
# MX Pro CMS可能返回不同编码
|
||||
if r.apparent_encoding:
|
||||
r.encoding = r.apparent_encoding
|
||||
return r
|
||||
|
||||
def _extract(self, text, pattern):
|
||||
m = re.search(pattern, text)
|
||||
return m.group(1) if m else ""
|
||||
|
||||
def _fetch_list(self, url):
|
||||
"""从HTML抓取影片列表"""
|
||||
r = self._get(url, timeout=15000)
|
||||
html = r.text if hasattr(r, 'text') else str(r)
|
||||
|
||||
items = []
|
||||
seen = set()
|
||||
|
||||
# Pattern1: MX Pro - <a href="/voddetail/{id}.html" title="名称">...<img data-original="封面">
|
||||
pattern1 = re.compile(
|
||||
r'<a\s+href="/voddetail/(\d+)\.html"[^>]+title="([^"]+)"'
|
||||
r'[\s\S]*?'
|
||||
r'<img[^>]+data-original="([^"]+\.(?:jpg|png|webp))"'
|
||||
)
|
||||
for vid, title, cover in pattern1.findall(html):
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": title.strip(),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": "",
|
||||
})
|
||||
|
||||
# Pattern1b: 搜索结果 - data-original在alt前面
|
||||
if not items:
|
||||
pattern1b = re.compile(
|
||||
r'<a\s+href="/voddetail/(\d+)\.html"[^>]*>'
|
||||
r'[\s\S]*?'
|
||||
r'<img[^>]+data-original="([^"]+\.(?:jpg|png|webp))"[^>]+alt="([^"]+)"'
|
||||
)
|
||||
for vid, cover, title in pattern1b.findall(html):
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": title.strip(),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": "",
|
||||
})
|
||||
|
||||
# Pattern1c: alt在data-original前面
|
||||
if not items:
|
||||
pattern1c = re.compile(
|
||||
r'<a\s+href="/voddetail/(\d+)\.html"[^>]*>'
|
||||
r'[\s\S]*?'
|
||||
r'<img[^>]+alt="([^"]+)"[^>]+data-original="([^"]+\.(?:jpg|png|webp))"'
|
||||
)
|
||||
for vid, title, cover in pattern1c.findall(html):
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": title.strip(),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": "",
|
||||
})
|
||||
|
||||
# Pattern2: 首页/推荐页 - <a href="/vodplay/{id}-1-1.html" title="名称">...<img data-original="封面">
|
||||
if not items:
|
||||
pattern2 = re.compile(
|
||||
r'<a\s+href="/vodplay/(\d+)-\d+-\d+\.html"[^>]+title="([^"]+)"'
|
||||
r'[\s\S]*?'
|
||||
r'<img[^>]+data-original="([^"]+\.(?:jpg|png|webp))"'
|
||||
)
|
||||
for vid, title, cover in pattern2.findall(html):
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
items.append({
|
||||
"vod_id": vid,
|
||||
"vod_name": title.strip(),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": "",
|
||||
})
|
||||
|
||||
# Pattern3: 备用 - 任何含title和data-original的a标签
|
||||
if not items:
|
||||
pattern3 = re.compile(
|
||||
r'<a[^>]+title="([^"]+)"[^>]*>'
|
||||
r'[\s\S]*?'
|
||||
r'<img[^>]+data-original="([^"]+\.(?:jpg|png|webp))"'
|
||||
)
|
||||
for title, cover in pattern3.findall(html):
|
||||
items.append({
|
||||
"vod_id": "",
|
||||
"vod_name": title.strip(),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": "",
|
||||
})
|
||||
return items
|
||||
|
||||
def _extract_player_url_from_path(self, path):
|
||||
"""从播放页路径提取真实m3u8 URL"""
|
||||
try:
|
||||
url = f"{HOST}{path}"
|
||||
r = self._get(url, timeout=15000)
|
||||
html = r.text
|
||||
m = re.search(r'player_aaaa\s*=\s*(\{[^}]+\})', html)
|
||||
if m:
|
||||
data = json.loads(m.group(1))
|
||||
return data.get("url", "")
|
||||
except:
|
||||
pass
|
||||
return ""
|
||||
|
||||
def _extract_player_data(self, path):
|
||||
"""从播放页路径提取完整播放数据(url, from等)"""
|
||||
try:
|
||||
url = f"{HOST}{path}"
|
||||
r = self._get(url, timeout=15000)
|
||||
html = r.text
|
||||
m = re.search(r'player_aaaa\s*=\s*(\{[^}]+\})', html)
|
||||
if m:
|
||||
data = json.loads(m.group(1))
|
||||
return data
|
||||
except:
|
||||
pass
|
||||
return None
|
||||
|
||||
def _get_from_source(self, vid, sid):
|
||||
"""获取某播放源的from字段(播放源类型)"""
|
||||
try:
|
||||
url = f"{HOST}/vodplay/{vid}-{sid}-1.html"
|
||||
r = self._get(url, timeout=10000)
|
||||
html = r.text
|
||||
m = re.search(r'player_aaaa\s*=\s*(\{[^}]+\})', html)
|
||||
if m:
|
||||
data = json.loads(m.group(1))
|
||||
return data.get("from", "")
|
||||
except:
|
||||
pass
|
||||
return ""
|
||||
|
||||
def _extract_player_url(self, vid, sid, nid):
|
||||
"""从播放页提取player_aaaa中的URL"""
|
||||
try:
|
||||
url = f"{HOST}/vodplay/{vid}-{sid}-{nid}.html"
|
||||
r = self._get(url, timeout=15000)
|
||||
html = r.text
|
||||
|
||||
# 匹配 player_aaaa={"url":"..."}
|
||||
m = re.search(r'player_aaaa\s*=\s*(\{[^}]+\})', html, re.DOTALL)
|
||||
if m:
|
||||
data = json.loads(m.group(1))
|
||||
play_url = data.get("url", "")
|
||||
return play_url
|
||||
except:
|
||||
pass
|
||||
return ""
|
||||
|
||||
def localProxy(self, param):
|
||||
pass
|
||||
@@ -0,0 +1,446 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""
|
||||
黄豆短剧爬虫
|
||||
站点: https://www.hdmgdj.com
|
||||
"""
|
||||
|
||||
import json
|
||||
import urllib.parse
|
||||
|
||||
import requests
|
||||
|
||||
try:
|
||||
from base.spider import Spider as BaseSpider
|
||||
except ImportError:
|
||||
class BaseSpider:
|
||||
pass
|
||||
|
||||
|
||||
class Spider(BaseSpider):
|
||||
"""黄豆短剧爬虫"""
|
||||
|
||||
BASE_URL = 'https://www.hdmgdj.com'
|
||||
API_BASE = 'https://hdmgdj.com/api'
|
||||
|
||||
HEADERS = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Accept': 'application/json, text/plain, */*',
|
||||
'Referer': 'https://www.hdmgdj.com/',
|
||||
'Origin': 'https://www.hdmgdj.com',
|
||||
}
|
||||
|
||||
_filter_cache = {} # 分类筛选缓存
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.name = ""
|
||||
self.error_play_url = "https://kjjsaas-sh.oss-cn-shanghai.aliyuncs.com/u/3401405881/20240818-936952-fc31b16575e80a7562cdb1f81a39c6b0.mp4"
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update(self.HEADERS)
|
||||
|
||||
# ==================== 标准接口 ====================
|
||||
|
||||
def init(self, extend="{}"):
|
||||
"""初始化"""
|
||||
if extend:
|
||||
try:
|
||||
self.extend = json.loads(extend)
|
||||
if 'name' in self.extend:
|
||||
self.name = self.extend['name']
|
||||
if 'base_url' in self.extend:
|
||||
self.BASE_URL = self.extend['base_url']
|
||||
self.API_BASE = self.extend['base_url'] + '/api'
|
||||
except Exception as e:
|
||||
print(e)
|
||||
return None
|
||||
|
||||
def getName(self):
|
||||
"""获取爬虫名称"""
|
||||
return "黄豆短剧"
|
||||
|
||||
def homeContent(self, filter):
|
||||
"""首页"""
|
||||
result = {
|
||||
"class": [],
|
||||
"filters": {},
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
try:
|
||||
# 获取分类
|
||||
genres_data = self._get('/genres')
|
||||
if genres_data and isinstance(genres_data, list):
|
||||
for g in genres_data:
|
||||
gid = str(g.get('id', ''))
|
||||
if not gid:
|
||||
continue
|
||||
result["class"].append({
|
||||
"type_id": gid,
|
||||
"type_name": g.get('name', ''),
|
||||
})
|
||||
|
||||
# 获取首页推荐
|
||||
home_data = self._get('/home')
|
||||
if home_data and isinstance(home_data, dict):
|
||||
# guess 猜你喜欢
|
||||
guess_list = home_data.get('guess', [])
|
||||
if isinstance(guess_list, list):
|
||||
for item in guess_list:
|
||||
result["list"].append(self._parse_vod(item))
|
||||
|
||||
# feature 精选
|
||||
feature_list = home_data.get('feature', [])
|
||||
if isinstance(feature_list, list):
|
||||
for item in feature_list:
|
||||
result["list"].append(self._parse_vod(item))
|
||||
|
||||
# 如果列表为空,用首页第一页数据
|
||||
if not result["list"]:
|
||||
dramas_data = self._get('/dramas?page=1&size=20')
|
||||
if dramas_data and isinstance(dramas_data, dict):
|
||||
for item in dramas_data.get('list', []):
|
||||
result["list"].append(self._parse_vod(item))
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
return result
|
||||
|
||||
def categoryContent(self, tid, pg, filter, extend):
|
||||
"""分类页"""
|
||||
result = {
|
||||
"page": pg,
|
||||
"pagecount": 999,
|
||||
"limit": 20,
|
||||
"total": 99999,
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
try:
|
||||
# 分类ID是genre id
|
||||
data = self._get(f'/dramas?genreId={tid}&page={pg}&size=20')
|
||||
if data and isinstance(data, dict):
|
||||
lst = data.get('list', [])
|
||||
total = data.get('total', 0)
|
||||
result["total"] = total
|
||||
result["pagecount"] = (total + 19) // 20 if total else 999
|
||||
for item in lst:
|
||||
result["list"].append(self._parse_vod(item))
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
return result
|
||||
|
||||
def detailContent(self, ids):
|
||||
"""详情页"""
|
||||
result = {
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
try:
|
||||
vid = ids[0]
|
||||
data = self._get(f'/dramas/{vid}')
|
||||
|
||||
if data and isinstance(data, dict):
|
||||
episodes = data.get('episodes', [])
|
||||
|
||||
# 组装播放地址
|
||||
play_url_parts = []
|
||||
for ep in episodes:
|
||||
ep_title = ep.get('title', f"第{ep.get('ep', 0)}集")
|
||||
play_url = ep.get('playUrl', '')
|
||||
if play_url:
|
||||
play_url_parts.append(f"{ep_title}${play_url}")
|
||||
|
||||
cover = data.get('cover', '')
|
||||
# 加密海报走本地代理解密
|
||||
if cover and ('encryptimages' in cover or '.bng' in cover):
|
||||
try:
|
||||
cover = f"{self.getProxyUrl()}&url={urllib.parse.quote(cover)}"
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
vod = {
|
||||
"vod_id": str(data['id']),
|
||||
"vod_name": data.get('t', ''),
|
||||
"vod_pic": cover,
|
||||
"type_name": data.get('sub', ''),
|
||||
"vod_year": '',
|
||||
"vod_area": '',
|
||||
"vod_remarks": f"{data.get('serial', '')}·{data.get('plays', '')}播放",
|
||||
"vod_actor": '',
|
||||
"vod_director": '未知',
|
||||
"vod_content": data.get('summary', '') or data.get('t', ''),
|
||||
"vod_play_from": '黄豆短剧',
|
||||
"vod_play_url": '#'.join(play_url_parts),
|
||||
}
|
||||
result["list"].append(vod)
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
return result
|
||||
|
||||
def searchContent(self, key, quick, pg="1"):
|
||||
"""搜索"""
|
||||
result = {
|
||||
"page": pg,
|
||||
"pagecount": 999,
|
||||
"limit": 20,
|
||||
"total": 99999,
|
||||
"list": [],
|
||||
"parse": 0,
|
||||
"jx": 0,
|
||||
}
|
||||
|
||||
try:
|
||||
data = self._get(f'/search?kw={urllib.parse.quote(key)}&page={pg}&size=20')
|
||||
if data and isinstance(data, dict):
|
||||
lst = data.get('list', [])
|
||||
total = data.get('total', 0)
|
||||
result["total"] = total
|
||||
result["pagecount"] = (total + 19) // 20 if total else 0
|
||||
for item in lst:
|
||||
result["list"].append(self._parse_vod(item))
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
return result
|
||||
|
||||
def playerContent(self, flag, id, vipFlags):
|
||||
"""播放页 - 直接返回 m3u8 data URI"""
|
||||
result = {
|
||||
"parse": 0,
|
||||
"playUrl": "",
|
||||
"url": self.error_play_url,
|
||||
"jx": 0,
|
||||
"header": "",
|
||||
}
|
||||
|
||||
if id:
|
||||
# 直接在 playerContent 里生成解密后的 m3u8,用 data URI 返回
|
||||
# 这样播放地址就不是 127.0.0.1 代理了
|
||||
m3u8_content = self._build_m3u8_with_key(id)
|
||||
if m3u8_content:
|
||||
import base64
|
||||
m3u8_b64 = base64.b64encode(m3u8_content.encode('utf-8')).decode('ascii')
|
||||
result["url"] = "data:application/vnd.apple.mpegurl;base64," + m3u8_b64
|
||||
result["parse"] = 0
|
||||
|
||||
return result
|
||||
|
||||
def _build_m3u8_with_key(self, url):
|
||||
"""构建 m3u8 内容(key 内嵌为 base64 data URI,ts 用原始绝对地址)"""
|
||||
import hashlib
|
||||
import re
|
||||
import base64
|
||||
if not url:
|
||||
return None
|
||||
|
||||
try:
|
||||
r = self.session.get(url, timeout=15, verify=False)
|
||||
content = r.text
|
||||
|
||||
# 计算 key
|
||||
key_bytes = self._get_key_bytes(url)
|
||||
if key_bytes:
|
||||
key_b64 = base64.b64encode(key_bytes).decode('ascii')
|
||||
key_data_uri = "data:text/plain;base64," + key_b64
|
||||
content = re.sub(
|
||||
r'(#EXT-X-KEY:.*?URI=")[^"]*(")',
|
||||
r'\1' + key_data_uri + r'\2',
|
||||
content
|
||||
)
|
||||
|
||||
# 把相对路径的 ts 改成绝对路径
|
||||
base_url = url.rsplit('/', 1)[0] + '/'
|
||||
lines = content.split('\n')
|
||||
new_lines = []
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'):
|
||||
if line.startswith('http'):
|
||||
new_lines.append(line)
|
||||
else:
|
||||
new_lines.append(base_url + line)
|
||||
else:
|
||||
new_lines.append(line)
|
||||
content = '\n'.join(new_lines)
|
||||
return content
|
||||
except Exception as e:
|
||||
print(f"_build_m3u8_with_key error: {e}")
|
||||
return None
|
||||
|
||||
def _get_key_bytes(self, url):
|
||||
"""从 m3u8 URL 计算解密 key"""
|
||||
import hashlib
|
||||
import re
|
||||
m = re.search(r'/hls/([0-9a-f]{64})/', url)
|
||||
if not m:
|
||||
return None
|
||||
video_id = m.group(1)
|
||||
ver_match = re.search(r'[?&]version=([^&#]+)', url)
|
||||
version = ver_match.group(1) if ver_match else 'v1'
|
||||
prefix = "xnaichanping"
|
||||
key_str = prefix + video_id + version
|
||||
return hashlib.md5(key_str.encode()).digest()
|
||||
|
||||
def localProxy(self, param):
|
||||
"""本地代理 - 解密海报图片"""
|
||||
try:
|
||||
url = param['url']
|
||||
r = self.session.get(url, timeout=15, verify=False)
|
||||
decrypted = self._aes_decrypt_img(r.content, url)
|
||||
# 确定图片类型
|
||||
content_type = "image/jpeg"
|
||||
if decrypted[:8] == b'\x89PNG\r\n\x1a\n':
|
||||
content_type = "image/png"
|
||||
elif decrypted[:6] in (b'GIF87a', b'GIF89a'):
|
||||
content_type = "image/gif"
|
||||
elif decrypted[:4] == b'RIFF' and decrypted[8:12] == b'WEBP':
|
||||
content_type = "image/webp"
|
||||
return [200, content_type, decrypted]
|
||||
except Exception as e:
|
||||
print(f"localProxy error: {e}")
|
||||
return [500, 'text/html', b'']
|
||||
|
||||
def _aes_decrypt_img(self, encrypted, url):
|
||||
"""AES 解密图片 - 网站自定义 CBC 算法"""
|
||||
import hashlib
|
||||
import re
|
||||
from Crypto.Cipher import AES
|
||||
|
||||
# 提取 imageId (64位哈希)
|
||||
m = re.search(r'([0-9a-f]{64})', url)
|
||||
if not m:
|
||||
return encrypted
|
||||
image_id = m.group(1)
|
||||
|
||||
# 提取 version
|
||||
ver_match = re.search(r'[?&]version=([^&#]+)', url)
|
||||
version = ver_match.group(1) if ver_match else 'v1'
|
||||
|
||||
# 计算解密 key
|
||||
prefix = "xnaichanping"
|
||||
key_str = prefix + image_id + version
|
||||
key_bytes = hashlib.md5(key_str.encode()).digest()
|
||||
|
||||
# 网站自定义 CBC 解密 (mC 函数)
|
||||
t = len(encrypted) // 16
|
||||
if t < 1:
|
||||
return encrypted
|
||||
|
||||
iv = bytes(16) # IV=0
|
||||
|
||||
# 取最后一块 XOR 16
|
||||
last_block = encrypted[(t - 1) * 16:t * 16]
|
||||
a = bytes([b ^ 16 for b in last_block])
|
||||
|
||||
# 加密 a
|
||||
cipher_enc = AES.new(key_bytes, AES.MODE_CBC, iv)
|
||||
o = cipher_enc.encrypt(a)[:16]
|
||||
|
||||
# 扩展密文并解密
|
||||
extended = encrypted + o
|
||||
cipher_dec = AES.new(key_bytes, AES.MODE_CBC, iv)
|
||||
c = cipher_dec.decrypt(extended)
|
||||
|
||||
# 自定义 CBC:每块 XOR 前一块密文
|
||||
u = bytearray(len(c))
|
||||
u[:16] = c[:16]
|
||||
for f in range(1, t):
|
||||
for h in range(16):
|
||||
u[f * 16 + h] = c[f * 16 + h] ^ encrypted[(f - 1) * 16 + h]
|
||||
|
||||
# 去掉 padding
|
||||
d = u[-1]
|
||||
if 1 <= d <= 16:
|
||||
u = u[:-d]
|
||||
|
||||
return bytes(u)
|
||||
|
||||
# ==================== 内部方法 ====================
|
||||
|
||||
def _parse_vod(self, item):
|
||||
"""解析视频条目 - 海报用 getProxyUrl 代理解密"""
|
||||
cover = item.get('cover', '')
|
||||
# 加密海报走本地代理解密
|
||||
if cover and ('encryptimages' in cover or '.bng' in cover):
|
||||
try:
|
||||
cover = f"{self.getProxyUrl()}&url={urllib.parse.quote(cover)}"
|
||||
except Exception:
|
||||
pass
|
||||
return {
|
||||
"vod_id": str(item['id']),
|
||||
"vod_name": item.get('t', ''),
|
||||
"vod_pic": cover,
|
||||
"vod_remarks": f"{item.get('serial', '')}·{item.get('eps', 0)}集",
|
||||
}
|
||||
|
||||
def _get(self, path):
|
||||
"""发送 GET 请求"""
|
||||
url = self.API_BASE + path
|
||||
try:
|
||||
r = self.session.get(url, timeout=15, verify=False)
|
||||
resp = r.json()
|
||||
if resp.get('code') == 0 and resp.get('data') is not None:
|
||||
return resp['data']
|
||||
return None
|
||||
except Exception as e:
|
||||
print(e)
|
||||
return None
|
||||
|
||||
|
||||
# 调试用
|
||||
if __name__ == '__main__':
|
||||
import urllib3
|
||||
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
||||
|
||||
s = Spider()
|
||||
s.init()
|
||||
|
||||
print('=== 首页 ===')
|
||||
home = s.homeContent(True)
|
||||
print(f'分类: {len(home["class"])}个')
|
||||
for c in home['class']:
|
||||
print(f' {c["type_id"]}: {c["type_name"]}')
|
||||
print(f'推荐: {len(home["list"])}个')
|
||||
for v in home['list'][:5]:
|
||||
print(f' {v["vod_id"]}: {v["vod_name"]} - {v["vod_remarks"]}')
|
||||
print()
|
||||
|
||||
print('=== 分类1(都市)第1页 ===')
|
||||
cr = s.categoryContent('1', 1, True, {})
|
||||
print(f'总数: {cr["total"]}, 本页: {len(cr["list"])}个')
|
||||
for v in cr['list'][:5]:
|
||||
print(f' {v["vod_id"]}: {v["vod_name"]}')
|
||||
print()
|
||||
|
||||
print('=== 搜索 穿越 ===')
|
||||
sr = s.searchContent('穿越', False, '1')
|
||||
print(f'结果: {len(sr["list"])}个, 总数: {sr["total"]}')
|
||||
for v in sr['list'][:5]:
|
||||
print(f' {v["vod_id"]}: {v["vod_name"]}')
|
||||
print()
|
||||
|
||||
if sr['list']:
|
||||
vid = sr['list'][0]['vod_id']
|
||||
print(f'=== 详情 {vid} ===')
|
||||
dr = s.detailContent([vid])
|
||||
if dr['list']:
|
||||
v = dr['list'][0]
|
||||
print(f'标题: {v["vod_name"]}')
|
||||
print(f'分类: {v["type_name"]}')
|
||||
print(f'备注: {v["vod_remarks"]}')
|
||||
print(f'播放源: {v["vod_play_from"]}')
|
||||
play_urls = v["vod_play_url"].split('#')
|
||||
print(f'集数: {len(play_urls)}集')
|
||||
print(f'第一集: {play_urls[0][:80]}...')
|
||||
Reference in New Issue
Block a user