两个BT
新韩剧网
This commit is contained in:
+142
-19
@@ -1,7 +1,7 @@
|
|||||||
import unittest
|
import unittest
|
||||||
from importlib.machinery import SourceFileLoader
|
from importlib.machinery import SourceFileLoader
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from unittest.mock import patch
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[1]
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
@@ -63,6 +63,31 @@ class TestLiangGeBTSpider(unittest.TestCase):
|
|||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_extract_cards_supports_current_play_cards(self):
|
||||||
|
html = """
|
||||||
|
<div class="relative group movie-card" data-vod-id="ch42nt5hw">
|
||||||
|
<a href="/play/ch42nt5hw" class="block">
|
||||||
|
<div class="relative">
|
||||||
|
<img data-src="https://img.example/poster.jpg" alt="夜魔侠:重生 第二季" />
|
||||||
|
<span>更新至6集</span>
|
||||||
|
</div>
|
||||||
|
<h3>夜魔侠:重生 第二季</h3>
|
||||||
|
</a>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
cards = self.spider._extract_cards(html)
|
||||||
|
self.assertEqual(
|
||||||
|
cards,
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"vod_id": "/play/ch42nt5hw",
|
||||||
|
"vod_name": "夜魔侠:重生 第二季",
|
||||||
|
"vod_pic": "https://img.example/poster.jpg",
|
||||||
|
"vod_remarks": "更新至6集",
|
||||||
|
}
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
@patch.object(Spider, "_request_html")
|
@patch.object(Spider, "_request_html")
|
||||||
def test_home_video_content_requests_home_page(self, mock_request_html):
|
def test_home_video_content_requests_home_page(self, mock_request_html):
|
||||||
mock_request_html.return_value = """
|
mock_request_html.return_value = """
|
||||||
@@ -80,53 +105,65 @@ class TestLiangGeBTSpider(unittest.TestCase):
|
|||||||
@patch.object(Spider, "_request_html")
|
@patch.object(Spider, "_request_html")
|
||||||
def test_category_content_builds_expected_urls(self, mock_request_html):
|
def test_category_content_builds_expected_urls(self, mock_request_html):
|
||||||
mock_request_html.return_value = """
|
mock_request_html.return_value = """
|
||||||
<li>
|
<div class="movie-card" data-vod-id="ch42nt5hw">
|
||||||
<a href="/movie/301.html" title="分类影片">
|
<a href="/play/ch42nt5hw">
|
||||||
<img src="/cate.jpg" />
|
<img data-src="/cate.jpg" alt="分类影片" />
|
||||||
|
<span>更新至6集</span>
|
||||||
|
<h3>分类影片</h3>
|
||||||
</a>
|
</a>
|
||||||
<h3>分类影片</h3>
|
</div>
|
||||||
<span class="rating">HD</span>
|
|
||||||
</li>
|
|
||||||
"""
|
"""
|
||||||
page_one = self.spider.categoryContent("meiju", "1", False, {})
|
page_one = self.spider.categoryContent("meiju", "1", False, {})
|
||||||
page_three = self.spider.categoryContent("movie_bt_tags/xiju", "3", False, {})
|
page_three = self.spider.categoryContent("movie_bt_tags/xiju", "3", False, {})
|
||||||
self.assertEqual(mock_request_html.call_args_list[0].args[0], "https://www.bttwoo.com/meiju")
|
self.assertEqual(
|
||||||
|
mock_request_html.call_args_list[0].args[0],
|
||||||
|
"https://www.bttwoo.com/filter?classify=2&tvclasses=21",
|
||||||
|
)
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
mock_request_html.call_args_list[1].args[0],
|
mock_request_html.call_args_list[1].args[0],
|
||||||
"https://www.bttwoo.com/movie_bt_tags/xiju?paged=3",
|
"https://www.bttwoo.com/filter?classify=1&types=5&page=3",
|
||||||
)
|
)
|
||||||
self.assertEqual(page_one["page"], 1)
|
self.assertEqual(page_one["page"], 1)
|
||||||
self.assertEqual(page_one["list"][0]["vod_name"], "分类影片")
|
self.assertEqual(page_one["list"][0]["vod_name"], "分类影片")
|
||||||
self.assertNotIn("pagecount", page_one)
|
self.assertNotIn("pagecount", page_one)
|
||||||
self.assertEqual(page_three["page"], 3)
|
self.assertEqual(page_three["page"], 3)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_category_content_maps_high_score_movies(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = ""
|
||||||
|
self.spider.categoryContent("gf", "2", False, {})
|
||||||
|
self.assertEqual(
|
||||||
|
mock_request_html.call_args.args[0],
|
||||||
|
"https://www.bttwoo.com/filter?classify=1&sort_by=score&order=desc&page=2",
|
||||||
|
)
|
||||||
|
|
||||||
@patch.object(Spider, "_request_html")
|
@patch.object(Spider, "_request_html")
|
||||||
def test_search_content_builds_query_and_filters_irrelevant_results(self, mock_request_html):
|
def test_search_content_builds_query_and_filters_irrelevant_results(self, mock_request_html):
|
||||||
mock_request_html.return_value = """
|
mock_request_html.return_value = """
|
||||||
<li>
|
<div class="movie-card" data-vod-id="ch-search-1">
|
||||||
<a href="/movie/401.html" title="繁花">
|
<a href="/play/ch-search-1" title="繁花">
|
||||||
<img src="/match.jpg" />
|
<img src="/match.jpg" />
|
||||||
|
<h3>繁花</h3>
|
||||||
</a>
|
</a>
|
||||||
<h3>繁花</h3>
|
</div>
|
||||||
</li>
|
<div class="movie-card" data-vod-id="ch-search-2">
|
||||||
<li>
|
<a href="/play/ch-search-2" title="无关结果">
|
||||||
<a href="/movie/402.html" title="无关结果">
|
|
||||||
<img src="/other.jpg" />
|
<img src="/other.jpg" />
|
||||||
|
<h3>无关结果</h3>
|
||||||
</a>
|
</a>
|
||||||
<h3>无关结果</h3>
|
</div>
|
||||||
</li>
|
|
||||||
"""
|
"""
|
||||||
result = self.spider.searchContent("繁花", False, "2")
|
result = self.spider.searchContent("繁花", False, "2")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
mock_request_html.call_args.args[0],
|
mock_request_html.call_args.args[0],
|
||||||
"https://www.bttwoo.com/xssssearch?q=%E7%B9%81%E8%8A%B1&p=2",
|
"https://www.bttwoo.com/search?q=%E7%B9%81%E8%8A%B1&page=2",
|
||||||
)
|
)
|
||||||
self.assertEqual(result["page"], 2)
|
self.assertEqual(result["page"], 2)
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
result["list"],
|
result["list"],
|
||||||
[
|
[
|
||||||
{
|
{
|
||||||
"vod_id": "401",
|
"vod_id": "/play/ch-search-1",
|
||||||
"vod_name": "繁花",
|
"vod_name": "繁花",
|
||||||
"vod_pic": "https://www.bttwoo.com/match.jpg",
|
"vod_pic": "https://www.bttwoo.com/match.jpg",
|
||||||
"vod_remarks": "",
|
"vod_remarks": "",
|
||||||
@@ -169,6 +206,45 @@ class TestLiangGeBTSpider(unittest.TestCase):
|
|||||||
self.assertEqual(self.spider._decode_play_id(first_id)["pid"], "play-1")
|
self.assertEqual(self.spider._decode_play_id(first_id)["pid"], "play-1")
|
||||||
self.assertEqual(self.spider._decode_play_id(first_id)["sid"], "900")
|
self.assertEqual(self.spider._decode_play_id(first_id)["sid"], "900")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_detail_content_supports_current_play_pages(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = """
|
||||||
|
<html>
|
||||||
|
<head>
|
||||||
|
<title>夜魔侠:重生 第二季 - 第1集 -两个BT影视</title>
|
||||||
|
<meta name="description" content="夜幕降临,魔影共舞。" />
|
||||||
|
<meta property="og:image" content="https://img.example/poster.jpg" />
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class="movie-poster">
|
||||||
|
<img src="https://img.example/poster.jpg" />
|
||||||
|
<h1>夜魔侠:重生 第二季</h1>
|
||||||
|
</div>
|
||||||
|
<div class="grid grid-cols-3 gap-2 text-xs">
|
||||||
|
<div>导演</div><div>贾斯汀·本森 / 艾伦·穆尔黑德</div>
|
||||||
|
<div>主演</div><div>查理·考克斯 / 文森特·多诺费奥</div>
|
||||||
|
</div>
|
||||||
|
<p>夜幕降临,魔影共舞。</p>
|
||||||
|
<div x-data="episodeManager(1, 1, [{ lineName: '线路一', episodeCount: 2 }])">
|
||||||
|
<a href="/play/ch42nt5hw">1</a>
|
||||||
|
<a href="/play/ch42nt5ic">2</a>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
|
"""
|
||||||
|
result = self.spider.detailContent(["/play/ch42nt5hw"])
|
||||||
|
vod = result["list"][0]
|
||||||
|
first_name, first_id = vod["vod_play_url"].split("#")[0].split("$", 1)
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.bttwoo.com/play/ch42nt5hw")
|
||||||
|
self.assertEqual(vod["vod_id"], "/play/ch42nt5hw")
|
||||||
|
self.assertEqual(vod["vod_name"], "夜魔侠:重生 第二季")
|
||||||
|
self.assertEqual(vod["vod_pic"], "https://img.example/poster.jpg")
|
||||||
|
self.assertEqual(vod["vod_content"], "夜幕降临,魔影共舞。")
|
||||||
|
self.assertEqual(vod["vod_actor"], "查理·考克斯 / 文森特·多诺费奥")
|
||||||
|
self.assertEqual(vod["vod_director"], "贾斯汀·本森 / 艾伦·穆尔黑德")
|
||||||
|
self.assertEqual(first_name, "1")
|
||||||
|
self.assertEqual(self.spider._decode_play_id(first_id)["pid"], "/play/ch42nt5hw")
|
||||||
|
|
||||||
def test_player_content_passthroughs_direct_media_url(self):
|
def test_player_content_passthroughs_direct_media_url(self):
|
||||||
result = self.spider.playerContent("两个BT", "https://media.example/direct.m3u8", {})
|
result = self.spider.playerContent("两个BT", "https://media.example/direct.m3u8", {})
|
||||||
self.assertEqual(result["parse"], 0)
|
self.assertEqual(result["parse"], 0)
|
||||||
@@ -176,6 +252,53 @@ class TestLiangGeBTSpider(unittest.TestCase):
|
|||||||
self.assertEqual(result["url"], "https://media.example/direct.m3u8")
|
self.assertEqual(result["url"], "https://media.example/direct.m3u8")
|
||||||
self.assertEqual(result["header"]["Referer"], "https://www.bttwoo.com/")
|
self.assertEqual(result["header"]["Referer"], "https://www.bttwoo.com/")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_player_content_supports_current_play_page_ids(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = "<html><body>empty</body></html>"
|
||||||
|
play_id = self.spider._encode_play_id("/play/ch42nt5hw", "/play/ch42nt5hw", "1")
|
||||||
|
result = self.spider.playerContent("两个BT", play_id, {})
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.bttwoo.com/play/ch42nt5hw")
|
||||||
|
self.assertEqual(result["parse"], 1)
|
||||||
|
self.assertEqual(result["jx"], 1)
|
||||||
|
self.assertEqual(result["url"], "https://www.bttwoo.com/play/ch42nt5hw")
|
||||||
|
|
||||||
|
@patch.object(Spider, "fetch")
|
||||||
|
@patch.object(Spider, "_build_wasm_play_api_url")
|
||||||
|
@patch.object(Spider, "_cache_wasm_assets")
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_player_content_resolves_current_play_page_via_play_api(
|
||||||
|
self, mock_request_html, mock_cache_wasm_assets, mock_build_wasm_play_api_url, mock_fetch
|
||||||
|
):
|
||||||
|
mock_request_html.return_value = """
|
||||||
|
<html>
|
||||||
|
<body>
|
||||||
|
<nav x-data="{isLoggedIn: false, userlink:'X1VaWEZeVwEFCw4FCgc7IUlfXUhe'}"></nav>
|
||||||
|
<div id="player-error-message-detail" data-v-id="4810"></div>
|
||||||
|
<div x-data="episodeManager(1, 1, [{ lineName: 'alists', episodeCount: 22 }])">
|
||||||
|
<a href="/play/ch440i68t" data-line="1" data-episode="1" dataid="33373">1</a>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
|
"""
|
||||||
|
mock_build_wasm_play_api_url.return_value = "https://www.bttwoo.com/video/play?p=33373"
|
||||||
|
mock_response = MagicMock()
|
||||||
|
mock_response.status_code = 200
|
||||||
|
mock_response.text = (
|
||||||
|
'{"code":200,"data":{"quality_urls":[{"url":"https://media.example/final.m3u8"}],"current_quality":0}}'
|
||||||
|
)
|
||||||
|
mock_fetch.return_value = mock_response
|
||||||
|
play_id = self.spider._encode_play_id("/play/ch440i68t", "/play/ch440i68t", "1")
|
||||||
|
result = self.spider.playerContent("两个BT", play_id, {})
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.bttwoo.com/play/ch440i68t")
|
||||||
|
self.assertEqual(
|
||||||
|
mock_build_wasm_play_api_url.call_args.args,
|
||||||
|
("33373", "ch440i68t", "1080", "X1VaWEZeVwEFCw4FCgc7IUlfXUhe"),
|
||||||
|
)
|
||||||
|
self.assertEqual(result["parse"], 0)
|
||||||
|
self.assertEqual(result["jx"], 0)
|
||||||
|
self.assertEqual(result["url"], "https://media.example/final.m3u8")
|
||||||
|
self.assertEqual(result["header"]["Referer"], "https://www.bttwoo.com/play/ch440i68t")
|
||||||
|
|
||||||
@patch.object(Spider, "_request_html")
|
@patch.object(Spider, "_request_html")
|
||||||
def test_player_content_extracts_media_from_play_page(self, mock_request_html):
|
def test_player_content_extracts_media_from_play_page(self, mock_request_html):
|
||||||
mock_request_html.return_value = """
|
mock_request_html.return_value = """
|
||||||
|
|||||||
@@ -0,0 +1,235 @@
|
|||||||
|
import base64
|
||||||
|
import unittest
|
||||||
|
from importlib.machinery import SourceFileLoader
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from Crypto.Cipher import AES
|
||||||
|
from Crypto.Util.Padding import pad
|
||||||
|
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
MODULE = SourceFileLoader("hanju7_spider", str(ROOT / "新韩剧网.py")).load_module()
|
||||||
|
Spider = MODULE.Spider
|
||||||
|
|
||||||
|
|
||||||
|
def encrypt_player_payload(url):
|
||||||
|
iv = b"0123456789abcdef"
|
||||||
|
key = MODULE.PLAYER_AES_KEY.encode("utf-8")
|
||||||
|
cipher = AES.new(key, AES.MODE_CBC, iv)
|
||||||
|
body = cipher.encrypt(pad(url.encode("utf-8"), AES.block_size))
|
||||||
|
return base64.b64encode(iv + body).decode("utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
HOME_HTML = """
|
||||||
|
<div class="list">
|
||||||
|
<ul>
|
||||||
|
<li>
|
||||||
|
<a href="/detail/abc123.html" title="苦尽柑来遇见你" data-original="//img.example.com/a.jpg">苦尽柑来遇见你</a>
|
||||||
|
<span>更新至第2集</span>
|
||||||
|
</li>
|
||||||
|
<li>
|
||||||
|
<a href="/detail/def456.html" data-original="/upload/b.jpg">机智住院医生生活</a>
|
||||||
|
<span>2026-04-24</span>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
LIST_HTML = """
|
||||||
|
<div class="list">
|
||||||
|
<ul>
|
||||||
|
<li>
|
||||||
|
<a class="tu" href="/detail/xyz001.html" title="电影A" data-original="//img.example.com/m1.jpg"></a>
|
||||||
|
<span class="tip">HD</span>
|
||||||
|
</li>
|
||||||
|
<li>
|
||||||
|
<a class="tu" href="/detail/xyz002.html" title="电影B"></a>
|
||||||
|
<span class="tip">完结</span>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
SEARCH_HTML = """
|
||||||
|
<div class="txt">
|
||||||
|
<ul>
|
||||||
|
<li id="t">header</li>
|
||||||
|
<li>
|
||||||
|
<p id="name"><a href="/detail/search001.html">新乌托邦(12)</a></p>
|
||||||
|
<p id="actor">朴正民 / 金智秀</p>
|
||||||
|
</li>
|
||||||
|
<li>
|
||||||
|
<p id="name"><a href="/detail/search002.html">协商的技术</a></p>
|
||||||
|
<p id="actor">李帝勋</p>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
DETAIL_HTML = """
|
||||||
|
<div class="detail">
|
||||||
|
<div class="pic"><img data-original="//img.example.com/poster.jpg" /></div>
|
||||||
|
<div class="info">
|
||||||
|
<dl><dd>比天堂还美丽</dd></dl>
|
||||||
|
<dl><dd>金惠子 / 孙锡久</dd></dl>
|
||||||
|
<dl><dd>韩国</dd></dl>
|
||||||
|
<dl><dd>剧情</dd></dl>
|
||||||
|
<dl><dd>更新至第3集</dd></dl>
|
||||||
|
<dl><dd>2026</dd></dl>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div class="juqing">一段剧情简介</div>
|
||||||
|
<div id="playlist">新韩剧线路</div>
|
||||||
|
<div class="play">
|
||||||
|
<ul>
|
||||||
|
<li><a onclick="bf('p001')">第1集</a></li>
|
||||||
|
<li><a onclick="bf('p002')">第2集</a></li>
|
||||||
|
</ul>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
HOT_HTML = """
|
||||||
|
<div class="txt">
|
||||||
|
<ul>
|
||||||
|
<li><a href="/detail/h1.html" title="热播1">热播1</a><span>Top1</span></li>
|
||||||
|
<li><a href="/detail/h2.html" title="热播2">热播2</a><span>Top2</span></li>
|
||||||
|
<li><a href="/detail/h3.html" title="热播3">热播3</a><span>Top3</span></li>
|
||||||
|
</ul>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
class FakeResponse:
|
||||||
|
def __init__(self, text="", status_code=200, headers=None):
|
||||||
|
self.text = text
|
||||||
|
self.status_code = status_code
|
||||||
|
self.headers = headers or {}
|
||||||
|
self.encoding = "utf-8"
|
||||||
|
|
||||||
|
|
||||||
|
class TestHanJu7Spider(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
Spider._instance = None
|
||||||
|
self.spider = Spider()
|
||||||
|
self.spider.init()
|
||||||
|
|
||||||
|
def test_home_content_returns_expected_classes(self):
|
||||||
|
content = self.spider.homeContent(False)
|
||||||
|
self.assertEqual(
|
||||||
|
[item["type_id"] for item in content["class"]],
|
||||||
|
["1", "3", "4", "hot", "new"],
|
||||||
|
)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_home_video_content_parses_list_cards(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = HOME_HTML
|
||||||
|
result = self.spider.homeVideoContent()
|
||||||
|
self.assertEqual(
|
||||||
|
result["list"],
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"vod_id": "abc123",
|
||||||
|
"vod_name": "苦尽柑来遇见你",
|
||||||
|
"vod_pic": "https://img.example.com/a.jpg",
|
||||||
|
"vod_remarks": "更新至第2集",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"vod_id": "def456",
|
||||||
|
"vod_name": "机智住院医生生活",
|
||||||
|
"vod_pic": "https://www.hanju7.com/upload/b.jpg",
|
||||||
|
"vod_remarks": "2026-04-24",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_category_content_uses_list_path_for_regular_category(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = LIST_HTML
|
||||||
|
result = self.spider.categoryContent("3", "2", False, {})
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.hanju7.com/list/3---1.html")
|
||||||
|
self.assertEqual(result["page"], 2)
|
||||||
|
self.assertEqual(result["list"][0]["vod_id"], "xyz001")
|
||||||
|
self.assertEqual(result["list"][1]["vod_pic"], "https://pics.hanju7.com/pics/xyz002.jpg")
|
||||||
|
self.assertNotIn("pagecount", result)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_category_content_paginates_hot_locally(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = HOT_HTML
|
||||||
|
result = self.spider.categoryContent("hot", "1", False, {})
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.hanju7.com/hot.html")
|
||||||
|
self.assertEqual(result["page"], 1)
|
||||||
|
self.assertEqual(result["total"], 3)
|
||||||
|
self.assertEqual(result["list"][0]["vod_id"], "h1")
|
||||||
|
|
||||||
|
def test_extract_redirect_location_from_native_search_response(self):
|
||||||
|
response = FakeResponse(status_code=302, headers={"Location": "/search.php?searchword=test"})
|
||||||
|
self.assertEqual(self.spider._extract_redirect_location(response), "/search.php?searchword=test")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
@patch.object(Spider, "_native_post_search")
|
||||||
|
def test_search_content_follows_redirect_and_parses_results(self, mock_native_search, mock_request_html):
|
||||||
|
mock_native_search.return_value = ("/search.php?page=1&searchword=%E6%96%B0%E4%B9%8C%E6%89%98%E9%82%A6", "k=v")
|
||||||
|
mock_request_html.return_value = SEARCH_HTML
|
||||||
|
result = self.spider.searchContent("新乌托邦", False, "1")
|
||||||
|
self.assertEqual(result["page"], 1)
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.hanju7.com/search.php?page=1&searchword=%E6%96%B0%E4%B9%8C%E6%89%98%E9%82%A6")
|
||||||
|
self.assertEqual(mock_request_html.call_args.kwargs["headers"]["Cookie"], "k=v")
|
||||||
|
self.assertEqual(
|
||||||
|
result["list"],
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"vod_id": "search001",
|
||||||
|
"vod_name": "新乌托邦",
|
||||||
|
"vod_pic": MODULE.DEFAULT_PIC,
|
||||||
|
"vod_remarks": "朴正民 / 金智秀",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"vod_id": "search002",
|
||||||
|
"vod_name": "协商的技术",
|
||||||
|
"vod_pic": MODULE.DEFAULT_PIC,
|
||||||
|
"vod_remarks": "李帝勋",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_detail_content_parses_metadata_and_playlists(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = DETAIL_HTML
|
||||||
|
result = self.spider.detailContent(["abc123"])
|
||||||
|
vod = result["list"][0]
|
||||||
|
self.assertEqual(mock_request_html.call_args.args[0], "https://www.hanju7.com/detail/abc123.html")
|
||||||
|
self.assertEqual(vod["vod_name"], "比天堂还美丽")
|
||||||
|
self.assertEqual(vod["vod_pic"], "https://img.example.com/poster.jpg")
|
||||||
|
self.assertEqual(vod["vod_actor"], "金惠子 / 孙锡久")
|
||||||
|
self.assertEqual(vod["vod_remarks"], "更新至第3集")
|
||||||
|
self.assertEqual(vod["vod_year"], "2026")
|
||||||
|
self.assertEqual(vod["vod_content"], "一段剧情简介")
|
||||||
|
self.assertEqual(vod["vod_play_from"], "新韩剧线路")
|
||||||
|
self.assertEqual(vod["vod_play_url"], "第1集$p001#第2集$p002")
|
||||||
|
|
||||||
|
def test_decrypt_play_url_decodes_prefixed_iv_payload(self):
|
||||||
|
encrypted = encrypt_player_payload("https://cdn.example.com/final.m3u8")
|
||||||
|
self.assertEqual(self.spider._decrypt_play_url(encrypted), "https://cdn.example.com/final.m3u8")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_text")
|
||||||
|
def test_player_content_returns_direct_media(self, mock_request_text):
|
||||||
|
mock_request_text.return_value = encrypt_player_payload("https://cdn.example.com/final.m3u8")
|
||||||
|
result = self.spider.playerContent("新韩剧线路", "p001", {})
|
||||||
|
self.assertEqual(mock_request_text.call_args.args[0], "https://www.hanju7.com/u/u1.php?ud=p001")
|
||||||
|
self.assertEqual(result["parse"], 0)
|
||||||
|
self.assertEqual(result["jx"], 0)
|
||||||
|
self.assertEqual(result["url"], "https://cdn.example.com/final.m3u8")
|
||||||
|
self.assertEqual(result["header"]["Referer"], "https://www.hanju7.com/")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_text")
|
||||||
|
def test_player_content_falls_back_to_parser_for_embed_url(self, mock_request_text):
|
||||||
|
mock_request_text.return_value = encrypt_player_payload("https://player.example.com/embed?id=1")
|
||||||
|
result = self.spider.playerContent("新韩剧线路", "p002", {})
|
||||||
|
self.assertEqual(result["parse"], 1)
|
||||||
|
self.assertEqual(result["jx"], 1)
|
||||||
|
self.assertEqual(result["url"], "https://player.example.com/embed?id=1")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
+240
-18
@@ -1,9 +1,13 @@
|
|||||||
# coding=utf-8
|
# coding=utf-8
|
||||||
import base64
|
import base64
|
||||||
|
import hashlib
|
||||||
import json
|
import json
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
from urllib.parse import quote, urljoin
|
import tempfile
|
||||||
|
from urllib.parse import quote, urlencode, urljoin
|
||||||
|
|
||||||
from lxml import html as lxml_html
|
from lxml import html as lxml_html
|
||||||
|
|
||||||
@@ -16,6 +20,7 @@ class Spider(BaseSpider):
|
|||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.name = "两个BT"
|
self.name = "两个BT"
|
||||||
self.host = "https://www.bttwoo.com"
|
self.host = "https://www.bttwoo.com"
|
||||||
|
self._wasm_asset_cache = {}
|
||||||
self.headers = {
|
self.headers = {
|
||||||
"User-Agent": (
|
"User-Agent": (
|
||||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||||
@@ -64,17 +69,17 @@ class Spider(BaseSpider):
|
|||||||
keyword = self._clean_text(key)
|
keyword = self._clean_text(key)
|
||||||
if not keyword:
|
if not keyword:
|
||||||
return {"page": page, "limit": 0, "total": 0, "list": []}
|
return {"page": page, "limit": 0, "total": 0, "list": []}
|
||||||
url = self.host + f"/xssssearch?q={quote(keyword)}"
|
url = self.host + f"/search?q={quote(keyword)}"
|
||||||
if page > 1:
|
if page > 1:
|
||||||
url += f"&p={page}"
|
url += f"&page={page}"
|
||||||
items = self._extract_cards(self._request_html(url), keyword=keyword)
|
items = self._extract_cards(self._request_html(url), keyword=keyword)
|
||||||
return {"page": page, "limit": len(items), "total": len(items), "list": items}
|
return {"page": page, "limit": len(items), "total": len(items), "list": items}
|
||||||
|
|
||||||
def detailContent(self, ids):
|
def detailContent(self, ids):
|
||||||
vod_id = str(ids[0] if isinstance(ids, list) and ids else ids or "").strip()
|
vod_id = self._normalize_vod_id(ids[0] if isinstance(ids, list) and ids else ids)
|
||||||
if not vod_id:
|
if not vod_id:
|
||||||
return {"list": []}
|
return {"list": []}
|
||||||
html = self._request_html(self.host + f"/movie/{vod_id}.html")
|
html = self._request_html(self._build_detail_url(vod_id))
|
||||||
detail = self._parse_detail(html, vod_id)
|
detail = self._parse_detail(html, vod_id)
|
||||||
return {"list": [detail]} if detail else {"list": []}
|
return {"list": [detail]} if detail else {"list": []}
|
||||||
|
|
||||||
@@ -85,7 +90,13 @@ class Spider(BaseSpider):
|
|||||||
|
|
||||||
meta = self._decode_play_id(play_id)
|
meta = self._decode_play_id(play_id)
|
||||||
pid = meta.get("pid") or play_id
|
pid = meta.get("pid") or play_id
|
||||||
play_page_url = self.host + f"/v_play/{pid}.html"
|
play_path = self._extract_play_path(pid)
|
||||||
|
if str(pid or "").startswith(("http://", "https://")):
|
||||||
|
play_page_url = str(pid)
|
||||||
|
elif play_path:
|
||||||
|
play_page_url = self._abs_url(play_path)
|
||||||
|
else:
|
||||||
|
play_page_url = self.host + f"/v_play/{pid}.html"
|
||||||
html = self._request_html(play_page_url, referer=self.host + "/")
|
html = self._request_html(play_page_url, referer=self.host + "/")
|
||||||
media_url = self._extract_media_url(html)
|
media_url = self._extract_media_url(html)
|
||||||
if media_url:
|
if media_url:
|
||||||
@@ -98,6 +109,17 @@ class Spider(BaseSpider):
|
|||||||
if iframe_media_url:
|
if iframe_media_url:
|
||||||
return self._build_player_result(iframe_media_url, iframe_url)
|
return self._build_player_result(iframe_media_url, iframe_url)
|
||||||
|
|
||||||
|
if play_path:
|
||||||
|
self._cache_wasm_assets(html)
|
||||||
|
dataid = self._extract_play_dataid(html, play_page_url)
|
||||||
|
secret_key = play_path.rsplit("/", 1)[-1]
|
||||||
|
userlink = self._extract_userlink(html) or "0"
|
||||||
|
api_url = self._build_wasm_play_api_url(dataid, secret_key, "1080", userlink)
|
||||||
|
api_data = self._request_json(api_url, referer=play_page_url)
|
||||||
|
media_url = self._extract_media_from_play_api(api_data)
|
||||||
|
if media_url:
|
||||||
|
return self._build_player_result(media_url, play_page_url)
|
||||||
|
|
||||||
return self._build_parse_result(play_page_url, play_page_url)
|
return self._build_parse_result(play_page_url, play_page_url)
|
||||||
|
|
||||||
def _request_html(self, url, referer=None):
|
def _request_html(self, url, referer=None):
|
||||||
@@ -113,9 +135,30 @@ class Spider(BaseSpider):
|
|||||||
|
|
||||||
def _build_category_url(self, tid, page):
|
def _build_category_url(self, tid, page):
|
||||||
path = str(tid or "").strip().lstrip("/")
|
path = str(tid or "").strip().lstrip("/")
|
||||||
|
filters = {
|
||||||
|
"zgjun": {"classify": "2", "tvclasses": "20"},
|
||||||
|
"meiju": {"classify": "2", "tvclasses": "21"},
|
||||||
|
"jpsrtv": {"classify": "2", "tvclasses": "22"},
|
||||||
|
"movie_bt_tags/xiju": {"classify": "1", "types": "5"},
|
||||||
|
"movie_bt_tags/aiqing": {"classify": "1", "types": "6"},
|
||||||
|
"movie_bt_tags/adt": {"classify": "1", "types": "18"},
|
||||||
|
"movie_bt_tags/at": {"classify": "1", "types": "10"},
|
||||||
|
"movie_bt_tags/donghua": {"classify": "1", "types": "11"},
|
||||||
|
"movie_bt_tags/qihuan": {"classify": "1", "types": "12"},
|
||||||
|
"movie_bt_tags/xuanni": {"classify": "1", "types": "2"},
|
||||||
|
"movie_bt_tags/kehuan": {"classify": "1", "types": "14"},
|
||||||
|
"movie_bt_tags/juqing": {"classify": "1", "types": "1"},
|
||||||
|
"movie_bt_tags/kongbu": {"classify": "1", "types": "3"},
|
||||||
|
"gf": {"classify": "1", "sort_by": "score", "order": "desc"},
|
||||||
|
}.get(path)
|
||||||
|
if filters:
|
||||||
|
params = dict(filters)
|
||||||
|
if page > 1:
|
||||||
|
params["page"] = str(page)
|
||||||
|
return self.host + "/filter?" + urlencode(params)
|
||||||
url = self.host + "/" + path
|
url = self.host + "/" + path
|
||||||
if page > 1:
|
if page > 1:
|
||||||
url += f"?paged={page}"
|
url += f"?page={page}"
|
||||||
return url
|
return url
|
||||||
|
|
||||||
def _extract_cards(self, html, keyword=None):
|
def _extract_cards(self, html, keyword=None):
|
||||||
@@ -124,13 +167,21 @@ class Spider(BaseSpider):
|
|||||||
return []
|
return []
|
||||||
results = []
|
results = []
|
||||||
seen = set()
|
seen = set()
|
||||||
for node in root.xpath("//li[.//a[contains(@href,'/movie/')]]"):
|
nodes = root.xpath(
|
||||||
href = self._first_attr(node, ".//a[contains(@href,'/movie/')][1]", "href")
|
"//*[contains(concat(' ', normalize-space(@class), ' '), ' movie-card ')]"
|
||||||
vod_id = self._extract_vod_id(href)
|
"|//li[.//a[contains(@href,'/movie/')]]"
|
||||||
|
)
|
||||||
|
for node in nodes:
|
||||||
|
href = (
|
||||||
|
self._first_attr(node, ".//a[contains(@href,'/play/')][1]", "href")
|
||||||
|
or self._first_attr(node, ".//a[contains(@href,'/movie/')][1]", "href")
|
||||||
|
)
|
||||||
|
vod_id = self._extract_card_id(href)
|
||||||
title = (
|
title = (
|
||||||
self._first_text(node, ".//h3//a[1]")
|
self._first_text(node, ".//h3//a[1]")
|
||||||
or self._first_text(node, ".//h3[1]")
|
or self._first_text(node, ".//h3[1]")
|
||||||
or self._clean_text(self._first_attr(node, ".//a[@title][1]", "title"))
|
or self._clean_text(self._first_attr(node, ".//a[@title][1]", "title"))
|
||||||
|
or self._clean_text(self._first_attr(node, ".//img[@alt][1]", "alt"))
|
||||||
or self._first_text(node, ".//*[contains(@class,'title')][1]")
|
or self._first_text(node, ".//*[contains(@class,'title')][1]")
|
||||||
or self._first_text(node, ".//*[contains(@class,'name')][1]")
|
or self._first_text(node, ".//*[contains(@class,'name')][1]")
|
||||||
)
|
)
|
||||||
@@ -146,6 +197,9 @@ class Spider(BaseSpider):
|
|||||||
remarks = (
|
remarks = (
|
||||||
self._first_text(node, ".//*[contains(@class,'rating')][1]")
|
self._first_text(node, ".//*[contains(@class,'rating')][1]")
|
||||||
or self._first_text(node, ".//*[contains(@class,'status')][1]")
|
or self._first_text(node, ".//*[contains(@class,'status')][1]")
|
||||||
|
or self._first_text(node, ".//span[contains(text(),'集')][1]")
|
||||||
|
or self._first_text(node, ".//span[contains(text(),'HD')][1]")
|
||||||
|
or self._first_text(node, ".//span[contains(text(),'4k')][1]")
|
||||||
)
|
)
|
||||||
seen.add(vod_id)
|
seen.add(vod_id)
|
||||||
results.append(
|
results.append(
|
||||||
@@ -177,39 +231,76 @@ class Spider(BaseSpider):
|
|||||||
matched = re.search(r"/movie/(\d+)\.html", str(href or "").strip())
|
matched = re.search(r"/movie/(\d+)\.html", str(href or "").strip())
|
||||||
return matched.group(1) if matched else ""
|
return matched.group(1) if matched else ""
|
||||||
|
|
||||||
|
def _extract_play_path(self, href):
|
||||||
|
raw = str(href or "").strip()
|
||||||
|
if "/play/" not in raw:
|
||||||
|
return ""
|
||||||
|
path = raw[raw.find("/play/") :]
|
||||||
|
return path.split("?", 1)[0].split("#", 1)[0]
|
||||||
|
|
||||||
|
def _extract_card_id(self, href):
|
||||||
|
return self._extract_play_path(href) or self._extract_vod_id(href)
|
||||||
|
|
||||||
def _extract_play_pid(self, href):
|
def _extract_play_pid(self, href):
|
||||||
matched = re.search(r"/v_play/([^.]+)\.html", str(href or "").strip())
|
matched = re.search(r"/v_play/([^.]+)\.html", str(href or "").strip())
|
||||||
return matched.group(1) if matched else ""
|
return matched.group(1) if matched else ""
|
||||||
|
|
||||||
|
def _normalize_vod_id(self, value):
|
||||||
|
raw = str(value or "").strip()
|
||||||
|
return self._extract_play_path(raw) or raw
|
||||||
|
|
||||||
|
def _build_detail_url(self, vod_id):
|
||||||
|
raw = str(vod_id or "").strip()
|
||||||
|
if not raw:
|
||||||
|
return ""
|
||||||
|
if raw.startswith(("http://", "https://")):
|
||||||
|
return raw
|
||||||
|
play_path = self._extract_play_path(raw)
|
||||||
|
if play_path:
|
||||||
|
return self._abs_url(play_path)
|
||||||
|
if raw.startswith("/movie/"):
|
||||||
|
return self._abs_url(raw)
|
||||||
|
return self.host + f"/movie/{raw}.html"
|
||||||
|
|
||||||
def _parse_detail(self, html, vod_id):
|
def _parse_detail(self, html, vod_id):
|
||||||
root = self._parse_html(html)
|
root = self._parse_html(html)
|
||||||
if root is None:
|
if root is None:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
vod_name = (
|
vod_name = (
|
||||||
self._first_text(root, "//h1[1]")
|
self._first_text(root, "//*[contains(@class,'movie-poster')]//h1[1]")
|
||||||
|
or self._first_text(root, "//h1[1]")
|
||||||
or self._first_text(root, "//h2[1]")
|
or self._first_text(root, "//h2[1]")
|
||||||
or self._extract_title_text(html)
|
or self._extract_title_text(html)
|
||||||
)
|
)
|
||||||
vod_pic = (
|
vod_pic = (
|
||||||
self._first_attr(root, "//img[contains(@class,'poster')][1]", "src")
|
self._first_attr(root, "//meta[@property='og:image'][1]", "content")
|
||||||
|
or self._first_attr(root, "//*[contains(@class,'movie-poster')]//img[1]", "src")
|
||||||
|
or self._first_attr(root, "//img[contains(@class,'poster')][1]", "src")
|
||||||
or self._first_attr(root, "//*[contains(@class,'poster')]//img[1]", "src")
|
or self._first_attr(root, "//*[contains(@class,'poster')]//img[1]", "src")
|
||||||
or self._first_attr(root, "//img[1]", "src")
|
or self._first_attr(root, "//img[1]", "src")
|
||||||
)
|
)
|
||||||
vod_content = (
|
vod_content = (
|
||||||
self._first_text(root, "//*[contains(@class,'intro')][1]")
|
self._first_attr(root, "//meta[@name='description'][1]", "content")
|
||||||
|
or self._first_text(root, "//*[contains(text(),'剧情简介')]/following::p[1]")
|
||||||
|
or self._first_text(root, "//*[contains(@class,'intro')][1]")
|
||||||
or self._first_text(root, "//*[contains(@class,'description')][1]")
|
or self._first_text(root, "//*[contains(@class,'description')][1]")
|
||||||
or self._first_text(root, "//*[contains(@class,'desc')][1]")
|
or self._first_text(root, "//*[contains(@class,'desc')][1]")
|
||||||
)
|
)
|
||||||
vod_actor = self._extract_meta_text(root, "主演")
|
vod_actor = self._extract_labeled_value(root, "主演") or self._extract_meta_text(root, "主演")
|
||||||
vod_director = self._extract_meta_text(root, "导演")
|
vod_director = self._extract_labeled_value(root, "导演") or self._extract_meta_text(root, "导演")
|
||||||
|
|
||||||
episodes = []
|
episodes = []
|
||||||
seen = set()
|
seen = set()
|
||||||
for index, node in enumerate(root.xpath("//a[contains(@href,'/v_play/')]")):
|
episode_nodes = root.xpath(
|
||||||
|
"//*[@x-data[contains(.,'episodeManager')]]//a[contains(@href,'/play/')]"
|
||||||
|
"|//*[contains(@class,'episode-link') and contains(@href,'/play/')]"
|
||||||
|
"|//a[contains(@href,'/v_play/')]"
|
||||||
|
)
|
||||||
|
for index, node in enumerate(episode_nodes):
|
||||||
href = str(node.get("href") or "").strip()
|
href = str(node.get("href") or "").strip()
|
||||||
pid = self._extract_play_pid(href)
|
pid = self._extract_play_path(href) or self._extract_play_pid(href)
|
||||||
name = self._clean_text(node.text_content()) or f"第{index + 1}集"
|
name = self._clean_text(node.text_content()) or self._clean_text(node.get("data-episode")) or f"第{index + 1}集"
|
||||||
if not pid or pid in seen:
|
if not pid or pid in seen:
|
||||||
continue
|
continue
|
||||||
seen.add(pid)
|
seen.add(pid)
|
||||||
@@ -257,6 +348,137 @@ class Spider(BaseSpider):
|
|||||||
text = self._first_text(root, f"//*[contains(text(),'{label}')][1]")
|
text = self._first_text(root, f"//*[contains(text(),'{label}')][1]")
|
||||||
return re.sub(rf"^{label}[::]?", "", text).strip()
|
return re.sub(rf"^{label}[::]?", "", text).strip()
|
||||||
|
|
||||||
|
def _extract_labeled_value(self, root, label):
|
||||||
|
if root is None:
|
||||||
|
return ""
|
||||||
|
for node in root.xpath(
|
||||||
|
f"//*[normalize-space(text())='{label}' or contains(text(),'{label}:') or contains(text(),'{label}:')]"
|
||||||
|
):
|
||||||
|
text = self._clean_text(node.text_content())
|
||||||
|
inline = re.sub(rf"^{label}[::]?", "", text).strip()
|
||||||
|
if inline and inline != text:
|
||||||
|
return inline
|
||||||
|
sibling = node.getnext()
|
||||||
|
if sibling is not None:
|
||||||
|
sibling_text = self._clean_text(sibling.text_content())
|
||||||
|
if sibling_text:
|
||||||
|
return sibling_text
|
||||||
|
return ""
|
||||||
|
|
||||||
|
def _extract_userlink(self, html):
|
||||||
|
matched = re.search(r"userlink:'([^']+)'", str(html or ""))
|
||||||
|
return str(matched.group(1) or "").strip() if matched else ""
|
||||||
|
|
||||||
|
def _extract_play_dataid(self, html, play_page_url):
|
||||||
|
root = self._parse_html(html)
|
||||||
|
if root is None:
|
||||||
|
return ""
|
||||||
|
play_path = self._extract_play_path(play_page_url)
|
||||||
|
if play_path:
|
||||||
|
for node in root.xpath(f"//a[@dataid and contains(@href,'{play_path}')]"):
|
||||||
|
dataid = str(node.get("dataid") or "").strip()
|
||||||
|
if dataid:
|
||||||
|
return dataid
|
||||||
|
return self._first_attr(root, "//a[@dataid][1]", "dataid")
|
||||||
|
|
||||||
|
def _cache_wasm_assets(self, html):
|
||||||
|
js_rel = self._first_attr(self._parse_html(html), "//*[@id='wasm-cfg'][1]", "data-js")
|
||||||
|
wasm_rel = self._first_attr(self._parse_html(html), "//*[@id='wasm-cfg'][1]", "data-bg")
|
||||||
|
if not js_rel or not wasm_rel:
|
||||||
|
return
|
||||||
|
cache_key = f"{js_rel}|{wasm_rel}"
|
||||||
|
cached = self._wasm_asset_cache.get(cache_key)
|
||||||
|
if cached and os.path.exists(cached.get("js", "")) and os.path.exists(cached.get("wasm", "")):
|
||||||
|
self._wasm_asset_cache["active"] = cached
|
||||||
|
return
|
||||||
|
tmp_dir = tempfile.gettempdir()
|
||||||
|
js_path = os.path.join(tmp_dir, f"lianggebt_{hashlib.md5(js_rel.encode('utf-8')).hexdigest()}.mjs")
|
||||||
|
wasm_path = os.path.join(tmp_dir, f"lianggebt_{hashlib.md5(wasm_rel.encode('utf-8')).hexdigest()}.wasm")
|
||||||
|
if not os.path.exists(js_path):
|
||||||
|
response = self.fetch(self._abs_url(js_rel), headers=self.headers, timeout=15, verify=False)
|
||||||
|
if response.status_code == 200:
|
||||||
|
with open(js_path, "w", encoding="utf-8") as handle:
|
||||||
|
handle.write(response.text or "")
|
||||||
|
if not os.path.exists(wasm_path):
|
||||||
|
response = self.fetch(self._abs_url(wasm_rel), headers=self.headers, timeout=15, verify=False)
|
||||||
|
if response.status_code == 200:
|
||||||
|
with open(wasm_path, "wb") as handle:
|
||||||
|
handle.write(response.content or b"")
|
||||||
|
active = {"js": js_path, "wasm": wasm_path}
|
||||||
|
self._wasm_asset_cache[cache_key] = active
|
||||||
|
self._wasm_asset_cache["active"] = active
|
||||||
|
|
||||||
|
def _build_wasm_play_api_url(self, dataid, secret_key, quality, userlink):
|
||||||
|
active = self._wasm_asset_cache.get("active") or {}
|
||||||
|
js_path = active.get("js", "")
|
||||||
|
wasm_path = active.get("wasm", "")
|
||||||
|
if not dataid or not secret_key or not js_path or not wasm_path:
|
||||||
|
return ""
|
||||||
|
if not os.path.exists(js_path) or not os.path.exists(wasm_path):
|
||||||
|
return ""
|
||||||
|
script = (
|
||||||
|
"import { pathToFileURL } from 'node:url';"
|
||||||
|
"const mod = await import(pathToFileURL(process.argv[1]).href);"
|
||||||
|
"await mod.default({module_or_path: await (await import('node:fs/promises')).readFile(process.argv[2])});"
|
||||||
|
"console.log(mod.build_play_url(process.argv[3], process.argv[4], process.argv[5], process.argv[6]));"
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
[
|
||||||
|
"node",
|
||||||
|
"--input-type=module",
|
||||||
|
"-e",
|
||||||
|
script,
|
||||||
|
js_path,
|
||||||
|
wasm_path,
|
||||||
|
str(dataid),
|
||||||
|
str(secret_key),
|
||||||
|
str(quality or "1080"),
|
||||||
|
str(userlink or "0"),
|
||||||
|
],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=20,
|
||||||
|
check=False,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
if result.returncode != 0:
|
||||||
|
return ""
|
||||||
|
return self._abs_url((result.stdout or "").strip())
|
||||||
|
|
||||||
|
def _request_json(self, url, referer=None):
|
||||||
|
target = str(url or "").strip()
|
||||||
|
if not target:
|
||||||
|
return {}
|
||||||
|
headers = dict(self.headers)
|
||||||
|
headers["Referer"] = referer or self.headers["Referer"]
|
||||||
|
headers["Accept"] = "application/json,text/plain,*/*"
|
||||||
|
try:
|
||||||
|
response = self.fetch(target, headers=headers, timeout=15, verify=False)
|
||||||
|
except Exception:
|
||||||
|
return {}
|
||||||
|
if response.status_code != 200:
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
return json.loads(response.text or "{}")
|
||||||
|
except Exception:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
def _extract_media_from_play_api(self, data):
|
||||||
|
payload = ((data or {}).get("data") or {}) if isinstance(data, dict) else {}
|
||||||
|
quality_urls = payload.get("quality_urls") or []
|
||||||
|
current_quality = self._to_int(payload.get("current_quality"), 0)
|
||||||
|
ordered = []
|
||||||
|
if 0 <= current_quality < len(quality_urls):
|
||||||
|
ordered.append(quality_urls[current_quality])
|
||||||
|
ordered.extend([item for index, item in enumerate(quality_urls) if index != current_quality])
|
||||||
|
for item in ordered:
|
||||||
|
url = str((item or {}).get("url") or "").strip()
|
||||||
|
if url and url != "1":
|
||||||
|
return self._abs_url(url)
|
||||||
|
return ""
|
||||||
|
|
||||||
def _extract_media_url(self, html):
|
def _extract_media_url(self, html):
|
||||||
body = str(html or "")
|
body = str(html or "")
|
||||||
patterns = [
|
patterns = [
|
||||||
|
|||||||
+316
@@ -0,0 +1,316 @@
|
|||||||
|
# coding=utf-8
|
||||||
|
import base64
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from urllib.parse import quote, urljoin
|
||||||
|
|
||||||
|
from Crypto.Cipher import AES
|
||||||
|
from Crypto.Util.Padding import unpad
|
||||||
|
|
||||||
|
from base.spider import Spider as BaseSpider
|
||||||
|
|
||||||
|
sys.path.append("..")
|
||||||
|
|
||||||
|
|
||||||
|
DEFAULT_PIC = "https://youke2.picui.cn/s1/2025/12/21/694796745c0c6.png"
|
||||||
|
PLAYER_AES_KEY = "my-to-newhan-2025" + ("\0" * 15)
|
||||||
|
|
||||||
|
|
||||||
|
class Spider(BaseSpider):
|
||||||
|
def __init__(self):
|
||||||
|
self.name = "新韩剧网"
|
||||||
|
self.host = "https://www.hanju7.com"
|
||||||
|
self.headers = {
|
||||||
|
"User-Agent": (
|
||||||
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||||
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||||
|
"Chrome/144.0.0.0 Safari/537.36"
|
||||||
|
),
|
||||||
|
"Referer": self.host + "/",
|
||||||
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||||
|
"Accept-Language": "zh-CN,zh;q=0.9",
|
||||||
|
}
|
||||||
|
self.classes = [
|
||||||
|
{"type_id": "1", "type_name": "韩剧"},
|
||||||
|
{"type_id": "3", "type_name": "韩国电影"},
|
||||||
|
{"type_id": "4", "type_name": "韩国综艺"},
|
||||||
|
{"type_id": "hot", "type_name": "排行榜"},
|
||||||
|
{"type_id": "new", "type_name": "最新更新"},
|
||||||
|
]
|
||||||
|
|
||||||
|
def init(self, extend=""):
|
||||||
|
return None
|
||||||
|
|
||||||
|
def getName(self):
|
||||||
|
return self.name
|
||||||
|
|
||||||
|
def homeContent(self, filter):
|
||||||
|
return {"class": self.classes}
|
||||||
|
|
||||||
|
def _build_url(self, value):
|
||||||
|
raw = str(value or "").strip()
|
||||||
|
if not raw:
|
||||||
|
return ""
|
||||||
|
if raw.startswith(("http://", "https://")):
|
||||||
|
return raw
|
||||||
|
if raw.startswith("//"):
|
||||||
|
return "https:" + raw
|
||||||
|
return urljoin(self.host + "/", raw)
|
||||||
|
|
||||||
|
def _extract_detail_id(self, href):
|
||||||
|
matched = re.search(r"/detail/([^/?#]+)\.html", self._build_url(href))
|
||||||
|
return matched.group(1) if matched else str(href or "").strip().strip("/")
|
||||||
|
|
||||||
|
def _request_text(self, path_or_url, headers=None, allow_redirects=True):
|
||||||
|
target = path_or_url if str(path_or_url).startswith("http") else self._build_url(path_or_url)
|
||||||
|
merged = dict(self.headers)
|
||||||
|
if headers:
|
||||||
|
merged.update(headers)
|
||||||
|
response = self.fetch(target, headers=merged, timeout=10, allow_redirects=allow_redirects)
|
||||||
|
if response.status_code != 200:
|
||||||
|
return ""
|
||||||
|
return response.text or ""
|
||||||
|
|
||||||
|
def _request_html(self, path_or_url, headers=None):
|
||||||
|
return self._request_text(path_or_url, headers=headers)
|
||||||
|
|
||||||
|
def _clean_text(self, text):
|
||||||
|
return re.sub(r"\s+", " ", str(text or "").replace("\xa0", " ")).strip()
|
||||||
|
|
||||||
|
def _node_text_without_children(self, node):
|
||||||
|
if node is None:
|
||||||
|
return ""
|
||||||
|
text = self._clean_text(node.text or "")
|
||||||
|
if text:
|
||||||
|
return text
|
||||||
|
return self._clean_text("".join(node.xpath(".//text()")))
|
||||||
|
|
||||||
|
def _build_list_item(self, anchor, remarks="", pic=""):
|
||||||
|
href = (anchor.xpath("./@href") or [""])[0]
|
||||||
|
vod_id = self._extract_detail_id(href)
|
||||||
|
if not vod_id:
|
||||||
|
return None
|
||||||
|
name = self._clean_text(
|
||||||
|
(anchor.xpath("./@title") or [""])[0]
|
||||||
|
or self._node_text_without_children(anchor)
|
||||||
|
or "".join(anchor.xpath(".//text()"))
|
||||||
|
)
|
||||||
|
image = self._build_url(pic or (anchor.xpath("./@data-original") or [""])[0] or (anchor.xpath("./@src") or [""])[0])
|
||||||
|
return {
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": name,
|
||||||
|
"vod_pic": image,
|
||||||
|
"vod_remarks": self._clean_text(remarks),
|
||||||
|
}
|
||||||
|
|
||||||
|
def _parse_home_cards(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
items = []
|
||||||
|
for node in root.xpath("//div[contains(@class,'list')]//ul/li"):
|
||||||
|
anchor = node.xpath(".//a[@href]")
|
||||||
|
if not anchor:
|
||||||
|
continue
|
||||||
|
item = self._build_list_item(anchor[0], remarks="".join(node.xpath("./span[1]//text()")))
|
||||||
|
if item and item["vod_name"]:
|
||||||
|
items.append(item)
|
||||||
|
return items
|
||||||
|
|
||||||
|
def _default_category_pic(self, vod_id):
|
||||||
|
return f"https://pics.hanju7.com/pics/{vod_id}.jpg" if vod_id else ""
|
||||||
|
|
||||||
|
def _parse_category_cards(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
items = []
|
||||||
|
for node in root.xpath("//div[contains(@class,'list')]//ul/li"):
|
||||||
|
anchor = node.xpath(".//a[contains(@class,'tu')][1]")
|
||||||
|
if not anchor:
|
||||||
|
continue
|
||||||
|
anchor = anchor[0]
|
||||||
|
vod_id = self._extract_detail_id((anchor.xpath("./@href") or [""])[0])
|
||||||
|
item = self._build_list_item(
|
||||||
|
anchor,
|
||||||
|
remarks="".join(node.xpath(".//span[contains(@class,'tip')][1]//text()")),
|
||||||
|
pic=(anchor.xpath("./@data-original") or [""])[0],
|
||||||
|
)
|
||||||
|
if item:
|
||||||
|
item["vod_pic"] = item["vod_pic"] or self._default_category_pic(vod_id)
|
||||||
|
items.append(item)
|
||||||
|
return items
|
||||||
|
|
||||||
|
def _parse_rank_cards(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
items = []
|
||||||
|
for node in root.xpath("//div[contains(@class,'txt') or contains(@class,'list_txt')]//ul/li"):
|
||||||
|
anchor = node.xpath(".//a[@href][1]")
|
||||||
|
if not anchor:
|
||||||
|
continue
|
||||||
|
anchor = anchor[0]
|
||||||
|
item = self._build_list_item(anchor, remarks="".join(node.xpath("./span[1]//text()")))
|
||||||
|
if item:
|
||||||
|
item["vod_pic"] = self._default_category_pic(item["vod_id"])
|
||||||
|
items.append(item)
|
||||||
|
return items
|
||||||
|
|
||||||
|
def homeVideoContent(self):
|
||||||
|
return {"list": self._parse_home_cards(self._request_html(self.host))[:100]}
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
page = int(pg)
|
||||||
|
if str(tid) in ["hot", "new"]:
|
||||||
|
items = self._parse_rank_cards(self._request_html(f"{self.host}/{tid}.html"))
|
||||||
|
page_size = 20
|
||||||
|
start = (page - 1) * page_size
|
||||||
|
result = items[start : start + page_size]
|
||||||
|
return {"page": page, "limit": page_size, "total": len(items), "list": result}
|
||||||
|
|
||||||
|
page_suffix = "" if page <= 1 else str(page - 1)
|
||||||
|
url = f"{self.host}/list/{tid}---{page_suffix}.html"
|
||||||
|
items = self._parse_category_cards(self._request_html(url))
|
||||||
|
return {"page": page, "limit": len(items), "total": page * 20 + len(items), "list": items}
|
||||||
|
|
||||||
|
def _extract_redirect_location(self, response):
|
||||||
|
if response is None:
|
||||||
|
return ""
|
||||||
|
return (
|
||||||
|
response.headers.get("Location")
|
||||||
|
or response.headers.get("location")
|
||||||
|
or response.headers.get("Location".lower())
|
||||||
|
or ""
|
||||||
|
)
|
||||||
|
|
||||||
|
def _native_post_search(self, keyword):
|
||||||
|
payload = f"show=searchkey&keyboard={quote(str(keyword or ''))}"
|
||||||
|
headers = dict(self.headers)
|
||||||
|
headers["Content-Type"] = "application/x-www-form-urlencoded"
|
||||||
|
response = self.post(
|
||||||
|
self.host + "/search/",
|
||||||
|
data=payload,
|
||||||
|
headers=headers,
|
||||||
|
timeout=10,
|
||||||
|
allow_redirects=False,
|
||||||
|
)
|
||||||
|
cookies = response.headers.get("Set-Cookie") or response.headers.get("set-cookie") or ""
|
||||||
|
if isinstance(cookies, list):
|
||||||
|
cookies = "; ".join(cookies)
|
||||||
|
return self._extract_redirect_location(response), cookies
|
||||||
|
|
||||||
|
def _normalize_search_redirect(self, location):
|
||||||
|
raw = str(location or "").strip()
|
||||||
|
if not raw:
|
||||||
|
return ""
|
||||||
|
if raw.startswith(("http://", "https://")):
|
||||||
|
return raw
|
||||||
|
if raw.startswith("/"):
|
||||||
|
return self.host + raw
|
||||||
|
return self.host + "/search/" + raw.lstrip("/")
|
||||||
|
|
||||||
|
def _parse_search_cards(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
items = []
|
||||||
|
for node in root.xpath("//div[contains(@class,'txt')]//ul/li"):
|
||||||
|
node_id = (node.xpath("./@id") or [""])[0]
|
||||||
|
if node_id == "t":
|
||||||
|
continue
|
||||||
|
anchor = node.xpath(".//*[@id='name']//a[@href][1]")
|
||||||
|
if not anchor:
|
||||||
|
continue
|
||||||
|
title = self._clean_text("".join(anchor[0].xpath(".//text()")))
|
||||||
|
title = re.sub(r"\(\d+\)$", "", title).strip()
|
||||||
|
items.append(
|
||||||
|
{
|
||||||
|
"vod_id": self._extract_detail_id((anchor[0].xpath("./@href") or [""])[0]),
|
||||||
|
"vod_name": title,
|
||||||
|
"vod_pic": DEFAULT_PIC,
|
||||||
|
"vod_remarks": self._clean_text("".join(node.xpath(".//*[@id='actor']//text()"))),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return items
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg="1"):
|
||||||
|
page = int(pg)
|
||||||
|
keyword = self._clean_text(key)
|
||||||
|
if not keyword:
|
||||||
|
return {"page": page, "total": 0, "list": []}
|
||||||
|
location, cookie = self._native_post_search(keyword)
|
||||||
|
url = self._normalize_search_redirect(location)
|
||||||
|
if not url:
|
||||||
|
return {"page": page, "total": 0, "list": []}
|
||||||
|
headers = {"Cookie": cookie} if cookie else None
|
||||||
|
items = self._parse_search_cards(self._request_html(url, headers=headers))
|
||||||
|
return {"page": page, "total": len(items), "list": items}
|
||||||
|
|
||||||
|
def _parse_play_groups(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
source_name = self._clean_text("".join(root.xpath("//*[@id='playlist'][1]//text()"))) or "新韩剧线路"
|
||||||
|
episodes = []
|
||||||
|
for anchor in root.xpath("//div[contains(@class,'play')]//ul/li//a[@onclick]"):
|
||||||
|
onclick = (anchor.xpath("./@onclick") or [""])[0]
|
||||||
|
matched = re.search(r"'([^']+)'", onclick)
|
||||||
|
if not matched:
|
||||||
|
continue
|
||||||
|
name = self._clean_text("".join(anchor.xpath(".//text()"))) or "正片"
|
||||||
|
episodes.append(f"{name}${matched.group(1)}")
|
||||||
|
if not episodes:
|
||||||
|
return []
|
||||||
|
return [(source_name, "#".join(episodes))]
|
||||||
|
|
||||||
|
def detailContent(self, ids):
|
||||||
|
raw = ids[0] if isinstance(ids, list) else ids
|
||||||
|
vod_id = self._extract_detail_id(raw)
|
||||||
|
html = self._request_html(f"{self.host}/detail/{vod_id}.html")
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return {"list": []}
|
||||||
|
play_groups = self._parse_play_groups(html)
|
||||||
|
info = root.xpath("//div[contains(@class,'detail')]//div[contains(@class,'info')]/dl/dd/text()")
|
||||||
|
info = [self._clean_text(item) for item in info]
|
||||||
|
pic = (root.xpath("//div[contains(@class,'detail')]//div[contains(@class,'pic')]//img/@data-original") or [""])[0]
|
||||||
|
vod = {
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": info[0] if len(info) > 0 else "",
|
||||||
|
"vod_pic": self._build_url(pic),
|
||||||
|
"vod_actor": info[1] if len(info) > 1 else "",
|
||||||
|
"vod_remarks": info[4] if len(info) > 4 else "",
|
||||||
|
"vod_year": info[5] if len(info) > 5 else "",
|
||||||
|
"vod_content": self._clean_text("".join(root.xpath("//div[contains(@class,'juqing')][1]//text()"))),
|
||||||
|
"vod_play_from": "$$$".join(name for name, _ in play_groups),
|
||||||
|
"vod_play_url": "$$$".join(urls for _, urls in play_groups),
|
||||||
|
}
|
||||||
|
return {"list": [vod]}
|
||||||
|
|
||||||
|
def _decrypt_play_url(self, payload):
|
||||||
|
raw = str(payload or "").strip()
|
||||||
|
if not raw:
|
||||||
|
return ""
|
||||||
|
try:
|
||||||
|
data = base64.b64decode(raw)
|
||||||
|
iv = data[:16]
|
||||||
|
ciphertext = data[16:]
|
||||||
|
cipher = AES.new(PLAYER_AES_KEY.encode("utf-8"), AES.MODE_CBC, iv)
|
||||||
|
return unpad(cipher.decrypt(ciphertext), AES.block_size).decode("utf-8").strip()
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
def _is_media_url(self, url):
|
||||||
|
return bool(re.search(r"\.(m3u8|mp4|flv|avi|mkv|ts)(?:[?#]|$)", str(url or ""), re.I))
|
||||||
|
|
||||||
|
def playerContent(self, flag, id, vipFlags):
|
||||||
|
raw_id = str(id or "").strip()
|
||||||
|
if self._is_media_url(raw_id):
|
||||||
|
return {"parse": 0, "jx": 0, "url": raw_id, "header": {"Referer": self.host + "/"}}
|
||||||
|
|
||||||
|
encrypted = self._request_text(f"{self.host}/u/u1.php?ud={raw_id}")
|
||||||
|
real_url = self._decrypt_play_url(encrypted)
|
||||||
|
if self._is_media_url(real_url):
|
||||||
|
return {"parse": 0, "jx": 0, "url": real_url, "header": {"Referer": self.host + "/"}}
|
||||||
|
return {"parse": 1, "jx": 1, "url": real_url or raw_id, "header": {"Referer": self.host + "/"}}
|
||||||
Reference in New Issue
Block a user