feat: add czzy category and search flow
This commit is contained in:
@@ -1,6 +1,7 @@
|
|||||||
import unittest
|
import unittest
|
||||||
from importlib.machinery import SourceFileLoader
|
from importlib.machinery import SourceFileLoader
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[1]
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
@@ -13,6 +14,10 @@ class TestCZZYSpider(unittest.TestCase):
|
|||||||
self.spider = Spider()
|
self.spider = Spider()
|
||||||
self.spider.init()
|
self.spider.init()
|
||||||
|
|
||||||
|
def test_default_host_prefers_czzy89(self):
|
||||||
|
self.assertEqual(self.spider.hosts[0], "https://www.czzy89.com")
|
||||||
|
self.assertEqual(self.spider.current_host, "https://www.czzy89.com")
|
||||||
|
|
||||||
def test_home_content_exposes_expected_categories(self):
|
def test_home_content_exposes_expected_categories(self):
|
||||||
content = self.spider.homeContent(False)
|
content = self.spider.homeContent(False)
|
||||||
class_ids = [item["type_id"] for item in content["class"]]
|
class_ids = [item["type_id"] for item in content["class"]]
|
||||||
@@ -43,6 +48,61 @@ class TestCZZYSpider(unittest.TestCase):
|
|||||||
],
|
],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@patch.object(Spider, "fetch")
|
||||||
|
def test_request_html_falls_back_to_second_host(self, mock_fetch):
|
||||||
|
class FakeResponse:
|
||||||
|
def __init__(self, text, status_code=200):
|
||||||
|
self.text = text
|
||||||
|
self.status_code = status_code
|
||||||
|
self.encoding = "utf-8"
|
||||||
|
|
||||||
|
mock_fetch.side_effect = [
|
||||||
|
Exception("first host down"),
|
||||||
|
FakeResponse("<html><body>ok</body></html>"),
|
||||||
|
]
|
||||||
|
|
||||||
|
html, host = self.spider._request_html("/movie_bt/page/1", expect_xpath="//body")
|
||||||
|
self.assertIn("ok", html)
|
||||||
|
self.assertEqual(host, "https://www.cz01.org")
|
||||||
|
self.assertEqual(self.spider.current_host, "https://www.cz01.org")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_category_content_builds_media_list(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = (
|
||||||
|
"""
|
||||||
|
<ul>
|
||||||
|
<li>
|
||||||
|
<a href="/movie/abc.html"><img src="/cover.jpg" alt="分类影片" /></a>
|
||||||
|
<span class="hdinfo">HD</span>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
""",
|
||||||
|
"https://www.czzy89.com",
|
||||||
|
)
|
||||||
|
|
||||||
|
result = self.spider.categoryContent("movie", "2", False, {})
|
||||||
|
self.assertEqual(result["page"], 2)
|
||||||
|
self.assertEqual(result["list"][0]["vod_name"], "分类影片")
|
||||||
|
self.assertEqual(result["list"][0]["vod_pic"], "https://www.czzy89.com/cover.jpg")
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_search_content_reuses_media_card_parser(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = (
|
||||||
|
"""
|
||||||
|
<div>
|
||||||
|
<li>
|
||||||
|
<a href="/movie/search-hit.html" title="搜索影片"></a>
|
||||||
|
<img data-src="https://img.example/search.jpg" />
|
||||||
|
</li>
|
||||||
|
</div>
|
||||||
|
""",
|
||||||
|
"https://www.czzy89.com",
|
||||||
|
)
|
||||||
|
|
||||||
|
result = self.spider.searchContent("繁花", False, "1")
|
||||||
|
self.assertEqual(result["list"][0]["vod_id"], "/movie/search-hit.html")
|
||||||
|
self.assertEqual(result["list"][0]["vod_name"], "搜索影片")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
+69
-2
@@ -1,6 +1,6 @@
|
|||||||
# coding=utf-8
|
# coding=utf-8
|
||||||
import sys
|
import sys
|
||||||
from urllib.parse import urljoin
|
from urllib.parse import quote, urljoin
|
||||||
|
|
||||||
from base.spider import Spider as BaseSpider
|
from base.spider import Spider as BaseSpider
|
||||||
|
|
||||||
@@ -11,8 +11,8 @@ class Spider(BaseSpider):
|
|||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.name = "厂长资源"
|
self.name = "厂长资源"
|
||||||
self.hosts = [
|
self.hosts = [
|
||||||
"https://www.cz01.org",
|
|
||||||
"https://www.czzy89.com",
|
"https://www.czzy89.com",
|
||||||
|
"https://www.cz01.org",
|
||||||
]
|
]
|
||||||
self.current_host = self.hosts[0]
|
self.current_host = self.hosts[0]
|
||||||
self.headers = {
|
self.headers = {
|
||||||
@@ -39,6 +39,23 @@ class Spider(BaseSpider):
|
|||||||
{"type_name": "韩剧", "type_id": "kr_drama"},
|
{"type_name": "韩剧", "type_id": "kr_drama"},
|
||||||
{"type_name": "海外剧", "type_id": "intl_drama"},
|
{"type_name": "海外剧", "type_id": "intl_drama"},
|
||||||
]
|
]
|
||||||
|
self.category_paths = {
|
||||||
|
"movie": "/movie_bt/movie_bt_series/dyy/page/{pg}",
|
||||||
|
"tv": "/movie_bt/movie_bt_series/dianshiju/page/{pg}",
|
||||||
|
"anime": "/movie_bt/movie_bt_series/dohua/page/{pg}",
|
||||||
|
"cn_movie": "/movie_bt/movie_bt_series/huayudianying/page/{pg}",
|
||||||
|
"in_movie": "/movie_bt/movie_bt_series/yindudianying/page/{pg}",
|
||||||
|
"ru_movie": "/movie_bt/movie_bt_series/eluosidianying/page/{pg}",
|
||||||
|
"ca_movie": "/movie_bt/movie_bt_series/jianadadianying/page/{pg}",
|
||||||
|
"jp_movie": "/movie_bt/movie_bt_series/ribendianying/page/{pg}",
|
||||||
|
"kr_movie": "/movie_bt/movie_bt_series/hanguodianying/page/{pg}",
|
||||||
|
"western_movie": "/movie_bt/movie_bt_series/meiguodianying/page/{pg}",
|
||||||
|
"cn_drama": "/movie_bt/movie_bt_series/guochanju/page/{pg}",
|
||||||
|
"jp_drama": "/movie_bt/movie_bt_series/rj/page/{pg}",
|
||||||
|
"us_drama": "/movie_bt/movie_bt_series/mj/page/{pg}",
|
||||||
|
"kr_drama": "/movie_bt/movie_bt_series/hj/page/{pg}",
|
||||||
|
"intl_drama": "/movie_bt/movie_bt_series/hwj/page/{pg}",
|
||||||
|
}
|
||||||
|
|
||||||
def init(self, extend=""):
|
def init(self, extend=""):
|
||||||
return None
|
return None
|
||||||
@@ -52,6 +69,44 @@ class Spider(BaseSpider):
|
|||||||
def homeVideoContent(self):
|
def homeVideoContent(self):
|
||||||
return {"list": []}
|
return {"list": []}
|
||||||
|
|
||||||
|
def _request_html(self, path_or_url, expect_xpath=None, referer=None):
|
||||||
|
candidates = [self.current_host] + [host for host in self.hosts if host != self.current_host]
|
||||||
|
last_error = None
|
||||||
|
|
||||||
|
for host in candidates:
|
||||||
|
target = path_or_url if path_or_url.startswith("http") else urljoin(host, path_or_url)
|
||||||
|
headers = dict(self.headers)
|
||||||
|
if referer:
|
||||||
|
headers["Referer"] = referer
|
||||||
|
try:
|
||||||
|
response = self.fetch(target, headers=headers, timeout=10)
|
||||||
|
if response.status_code != 200:
|
||||||
|
continue
|
||||||
|
html = response.text or ""
|
||||||
|
if expect_xpath:
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None or not root.xpath(expect_xpath):
|
||||||
|
continue
|
||||||
|
self.current_host = host
|
||||||
|
return html, host
|
||||||
|
except Exception as exc:
|
||||||
|
last_error = exc
|
||||||
|
|
||||||
|
if last_error:
|
||||||
|
raise last_error
|
||||||
|
return "", self.current_host
|
||||||
|
|
||||||
|
def _page_result(self, items, pg):
|
||||||
|
page = int(pg)
|
||||||
|
pagecount = page + 1 if items else page
|
||||||
|
return {
|
||||||
|
"list": items,
|
||||||
|
"page": page,
|
||||||
|
"pagecount": pagecount,
|
||||||
|
"limit": len(items),
|
||||||
|
"total": pagecount * max(len(items), 1),
|
||||||
|
}
|
||||||
|
|
||||||
def _parse_media_cards(self, html, host):
|
def _parse_media_cards(self, html, host):
|
||||||
root = self.html(html)
|
root = self.html(html)
|
||||||
results = []
|
results = []
|
||||||
@@ -94,3 +149,15 @@ class Spider(BaseSpider):
|
|||||||
)
|
)
|
||||||
|
|
||||||
return results
|
return results
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
path = self.category_paths.get(tid, self.category_paths["movie"]).format(pg=pg)
|
||||||
|
html, host = self._request_html(path, expect_xpath="//a[@href]")
|
||||||
|
items = self._parse_media_cards(html, host)
|
||||||
|
return self._page_result(items, pg)
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg="1"):
|
||||||
|
path = "/boss1O1?q={keyword}".format(keyword=quote(key))
|
||||||
|
html, host = self._request_html(path, expect_xpath="//a[@href]")
|
||||||
|
items = self._parse_media_cards(html, host)
|
||||||
|
return self._page_result(items, pg)
|
||||||
|
|||||||
Reference in New Issue
Block a user