From 8f5856c7b2e7179bdd50ad65c2386d9567ce161c Mon Sep 17 00:00:00 2001
From: Harold <8866033@gmail.com>
Date: Sat, 18 Apr 2026 11:20:36 +0800
Subject: [PATCH] feat: add czzy detail parsing
---
py/tests/test_czzy.py | 44 +++++++++++++++++++++++++++++
py/厂长资源.py | 64 +++++++++++++++++++++++++++++++++++++++++++
2 files changed, 108 insertions(+)
diff --git a/py/tests/test_czzy.py b/py/tests/test_czzy.py
index d323968..513f33f 100644
--- a/py/tests/test_czzy.py
+++ b/py/tests/test_czzy.py
@@ -103,6 +103,50 @@ class TestCZZYSpider(unittest.TestCase):
self.assertEqual(result["list"][0]["vod_id"], "/movie/search-hit.html")
self.assertEqual(result["list"][0]["vod_name"], "搜索影片")
+ def test_parse_detail_page_splits_direct_and_pan_sources(self):
+ html = """
+
+
繁花
+

+
+
年份:2024
+ 地区:中国大陆
+ 导演:王家卫
+ 主演:胡歌
+
+
一段剧情简介
+
+
+
+ """
+
+ detail = self.spider._parse_detail_page(html, "https://www.czzy89.com", "/movie/fanhua.html")
+ vod = detail["list"][0]
+ self.assertEqual(vod["vod_name"], "繁花")
+ self.assertEqual(vod["vod_year"], "2024")
+ self.assertEqual(vod["vod_play_from"], "厂长资源$$$网盘资源")
+ self.assertIn("第1集$https://www.czzy89.com/v_play/1.html", vod["vod_play_url"])
+ self.assertIn("阿里云盘$https://www.alipan.com/s/demo", vod["vod_play_url"])
+
+ @patch.object(Spider, "_request_html")
+ def test_detail_content_reads_from_vod_id_path(self, mock_request_html):
+ mock_request_html.return_value = (
+ """
+ 示例影片
+
+ """,
+ "https://www.czzy89.com",
+ )
+
+ result = self.spider.detailContent(["/movie/example.html"])
+ self.assertEqual(result["list"][0]["vod_id"], "/movie/example.html")
+ self.assertEqual(result["list"][0]["vod_name"], "示例影片")
+
if __name__ == "__main__":
unittest.main()
diff --git a/py/厂长资源.py b/py/厂长资源.py
index b75cb1e..e1924bf 100644
--- a/py/厂长资源.py
+++ b/py/厂长资源.py
@@ -96,6 +96,22 @@ class Spider(BaseSpider):
raise last_error
return "", self.current_host
+ def _normalize_url(self, value, host):
+ value = (value or "").strip()
+ if not value:
+ return ""
+ if value.startswith(("http://", "https://")):
+ return value
+ return urljoin(host, value)
+
+ def _extract_meta_value(self, root, labels):
+ for text in root.xpath("//li/text()"):
+ clean = text.strip()
+ for label in labels:
+ if clean.startswith(label):
+ return clean.split(":", 1)[-1].strip()
+ return ""
+
def _page_result(self, items, pg):
page = int(pg)
pagecount = page + 1 if items else page
@@ -150,12 +166,60 @@ class Spider(BaseSpider):
return results
+ def _parse_detail_page(self, html, host, vod_id):
+ root = self.html(html)
+ title = ((root.xpath("//h1/text()") or [""])[0]).strip()
+ pic = ((root.xpath("//img[@src][1]/@src") or [""])[0]).strip()
+ content = "".join(root.xpath("//*[contains(@class,'yp_context')][1]//text()")).strip()
+
+ direct = []
+ for anchor in root.xpath("//*[contains(@class,'paly_list_btn')]//a[@href]"):
+ name = "".join(anchor.xpath(".//text()")).strip() or "播放"
+ href = self._normalize_url((anchor.xpath("./@href") or [""])[0], host)
+ if href:
+ direct.append(f"{name}${href}")
+
+ pan = []
+ for anchor in root.xpath("//*[contains(@class,'ypbt_down_list')]//a[@href]"):
+ name = "".join(anchor.xpath(".//text()")).strip() or "网盘资源"
+ href = self._normalize_url((anchor.xpath("./@href") or [""])[0], host)
+ if href:
+ pan.append(f"{name}${href}")
+
+ play_from = []
+ play_url = []
+ if direct:
+ play_from.append("厂长资源")
+ play_url.append("#".join(dict.fromkeys(direct)))
+ if pan:
+ play_from.append("网盘资源")
+ play_url.append("#".join(dict.fromkeys(pan)))
+
+ vod = {
+ "vod_id": vod_id,
+ "vod_name": title,
+ "vod_pic": self._normalize_url(pic, host),
+ "vod_year": self._extract_meta_value(root, ["年份:"]),
+ "vod_area": self._extract_meta_value(root, ["地区:"]),
+ "vod_actor": self._extract_meta_value(root, ["主演:"]),
+ "vod_director": self._extract_meta_value(root, ["导演:"]),
+ "vod_content": content,
+ "vod_play_from": "$$$".join(play_from),
+ "vod_play_url": "$$$".join(play_url),
+ }
+ return {"list": [vod]}
+
def categoryContent(self, tid, pg, filter, extend):
path = self.category_paths.get(tid, self.category_paths["movie"]).format(pg=pg)
html, host = self._request_html(path, expect_xpath="//a[@href]")
items = self._parse_media_cards(html, host)
return self._page_result(items, pg)
+ def detailContent(self, ids):
+ vod_id = ids[0]
+ html, host = self._request_html(vod_id, expect_xpath="//h1|//*[contains(@class,'paly_list_btn')]")
+ return self._parse_detail_page(html, host, vod_id)
+
def searchContent(self, key, quick, pg="1"):
path = "/boss1O1?q={keyword}".format(keyword=quote(key))
html, host = self._request_html(path, expect_xpath="//a[@href]")