From 4b40851bab4c161b7c86e94c6108895e2bbba437 Mon Sep 17 00:00:00 2001 From: Harold <8866033@gmail.com> Date: Sat, 18 Apr 2026 11:16:07 +0800 Subject: [PATCH] feat: scaffold czzy spider parsing --- py/tests/test_czzy.py | 48 ++++++++++++++++++++++ py/厂长资源.py | 96 +++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 144 insertions(+) create mode 100644 py/tests/test_czzy.py create mode 100644 py/厂长资源.py diff --git a/py/tests/test_czzy.py b/py/tests/test_czzy.py new file mode 100644 index 0000000..6752c64 --- /dev/null +++ b/py/tests/test_czzy.py @@ -0,0 +1,48 @@ +import unittest +from importlib.machinery import SourceFileLoader +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +MODULE = SourceFileLoader("czzy_spider", str(ROOT / "厂长资源.py")).load_module() +Spider = MODULE.Spider + + +class TestCZZYSpider(unittest.TestCase): + def setUp(self): + self.spider = Spider() + self.spider.init() + + def test_home_content_exposes_expected_categories(self): + content = self.spider.homeContent(False) + class_ids = [item["type_id"] for item in content["class"]] + self.assertEqual(class_ids[:3], ["movie", "tv", "anime"]) + self.assertIn("cn_drama", class_ids) + + def test_parse_media_cards_extracts_basic_fields(self): + html = """ + + """ + cards = self.spider._parse_media_cards(html, "https://www.cz01.org") + self.assertEqual( + cards, + [ + { + "vod_id": "/movie/abc.html", + "vod_name": "测试影片", + "vod_pic": "https://img.example/cover.jpg", + "vod_remarks": "更新至10集", + } + ], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/py/厂长资源.py b/py/厂长资源.py new file mode 100644 index 0000000..3a2c81a --- /dev/null +++ b/py/厂长资源.py @@ -0,0 +1,96 @@ +# coding=utf-8 +import sys +from urllib.parse import urljoin + +from base.spider import Spider as BaseSpider + +sys.path.append("..") + + +class Spider(BaseSpider): + def __init__(self): + self.name = "厂长资源" + self.hosts = [ + "https://www.cz01.org", + "https://www.czzy89.com", + ] + self.current_host = self.hosts[0] + self.headers = { + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/124.0.0.0 Safari/537.36" + ) + } + self.categories = [ + {"type_name": "电影", "type_id": "movie"}, + {"type_name": "电视剧", "type_id": "tv"}, + {"type_name": "动漫", "type_id": "anime"}, + {"type_name": "华语电影", "type_id": "cn_movie"}, + {"type_name": "印度电影", "type_id": "in_movie"}, + {"type_name": "俄罗斯电影", "type_id": "ru_movie"}, + {"type_name": "加拿大电影", "type_id": "ca_movie"}, + {"type_name": "日本电影", "type_id": "jp_movie"}, + {"type_name": "韩国电影", "type_id": "kr_movie"}, + {"type_name": "欧美电影", "type_id": "western_movie"}, + {"type_name": "国产剧", "type_id": "cn_drama"}, + {"type_name": "日剧", "type_id": "jp_drama"}, + {"type_name": "美剧", "type_id": "us_drama"}, + {"type_name": "韩剧", "type_id": "kr_drama"}, + {"type_name": "海外剧", "type_id": "intl_drama"}, + ] + + def init(self, extend=""): + return None + + def getName(self): + return self.name + + def homeContent(self, filter): + return {"class": self.categories} + + def homeVideoContent(self): + return {"list": []} + + def _parse_media_cards(self, html, host): + root = self.html(html) + results = [] + if root is None: + return results + + for item in root.xpath("//li[.//a[@href]]"): + href = (item.xpath(".//a[@href][1]/@href") or [""])[0].strip() + if not href: + continue + + title = ( + (item.xpath(".//img[@alt][1]/@alt") or [""])[0].strip() + or (item.xpath(".//a[@title][1]/@title") or [""])[0].strip() + or "".join(item.xpath(".//a[1]//text()")).strip() + ) + + pic = "" + for expr in [ + ".//img[@data-original][1]/@data-original", + ".//img[@data-src][1]/@data-src", + ".//img[@src][1]/@src", + ]: + pic = (item.xpath(expr) or [""])[0].strip() + if pic: + break + + remarks = ( + (item.xpath(".//*[contains(@class,'jidi')][1]/text()") or [""])[0].strip() + or (item.xpath(".//*[contains(@class,'hdinfo')][1]/text()") or [""])[0].strip() + ) + + results.append( + { + "vod_id": href, + "vod_name": title or "未命名", + "vod_pic": urljoin(host, pic) if pic.startswith("/") else pic, + "vod_remarks": remarks, + } + ) + + return results