From 8d1ad1f82d769d19ff808435789d37a146ea3d4f Mon Sep 17 00:00:00 2001 From: Harold <8866033@gmail.com> Date: Mon, 20 Apr 2026 14:40:02 +0800 Subject: [PATCH] =?UTF-8?q?=E7=8E=A9=E5=81=B6=E8=81=9A=E5=90=88?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- py/tests/test_玩偶聚合.py | 357 -------------------------- py/玩偶聚合.py | 516 -------------------------------------- 2 files changed, 873 deletions(-) delete mode 100644 py/tests/test_玩偶聚合.py delete mode 100644 py/玩偶聚合.py diff --git a/py/tests/test_玩偶聚合.py b/py/tests/test_玩偶聚合.py deleted file mode 100644 index 51b6536..0000000 --- a/py/tests/test_玩偶聚合.py +++ /dev/null @@ -1,357 +0,0 @@ -import base64 -import json -import unittest -from importlib.machinery import SourceFileLoader -from pathlib import Path -from types import SimpleNamespace -from unittest.mock import patch - - -ROOT = Path(__file__).resolve().parents[1] -MODULE = SourceFileLoader("wanou_aggregate_spider", str(ROOT / "玩偶聚合.py")).load_module() -Spider = MODULE.Spider - - -class TestWanouAggregateSpider(unittest.TestCase): - def setUp(self): - Spider._instance = None - self.spider = Spider() - self.spider.init() - - def test_home_content_exposes_site_classes_and_category_filter(self): - content = self.spider.homeContent(False) - self.assertEqual( - [item["type_id"] for item in content["class"][:3]], - ["site_wanou", "site_muou", "site_labi"], - ) - self.assertEqual(content["class"][0]["type_name"], "玩偶") - self.assertEqual(content["filters"]["site_wanou"][0]["key"], "categoryId") - self.assertEqual(content["filters"]["site_wanou"][0]["value"][1], {"n": "电影", "v": "1"}) - - @patch.object( - Spider, - "_load_local_filter_groups", - return_value=[ - { - "key": "year", - "name": "年份", - "init": "", - "value": [{"n": "全部", "v": ""}, {"n": "2025", "v": "2025"}], - } - ], - ) - def test_home_content_appends_local_filter_groups_after_category_filter(self, mock_load_local_filter_groups): - content = self.spider.homeContent(False) - self.assertEqual( - [item["key"] for item in content["filters"]["site_wanou"][:2]], - ["categoryId", "year"], - ) - - def test_encode_and_decode_site_vod_id_round_trip(self): - vod_id = self.spider._encode_site_vod_id("wanou", "/voddetail/12345.html") - self.assertEqual(vod_id, "site:wanou:/voddetail/12345.html") - self.assertEqual( - self.spider._decode_site_vod_id(vod_id), - {"site": "wanou", "path": "/voddetail/12345.html"}, - ) - - def test_encode_and_decode_aggregate_vod_id_round_trip(self): - payload = [ - {"site": "wanou", "path": "/voddetail/1.html", "name": "繁花", "year": "2024"}, - {"site": "muou", "path": "/voddetail/2.html", "name": "繁花", "year": "2024"}, - ] - vod_id = self.spider._encode_aggregate_vod_id(payload) - self.assertTrue(vod_id.startswith("agg:")) - decoded = self.spider._decode_aggregate_vod_id(vod_id) - self.assertEqual(decoded, payload) - - def test_normalize_title_removes_spaces_punctuation_and_resolution_tags(self): - self.assertEqual( - self.spider._normalize_title(" 繁花 4K.HDR-玩偶 "), - "繁花", - ) - - def test_is_same_title_rejects_year_conflict(self): - left = {"vod_name": "繁花", "vod_year": "2024"} - right = {"vod_name": "繁花", "vod_year": "2023"} - self.assertFalse(self.spider._is_same_title(left, right)) - - def test_build_category_url_uses_selected_category_and_filters(self): - site = { - "id": "wanou", - "domains": ["https://www.wogg.net"], - "category_url": "/vodshow/{categoryId}--------{page}---.html", - "category_url_with_filters": "/vodshow/{categoryId}-{area}-{by}-{class}-----{page}---{year}.html", - } - url = self.spider._build_category_url( - site, - "1", - "2", - {"categoryId": "1", "area": "香港", "by": "score", "class": "动作", "year": "2025"}, - ) - self.assertEqual( - url, - "https://www.wogg.net/vodshow/1-%E9%A6%99%E6%B8%AF-score-%E5%8A%A8%E4%BD%9C-----2---2025.html", - ) - - @patch.object(Spider, "fetch") - def test_request_with_failover_tries_next_domain_when_first_fails(self, mock_fetch): - def fake_fetch(url, headers=None, timeout=10): - if url.startswith("https://bad.example"): - raise RuntimeError("boom") - return SimpleNamespace(status_code=200, text="ok") - - mock_fetch.side_effect = fake_fetch - site = {"domains": ["https://bad.example", "https://good.example"]} - html = self.spider._request_with_failover(site, "/vodshow/1--------1---.html") - self.assertIn("ok", html) - self.assertEqual(site["domains"][0], "https://good.example") - - def test_parse_cards_extracts_short_site_vod_id_title_cover_and_remarks(self): - site = {"id": "wanou", "domains": ["https://www.wogg.net"], "list_xpath": "//*[contains(@class,'module-item')]"} - html = """ -
-
- - 示例影片 -
-
HD
-
- """ - cards = self.spider._parse_cards(site, html) - self.assertEqual( - cards, - [ - { - "vod_id": "site:wanou:/voddetail/123.html", - "vod_name": "示例影片", - "vod_pic": "https://www.wogg.net/poster.jpg", - "vod_remarks": "HD", - "vod_year": "", - "_site": "wanou", - "_detail_path": "/voddetail/123.html", - } - ], - ) - - @patch.object(Spider, "_request_with_failover") - def test_category_content_uses_default_category_when_extend_missing(self, mock_request_with_failover): - mock_request_with_failover.return_value = """ -
-
- - 分类影片 -
-
更新至10集
-
- """ - result = self.spider.categoryContent("site_wanou", "2", False, {}) - self.assertEqual(result["page"], 2) - self.assertEqual(result["list"][0]["vod_name"], "分类影片") - - def test_aggregate_search_results_merges_same_title_and_keeps_highest_priority_source(self): - raw_results = [ - { - "vod_id": "site:muou:/voddetail/2.html", - "vod_name": "繁花", - "vod_pic": "https://img.example/m.jpg", - "vod_remarks": "木偶版", - "vod_year": "2024", - "_site": "muou", - "_detail_path": "/voddetail/2.html", - }, - { - "vod_id": "site:wanou:/voddetail/1.html", - "vod_name": "繁花", - "vod_pic": "https://img.example/w.jpg", - "vod_remarks": "玩偶版", - "vod_year": "2024", - "_site": "wanou", - "_detail_path": "/voddetail/1.html", - }, - ] - aggregated = self.spider._aggregate_search_results(raw_results) - self.assertEqual(len(aggregated), 1) - self.assertEqual(aggregated[0]["vod_name"], "繁花") - self.assertEqual(aggregated[0]["vod_pic"], "https://img.example/w.jpg") - self.assertEqual(aggregated[0]["vod_remarks"], "玩偶版") - self.assertTrue(aggregated[0]["vod_id"].startswith("agg:")) - - def test_aggregate_search_results_keeps_year_conflict_as_two_items(self): - raw_results = [ - { - "vod_id": "site:wanou:/voddetail/1.html", - "vod_name": "倚天屠龙记", - "vod_year": "2019", - "_site": "wanou", - "_detail_path": "/voddetail/1.html", - }, - { - "vod_id": "site:muou:/voddetail/2.html", - "vod_name": "倚天屠龙记", - "vod_year": "2022", - "_site": "muou", - "_detail_path": "/voddetail/2.html", - }, - ] - aggregated = self.spider._aggregate_search_results(raw_results) - self.assertEqual(len(aggregated), 2) - - @patch.object(Spider, "_fetch_site_search") - def test_search_content_queries_sites_and_returns_aggregated_items(self, mock_fetch_site_search): - mock_fetch_site_search.side_effect = [ - [ - { - "vod_id": "site:wanou:/voddetail/1.html", - "vod_name": "繁花", - "vod_pic": "https://img.example/w.jpg", - "vod_remarks": "玩偶版", - "vod_year": "2024", - "_site": "wanou", - "_detail_path": "/voddetail/1.html", - } - ], - [ - { - "vod_id": "site:muou:/voddetail/2.html", - "vod_name": "繁花", - "vod_pic": "https://img.example/m.jpg", - "vod_remarks": "木偶版", - "vod_year": "2024", - "_site": "muou", - "_detail_path": "/voddetail/2.html", - } - ], - [], - ] - result = self.spider.searchContent("繁花", False, "1") - self.assertEqual(len(result["list"]), 1) - self.assertEqual(result["list"][0]["vod_name"], "繁花") - self.assertNotIn("pagecount", result) - - def test_parse_detail_page_extracts_meta_and_netdisk_links(self): - site = { - "id": "wanou", - "name": "玩偶", - "domains": ["https://www.wogg.net"], - "detail_pan_xpath": "//*[contains(@class,'module-row-info')]//p", - } - html = """ -
示例剧
-
-
导演
导演甲
-
主演
演员甲演员乙
-
剧情

一段剧情简介

-
-

https://pan.quark.cn/s/q1

-

https://pan.baidu.com/s/b1

-
- """ - detail = self.spider._parse_detail_page(site, "/voddetail/123.html", html) - self.assertEqual(detail["vod_name"], "示例剧") - self.assertEqual(detail["vod_pic"], "https://www.wogg.net/poster.jpg") - self.assertEqual(detail["vod_director"], "导演甲") - self.assertEqual(detail["vod_actor"], "演员甲,演员乙") - self.assertEqual(detail["pan_urls"], ["https://pan.quark.cn/s/q1", "https://pan.baidu.com/s/b1"]) - - @patch.object(Spider, "_fetch_site_detail") - def test_detail_content_for_aggregate_id_merges_lines_from_multiple_sites(self, mock_fetch_site_detail): - mock_fetch_site_detail.side_effect = [ - { - "vod_name": "繁花", - "vod_pic": "https://img.example/w.jpg", - "vod_year": "2024", - "vod_director": "导演甲", - "vod_actor": "演员甲", - "vod_content": "玩偶简介", - "pan_urls": ["https://pan.baidu.com/s/b1", "https://pan.quark.cn/s/q1"], - "_site_name": "玩偶", - }, - { - "vod_name": "繁花", - "vod_pic": "https://img.example/m.jpg", - "vod_year": "2024", - "vod_director": "导演乙", - "vod_actor": "演员乙", - "vod_content": "木偶简介", - "pan_urls": ["https://pan.quark.cn/s/q2", "https://pan.baidu.com/s/b1"], - "_site_name": "木偶", - }, - ] - payload = [ - {"site": "wanou", "path": "/voddetail/1.html", "name": "繁花", "year": "2024"}, - {"site": "muou", "path": "/voddetail/2.html", "name": "繁花", "year": "2024"}, - ] - result = self.spider.detailContent([self.spider._encode_aggregate_vod_id(payload)]) - vod = result["list"][0] - self.assertEqual(vod["vod_name"], "繁花") - self.assertEqual(vod["vod_pic"], "https://img.example/w.jpg") - self.assertEqual(vod["vod_play_from"], "baidu#玩偶$$$quark#玩偶$$$quark#木偶") - self.assertEqual( - vod["vod_play_url"], - "百度资源$https://pan.baidu.com/s/b1$$$夸克资源$https://pan.quark.cn/s/q1$$$夸克资源$https://pan.quark.cn/s/q2", - ) - - def test_player_content_passthroughs_known_pan_domains(self): - self.assertEqual( - self.spider.playerContent("quark#玩偶", "https://pan.quark.cn/s/demo", {}), - {"parse": 0, "playUrl": "", "url": "https://pan.quark.cn/s/demo"}, - ) - self.assertEqual( - self.spider.playerContent("baidu#玩偶", "https://pan.baidu.com/s/demo", {}), - {"parse": 0, "playUrl": "", "url": "https://pan.baidu.com/s/demo"}, - ) - - def test_player_content_rejects_non_pan_url(self): - self.assertEqual( - self.spider.playerContent("zxzj", "/vodplay/1-1-1.html", {}), - {"parse": 0, "playUrl": "", "url": ""}, - ) - - @patch.object(Spider, "_request_with_failover") - def test_fetch_site_search_builds_search_url_and_parses_results(self, mock_request_with_failover): - mock_request_with_failover.return_value = """ -
-
-
搜索影片
-
抢先版
-
- """ - site = { - "id": "wanou", - "domains": ["https://www.wogg.net"], - "list_xpath": "//*[contains(@class,'module-item')]", - "search_xpath": "//*[contains(@class,'module-search-item')]", - "search_url": "/vodsearch/-------------.html?wd={keyword}&page={page}", - } - results = self.spider._fetch_site_search(site, "繁花", 1) - self.assertEqual(results[0]["vod_id"], "site:wanou:/voddetail/789.html") - self.assertEqual(results[0]["vod_name"], "搜索影片") - - @patch.object(Spider, "_fetch_site_search") - def test_search_content_skips_site_errors(self, mock_fetch_site_search): - mock_fetch_site_search.side_effect = [ - RuntimeError("boom"), - [ - { - "vod_id": "site:muou:/voddetail/2.html", - "vod_name": "繁花", - "vod_pic": "", - "vod_remarks": "", - "vod_year": "2024", - "_site": "muou", - "_detail_path": "/voddetail/2.html", - } - ], - [], - ] - result = self.spider.searchContent("繁花", False, "1") - self.assertEqual(result["total"], 1) - self.assertEqual(result["list"][0]["vod_name"], "繁花") - - def test_home_content_does_not_expose_content_first_categories(self): - content = self.spider.homeContent(False) - self.assertNotIn("movie", [item["type_id"] for item in content["class"]]) - - def test_search_content_returns_empty_list_for_blank_keyword(self): - self.assertEqual(self.spider.searchContent("", False, "1"), {"page": 1, "total": 0, "list": []}) diff --git a/py/玩偶聚合.py b/py/玩偶聚合.py deleted file mode 100644 index a4732f5..0000000 --- a/py/玩偶聚合.py +++ /dev/null @@ -1,516 +0,0 @@ -# coding=utf-8 -import base64 -import json -import os -import re -import sys -from urllib.parse import quote - -from base.spider import Spider as BaseSpider - -sys.path.append("..") - - -class Spider(BaseSpider): - def __init__(self): - self.name = "玩偶聚合" - self.filter_root = os.path.join(os.path.dirname(__file__), "../筛选") - self.headers = { - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/136.0.0.0 Safari/537.36" - ) - } - self.site_priority = [ - "wanou", - "muou", - "labi", - "zhizhen", - "erxiao", - "huban", - "kuaiying", - "shandian", - "ouge", - ] - self.pan_priority = { - "baidu": 1, - "a139": 2, - "a189": 3, - "a123": 4, - "a115": 5, - "quark": 6, - "xunlei": 7, - "aliyun": 8, - "uc": 9, - } - self.sites = [ - { - "id": "wanou", - "name": "玩偶", - "domains": ["https://wogg.xxooo.cf"], - "filter_files": ["wogg.json"], - "list_xpath": "//*[contains(@class,'module-item')]", - "search_xpath": "//*[contains(@class,'module-search-item')]", - "detail_pan_xpath": "//*[contains(@class,'module-row-info')]//p", - "category_url": "/vodshow/{categoryId}--------{page}---.html", - "category_url_with_filters": "/vodshow/{categoryId}-{area}-{by}-{class}-----{page}---{year}.html", - "search_url": "/vodsearch/-------------.html?wd={keyword}&page={page}", - "default_categories": [("1", "电影"), ("2", "电视剧"), ("3", "动漫"), ("4", "综艺")], - }, - { - "id": "muou", - "name": "木偶", - "domains": ["https://www.muou.site", "http://123.666291.xyz"], - "filter_files": ["mogg.json"], - "list_xpath": "//*[contains(@class,'module-item')]", - "search_xpath": "//*[contains(@class,'module-search-item')]", - "detail_pan_xpath": "//*[contains(@class,'module-row-info')]//p", - "category_url": "/vodshow/{categoryId}--------{page}---.html", - "search_url": "/vodsearch/-------------.html?wd={keyword}&page={page}", - "default_categories": [("1", "电影"), ("2", "电视剧"), ("3", "动漫"), ("29", "综艺")], - }, - { - "id": "labi", - "name": "蜡笔", - "domains": ["http://xiaocge.fun"], - "filter_files": ["labi.json"], - "list_xpath": "//*[contains(@class,'module-item')]", - "search_xpath": "//*[contains(@class,'module-search-item')]", - "detail_pan_xpath": "//*[contains(@class,'module-row-info')]//p", - "category_url": "/vodshow/{categoryId}--------{page}---.html", - "search_url": "/vodsearch/-------------.html?wd={keyword}&page={page}", - "default_categories": [("1", "电影"), ("2", "电视剧"), ("3", "动漫"), ("4", "综艺")], - }, - ] - - def init(self, extend=""): - return None - - def getName(self): - return self.name - - def homeVideoContent(self): - return {"list": []} - - def _load_local_filter_groups(self, site): - return [] - - def _build_site_filters(self, site): - groups = [ - { - "key": "categoryId", - "name": "分类", - "init": site["default_categories"][0][0], - "value": [{"n": "全部", "v": ""}] - + [{"n": name, "v": cid} for cid, name in site["default_categories"]], - } - ] - groups.extend(self._load_local_filter_groups(site)) - return groups - - def homeContent(self, filter): - classes = [{"type_id": f"site_{site['id']}", "type_name": site["name"]} for site in self.sites] - filters = {f"site_{site['id']}": self._build_site_filters(site) for site in self.sites} - return {"class": classes, "filters": filters} - - def _encode_site_vod_id(self, site_id, path): - return f"site:{site_id}:{path}" - - def _decode_site_vod_id(self, value): - prefix, site_id, path = str(value).split(":", 2) - if prefix != "site": - raise ValueError("not site vod id") - return {"site": site_id, "path": path} - - def _encode_aggregate_vod_id(self, payload): - raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8") - return "agg:" + base64.urlsafe_b64encode(raw).decode("utf-8").rstrip("=") - - def _decode_aggregate_vod_id(self, value): - encoded = str(value)[4:] - padded = encoded + "=" * (-len(encoded) % 4) - return json.loads(base64.urlsafe_b64decode(padded.encode("utf-8")).decode("utf-8")) - - def _normalize_title(self, value): - text = str(value or "").lower() - text = re.sub(r"(4k|hdr|2160p|1080p|720p|玩偶|木偶|蜡笔)", "", text, flags=re.I) - text = re.sub(r"[\s\-_.·,,。!!?::()()\[\]]+", "", text) - return text - - def _is_same_title(self, left, right): - left_year = str(left.get("vod_year") or "").strip() - right_year = str(right.get("vod_year") or "").strip() - if left_year and right_year and left_year != right_year: - return False - return self._normalize_title(left.get("vod_name")) == self._normalize_title(right.get("vod_name")) - - def _get_site(self, site_id): - for site in self.sites: - if site["id"] == site_id: - return site - return None - - def _build_absolute_url(self, base, path): - raw = str(path or "").strip() - if not raw: - return "" - if raw.startswith(("http://", "https://")): - return raw - if raw.startswith("//"): - return "https:" + raw - return str(base).rstrip("/") + "/" + raw.lstrip("/") - - def _build_category_url(self, site, category_id, pg, extend): - values = dict(extend or {}) - values.setdefault("categoryId", category_id) - values.setdefault("area", "") - values.setdefault("by", values.get("sort", "")) - values.setdefault("class", "") - values.setdefault("year", "") - if site.get("category_url_with_filters") and any(values.get(key) for key in ("area", "by", "class", "year")): - path = site["category_url_with_filters"].format( - **{ - "categoryId": values["categoryId"], - "area": quote(str(values["area"])), - "by": quote(str(values["by"])), - "class": quote(str(values["class"])), - "page": int(pg), - "year": quote(str(values["year"])), - } - ) - else: - path = site["category_url"].format(categoryId=values["categoryId"], page=int(pg)) - return self._build_absolute_url(site["domains"][0], path) - - def _request_with_failover(self, site, path_or_url, referer=None): - last_error = None - for index, domain in enumerate(list(site["domains"])): - target = path_or_url if str(path_or_url).startswith("http") else self._build_absolute_url(domain, path_or_url) - try: - headers = dict(self.headers) - headers["Referer"] = referer or self._build_absolute_url(domain, "/") - response = self.fetch(target, headers=headers, timeout=10) - if response.status_code == 200 and response.text: - if index > 0: - site["domains"].insert(0, site["domains"].pop(index)) - return response.text - except Exception as exc: - last_error = exc - raise RuntimeError(str(last_error or "all domains failed")) - - def _parse_cards(self, site, html): - root = self.html(html) - if root is None: - return [] - - items = [] - seen = set() - for card in root.xpath(site["list_xpath"]): - href = ((card.xpath(".//a[@href][1]/@href") or [""])[0]).strip() - title = ( - ((card.xpath(".//img[@alt][1]/@alt") or [""])[0]).strip() - or ((card.xpath(".//a[@title][1]/@title") or [""])[0]).strip() - ) - pic = ( - ((card.xpath(".//img[@data-src][1]/@data-src") or [""])[0]).strip() - or ((card.xpath(".//img[@src][1]/@src") or [""])[0]).strip() - ) - remarks = "".join(card.xpath(".//*[contains(@class,'module-item-text')][1]//text()")).strip() - if not href or not title or href in seen: - continue - seen.add(href) - items.append( - { - "vod_id": self._encode_site_vod_id(site["id"], href), - "vod_name": title, - "vod_pic": self._build_absolute_url(site["domains"][0], pic), - "vod_remarks": remarks, - "vod_year": "", - "_site": site["id"], - "_detail_path": href, - } - ) - return items - - def categoryContent(self, tid, pg, filter, extend): - site_id = str(tid).replace("site_", "", 1) - site = self._get_site(site_id) - values = extend if isinstance(extend, dict) else {} - category_id = values.get("categoryId") or site["default_categories"][0][0] - html = self._request_with_failover(site, self._build_category_url(site, category_id, pg, values)) - items = self._parse_cards(site, html) - return {"list": items, "page": int(pg), "limit": len(items), "total": int(pg) * 20 + len(items)} - - def _site_rank(self, site_id): - try: - return self.site_priority.index(site_id) - except ValueError: - return 999 - - def _parse_search_cards(self, site, html): - root = self.html(html) - if root is None: - return [] - - items = [] - seen = set() - xpath = site.get("search_xpath") or site["list_xpath"] - for card in root.xpath(xpath): - href = ( - ((card.xpath(".//*[contains(@class,'video-serial')][1]/@href") or [""])[0]).strip() - or ((card.xpath(".//*[@href][1]/@href") or [""])[0]).strip() - ) - title = ( - ((card.xpath(".//*[contains(@class,'video-serial')][1]/@title") or [""])[0]).strip() - or ((card.xpath(".//img[@alt][1]/@alt") or [""])[0]).strip() - or ((card.xpath(".//*[@title][1]/@title") or [""])[0]).strip() - ) - pic = ( - ((card.xpath(".//img[@data-src][1]/@data-src") or [""])[0]).strip() - or ((card.xpath(".//img[@src][1]/@src") or [""])[0]).strip() - ) - remarks = "".join(card.xpath(".//*[contains(@class,'module-item-text')][1]//text()")).strip() - if not href or not title or href in seen: - continue - seen.add(href) - items.append( - { - "vod_id": self._encode_site_vod_id(site["id"], href), - "vod_name": title, - "vod_pic": self._build_absolute_url(site["domains"][0], pic), - "vod_remarks": remarks, - "vod_year": "", - "_site": site["id"], - "_detail_path": href, - } - ) - return items - - def _fetch_site_search(self, site, keyword, pg): - search_path = site["search_url"].format(keyword=quote(str(keyword)), page=int(pg)) - html = self._request_with_failover(site, search_path) - return self._parse_search_cards(site, html) - - def _aggregate_search_results(self, items): - groups = [] - for item in items: - matched_group = None - for group in groups: - if self._is_same_title(group[0], item): - matched_group = group - break - if matched_group is None: - groups.append([item]) - else: - matched_group.append(item) - - result = [] - for group in groups: - group.sort(key=lambda item: self._site_rank(item["_site"])) - primary = group[0] - payload = [ - { - "site": item["_site"], - "path": item["_detail_path"], - "name": item.get("vod_name", ""), - "year": item.get("vod_year", ""), - } - for item in group - ] - result.append( - { - "vod_id": self._encode_aggregate_vod_id(payload), - "vod_name": primary["vod_name"], - "vod_pic": primary.get("vod_pic", ""), - "vod_remarks": primary.get("vod_remarks", ""), - "vod_year": primary.get("vod_year", ""), - } - ) - return result - - def searchContent(self, key, quick, pg="1"): - page = int(pg) - keyword = str(key or "").strip() - if not keyword: - return {"page": page, "total": 0, "list": []} - - all_items = [] - for site in self.sites: - try: - all_items.extend(self._fetch_site_search(site, keyword, page)) - except Exception: - continue - - merged = self._aggregate_search_results(all_items) - return {"page": page, "total": len(merged), "list": merged} - - def _detect_pan_type(self, url): - value = str(url or "").lower() - if "pan.baidu.com" in value: - return "baidu", "百度资源" - if "pan.quark.cn" in value: - return "quark", "夸克资源" - if "drive.uc.cn" in value: - return "uc", "UC资源" - if "alipan.com" in value or "aliyundrive.com" in value: - return "aliyun", "阿里资源" - if "pan.xunlei.com" in value: - return "xunlei", "迅雷资源" - if "123pan.com" in value: - return "a123", "123资源" - if "115.com" in value: - return "a115", "115资源" - if "189.cn" in value: - return "a189", "天翼资源" - if "139.com" in value: - return "a139", "移动云资源" - return "", "" - - def _join_next_sibling_links(self, root, label): - values = [] - labels = root.xpath(f"//*[contains(@class,'video-info-itemtitle') and contains(normalize-space(.), '{label}')]") - for node in labels: - sibling = node.getnext() - if sibling is None: - continue - for text in sibling.xpath(".//a/text()"): - clean = str(text).strip() - if clean: - values.append(clean) - unique = [] - for value in values: - if value not in unique: - unique.append(value) - return ",".join(unique) - - def _join_next_sibling_text(self, root, label): - labels = root.xpath(f"//*[contains(@class,'video-info-itemtitle') and contains(normalize-space(.), '{label}')]") - for node in labels: - sibling = node.getnext() - if sibling is None: - continue - text = "".join(sibling.xpath(".//text()")).strip() - if text: - return text - return "" - - def _parse_detail_page(self, site, detail_path, html): - root = self.html(html) - if root is None: - return { - "vod_name": "", - "vod_pic": "", - "vod_year": "", - "vod_director": "", - "vod_actor": "", - "vod_content": "", - "pan_urls": [], - "_site_name": site["name"], - } - - title = "".join(root.xpath("//*[contains(@class,'page-title')][1]//text()")).strip() - pic = ((root.xpath("//*[contains(@class,'mobile-play')]//img[1]/@data-src") or [""])[0]).strip() - pan_urls = [] - for node in root.xpath(site["detail_pan_xpath"]): - text = "".join(node.xpath(".//text()")).strip() - if text.startswith("http"): - pan_urls.append(text) - - return { - "vod_name": title, - "vod_pic": self._build_absolute_url(site["domains"][0], pic), - "vod_year": "", - "vod_director": self._join_next_sibling_links(root, "导演"), - "vod_actor": self._join_next_sibling_links(root, "主演"), - "vod_content": self._join_next_sibling_text(root, "剧情"), - "pan_urls": pan_urls, - "_site_name": site["name"], - } - - def _fetch_site_detail(self, site, detail_path): - html = self._request_with_failover(site, detail_path) - return self._parse_detail_page(site, detail_path, html) - - def _build_pan_lines(self, detail): - lines = [] - seen = set() - for url in detail.get("pan_urls", []): - pan_type, title = self._detect_pan_type(url) - if not pan_type or url in seen: - continue - seen.add(url) - lines.append( - ( - self.pan_priority.get(pan_type, 999), - f"{pan_type}#{detail['_site_name']}", - f"{title}${url}", - ) - ) - lines.sort(key=lambda item: item[0]) - return lines - - def detailContent(self, ids): - vod_id = ids[0] - if str(vod_id).startswith("agg:"): - payload = self._decode_aggregate_vod_id(vod_id) - details = [] - for item in payload: - site = self._get_site(item["site"]) - try: - details.append(self._fetch_site_detail(site, item["path"])) - except Exception: - continue - - primary = details[0] - all_lines = [] - seen_urls = set() - for detail in details: - for _, line_from, line_url in self._build_pan_lines(detail): - share_url = line_url.split("$", 1)[1] - if share_url in seen_urls: - continue - seen_urls.add(share_url) - all_lines.append((line_from, line_url)) - - return { - "list": [ - { - "vod_id": vod_id, - "vod_name": primary["vod_name"], - "vod_pic": primary["vod_pic"], - "vod_year": primary["vod_year"], - "vod_director": primary["vod_director"], - "vod_actor": primary["vod_actor"], - "vod_content": primary["vod_content"], - "vod_play_from": "$$$".join([item[0] for item in all_lines]), - "vod_play_url": "$$$".join([item[1] for item in all_lines]), - } - ] - } - - info = self._decode_site_vod_id(vod_id) - site = self._get_site(info["site"]) - detail = self._fetch_site_detail(site, info["path"]) - lines = self._build_pan_lines(detail) - return { - "list": [ - { - "vod_id": vod_id, - "vod_name": detail["vod_name"], - "vod_pic": detail["vod_pic"], - "vod_year": detail["vod_year"], - "vod_director": detail["vod_director"], - "vod_actor": detail["vod_actor"], - "vod_content": detail["vod_content"], - "vod_play_from": "$$$".join([item[1] for item in lines]), - "vod_play_url": "$$$".join([item[2] for item in lines]), - } - ] - } - - def playerContent(self, flag, id, vipFlags): - pan_type, _ = self._detect_pan_type(id) - if pan_type: - return {"parse": 0, "playUrl": "", "url": id} - return {"parse": 0, "playUrl": "", "url": ""}