diff --git a/py/tests/test_腾讯视频.py b/py/tests/test_腾讯视频.py new file mode 100644 index 0000000..bbc7a2e --- /dev/null +++ b/py/tests/test_腾讯视频.py @@ -0,0 +1,76 @@ +import unittest +from importlib.machinery import SourceFileLoader +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + + +ROOT = Path(__file__).resolve().parents[1] +MODULE = SourceFileLoader("tengxun_spider", str(ROOT / "腾讯视频.py")).load_module() +Spider = MODULE.Spider + + +HOME_HTML = """ +
+ 海底小纵队 + 更新至10集 +
+
+ 熊出没 + 全52集 +
+""" + + +class TestTencentSpider(unittest.TestCase): + def setUp(self): + Spider._instance = None + self.spider = Spider() + self.spider.init() + + def test_parse_list_items_extracts_cards(self): + cards = self.spider._parse_list_items(HOME_HTML, with_channel=False) + self.assertEqual( + cards, + [ + { + "vod_id": "/x/cover/mzc00200abc1111.html", + "vod_name": "海底小纵队", + "vod_pic": "https://img.test/a.jpg", + "vod_remarks": "更新至10集", + }, + { + "vod_id": "/x/cover/mzc00200abc2222.html", + "vod_name": "熊出没", + "vod_pic": "https://img.test/b.jpg", + "vod_remarks": "全52集", + }, + ], + ) + + @patch.object(Spider, "fetch") + def test_home_content_returns_fixed_classes_and_top_20_cards(self, mock_fetch): + mock_fetch.return_value = SimpleNamespace(text=HOME_HTML) + result = self.spider.homeContent(False) + self.assertEqual( + [item["type_id"] for item in result["class"]], + ["choice", "movie", "tv", "variety", "cartoon", "child", "doco"], + ) + self.assertEqual(result["list"][0]["vod_name"], "海底小纵队") + self.assertNotIn("filters", result) + + def test_player_content_passthroughs_raw_url(self): + result = self.spider.playerContent("腾讯视频", "https://v.qq.com/x/cover/demo.html", {}) + self.assertEqual( + result, + { + "parse": 1, + "jx": 1, + "url": "https://v.qq.com/x/cover/demo.html", + "header": {"User-Agent": "PC_UA"}, + }, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/py/腾讯视频.py b/py/腾讯视频.py new file mode 100644 index 0000000..b6b56d2 --- /dev/null +++ b/py/腾讯视频.py @@ -0,0 +1,76 @@ +# coding=utf-8 +import re +import sys + +from base.spider import Spider as BaseSpider + +sys.path.append("..") + + +class Spider(BaseSpider): + def __init__(self): + self.name = "腾讯视频" + self.base_host = "https://v.qq.com" + self.header = {"User-Agent": "PC_UA"} + self.classes = [ + {"type_id": "choice", "type_name": "精选"}, + {"type_id": "movie", "type_name": "电影"}, + {"type_id": "tv", "type_name": "电视剧"}, + {"type_id": "variety", "type_name": "综艺"}, + {"type_id": "cartoon", "type_name": "动漫"}, + {"type_id": "child", "type_name": "少儿"}, + {"type_id": "doco", "type_name": "纪录片"}, + ] + + def init(self, extend=""): + return None + + def getName(self): + return self.name + + def _headers(self): + return dict(self.header) + + def _parse_list_items(self, html, with_channel=False, channel_id=""): + videos = [] + list_items = re.findall( + r']*class=["\']?list_item["\']?[^>]*>([\s\S]*?)', + str(html or ""), + re.I, + ) + for item in list_items: + title_match = re.search(r']*alt=["\']?([^"\']*)["\']?', item, re.I) + pic_match = re.search(r']*src=["\']?([^"\'\s>]+)["\']?', item, re.I) + desc_values = [ + re.sub(r"<[^>]*>", "", value).strip() + for value in re.findall(r']*>([\s\S]*?)', item, re.I) + ] + url_match = re.search(r']*data-float=["\']?([^"\'\s>]+)["\']?', item, re.I) + if not title_match or not pic_match: + continue + source_id = url_match.group(1) if url_match else "" + vod_id = f"{channel_id}${source_id}" if with_channel else source_id + videos.append( + { + "vod_id": vod_id, + "vod_name": title_match.group(1) or "", + "vod_pic": pic_match.group(1) or "", + "vod_remarks": next((value for value in reversed(desc_values) if value), ""), + } + ) + return videos + + def homeContent(self, filter): + url = ( + f"{self.base_host}/x/bu/pagesheet/list?_all=1&append=1&channel=cartoon" + "&listpage=1&offset=0&pagesize=21&iarea=-1&sort=18" + ) + response = self.fetch(url, headers=self._headers()) + videos = self._parse_list_items(getattr(response, "text", ""), with_channel=False) + return {"class": self.classes, "list": videos[:20]} + + def homeVideoContent(self): + return {"list": self.homeContent(False).get("list", [])} + + def playerContent(self, flag, id, vipFlags): + return {"parse": 1, "jx": 1, "url": id, "header": self._headers()}