feat: add rrdy category parsing
This commit is contained in:
@@ -1,6 +1,7 @@
|
|||||||
import unittest
|
import unittest
|
||||||
from importlib.machinery import SourceFileLoader
|
from importlib.machinery import SourceFileLoader
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[1]
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
@@ -47,3 +48,48 @@ class TestRenRenDianYingSpider(unittest.TestCase):
|
|||||||
self.assertEqual(self.spider._normalize_title("《繁花》"), "繁花")
|
self.assertEqual(self.spider._normalize_title("《繁花》"), "繁花")
|
||||||
self.assertEqual(self.spider._normalize_title("「诛仙」特别篇"), "诛仙")
|
self.assertEqual(self.spider._normalize_title("「诛仙」特别篇"), "诛仙")
|
||||||
self.assertEqual(self.spider._normalize_title("普通标题"), "普通标题")
|
self.assertEqual(self.spider._normalize_title("普通标题"), "普通标题")
|
||||||
|
|
||||||
|
def test_parse_cards_extracts_expected_fields(self):
|
||||||
|
html = """
|
||||||
|
<ul id="movielist">
|
||||||
|
<li>
|
||||||
|
<div class="pure-img"><img class="pure-img" data-original="/poster.jpg" /></div>
|
||||||
|
<div class="intro">
|
||||||
|
<h2><a href="/movie/123.html" title="《示例电影》">《示例电影》</a></h2>
|
||||||
|
</div>
|
||||||
|
<div class="dou"><b>8.8</b></div>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
"""
|
||||||
|
self.assertEqual(
|
||||||
|
self.spider._parse_cards(html),
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"vod_id": "/movie/123.html",
|
||||||
|
"vod_name": "示例电影",
|
||||||
|
"vod_pic": "https://www.rrdynb.com/poster.jpg",
|
||||||
|
"vod_remarks": "8.8",
|
||||||
|
}
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
@patch.object(Spider, "_request_html")
|
||||||
|
def test_category_content_builds_reference_url_and_page_payload(self, mock_request_html):
|
||||||
|
mock_request_html.return_value = """
|
||||||
|
<ul id="movielist">
|
||||||
|
<li>
|
||||||
|
<img class="pure-img" data-original="/cate.jpg" />
|
||||||
|
<div class="intro"><h2><a href="/movie/456.html" title="分类影片">分类影片</a></h2></div>
|
||||||
|
<div class="dou"><b>更新中</b></div>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
"""
|
||||||
|
result = self.spider.categoryContent("movie/list_2", "3", False, {})
|
||||||
|
self.assertEqual(
|
||||||
|
mock_request_html.call_args.args[0],
|
||||||
|
"https://www.rrdynb.com/movie/list_2_3.html",
|
||||||
|
)
|
||||||
|
self.assertEqual(result["page"], 3)
|
||||||
|
self.assertEqual(result["limit"], 1)
|
||||||
|
self.assertEqual(result["list"][0]["vod_name"], "分类影片")
|
||||||
|
self.assertNotIn("pagecount", result)
|
||||||
|
|||||||
+50
@@ -78,3 +78,53 @@ class Spider(BaseSpider):
|
|||||||
if any(re.search(pattern, raw, re.I) for pattern in self.excluded_pan_patterns):
|
if any(re.search(pattern, raw, re.I) for pattern in self.excluded_pan_patterns):
|
||||||
return False
|
return False
|
||||||
return any(re.search(pattern, raw, re.I) for pattern in self.supported_pan_patterns)
|
return any(re.search(pattern, raw, re.I) for pattern in self.supported_pan_patterns)
|
||||||
|
|
||||||
|
def _request_html(self, path_or_url):
|
||||||
|
target = path_or_url if str(path_or_url).startswith("http") else self._build_url(path_or_url)
|
||||||
|
response = self.fetch(target, headers=dict(self.headers), timeout=10)
|
||||||
|
if response.status_code != 200:
|
||||||
|
return ""
|
||||||
|
return response.text or ""
|
||||||
|
|
||||||
|
def _clean_text(self, text):
|
||||||
|
return re.sub(r"\s+", " ", str(text or "").replace("\xa0", " ")).strip()
|
||||||
|
|
||||||
|
def _parse_cards(self, html):
|
||||||
|
root = self.html(html)
|
||||||
|
if root is None:
|
||||||
|
return []
|
||||||
|
|
||||||
|
items = []
|
||||||
|
seen = set()
|
||||||
|
for node in root.xpath("//*[@id='movielist']//li"):
|
||||||
|
href = "".join(node.xpath(".//*[contains(@class,'intro')]//h2//a[1]/@href")).strip()
|
||||||
|
title = (
|
||||||
|
"".join(node.xpath(".//*[contains(@class,'intro')]//h2//a[1]/@title")).strip()
|
||||||
|
or "".join(node.xpath(".//*[contains(@class,'intro')]//h2//a[1]//text()")).strip()
|
||||||
|
)
|
||||||
|
pic = (
|
||||||
|
"".join(node.xpath(".//*[contains(@class,'pure-img')][1]/@data-original")).strip()
|
||||||
|
or "".join(node.xpath(".//*[contains(@class,'pure-img')][1]/@src")).strip()
|
||||||
|
or "".join(node.xpath(".//*[contains(@class,'pure-img')]//img[1]/@data-original")).strip()
|
||||||
|
or "".join(node.xpath(".//*[contains(@class,'pure-img')]//img[1]/@src")).strip()
|
||||||
|
)
|
||||||
|
remarks = self._clean_text("".join(node.xpath(".//*[contains(@class,'dou')][1]//text()")))
|
||||||
|
if not href or not title or href in seen:
|
||||||
|
continue
|
||||||
|
seen.add(href)
|
||||||
|
items.append(
|
||||||
|
{
|
||||||
|
"vod_id": href,
|
||||||
|
"vod_name": self._normalize_title(title),
|
||||||
|
"vod_pic": self._build_url(pic),
|
||||||
|
"vod_remarks": remarks,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return items
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
page = int(pg)
|
||||||
|
class_path = str(tid or "").lstrip("/")
|
||||||
|
url = self._build_url(f"{class_path}_{page}.html")
|
||||||
|
items = self._parse_cards(self._request_html(url))
|
||||||
|
return {"page": page, "limit": len(items), "total": page * 20 + len(items), "list": items}
|
||||||
|
|||||||
Reference in New Issue
Block a user