feat: add sjmusic category and search parsing
This commit is contained in:
@@ -30,6 +30,50 @@ HOME_HTML = """
|
|||||||
</html>
|
</html>
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
PLAYLIST_HTML = """
|
||||||
|
<html><body>
|
||||||
|
<ul class="video_list">
|
||||||
|
<li>
|
||||||
|
<div class="name"><a href="/playlist/gedan001.html">华语经典</a></div>
|
||||||
|
<div class="pic"><img src="/img/list.jpg"></div>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</body></html>
|
||||||
|
"""
|
||||||
|
|
||||||
|
SINGER_HTML = """
|
||||||
|
<html><body>
|
||||||
|
<ul class="singer_list">
|
||||||
|
<li>
|
||||||
|
<div class="pic"><a href="/singer/zhoujielun.html"><img src="/img/singer.jpg"></a></div>
|
||||||
|
<div class="name"><a>周杰伦</a></div>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</body></html>
|
||||||
|
"""
|
||||||
|
|
||||||
|
MV_HTML = """
|
||||||
|
<html><body>
|
||||||
|
<ul class="video_list">
|
||||||
|
<li>
|
||||||
|
<div class="name"><a href="/mp4/999.html">晴天MV</a></div>
|
||||||
|
<div class="pic"><img src="/img/mv2.jpg"></div>
|
||||||
|
</li>
|
||||||
|
</ul>
|
||||||
|
</body></html>
|
||||||
|
"""
|
||||||
|
|
||||||
|
SEARCH_HTML = """
|
||||||
|
<html><body>
|
||||||
|
<ul class="play_list">
|
||||||
|
<li><div class="name"><a href="/mp3/777.html">七里香</a></div><img src="/img/1.jpg"></li>
|
||||||
|
<li><div class="name"><a href="/mp4/778.html">七里香MV</a></div><img src="/img/2.jpg"></li>
|
||||||
|
<li><div class="name"><a href="/playlist/top777.html">周董精选</a></div><img src="/img/3.jpg"></li>
|
||||||
|
<li><div class="name"><a href="/singer/jay.html">周杰伦</a></div><img src="/img/4.jpg"></li>
|
||||||
|
</ul>
|
||||||
|
</body></html>
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
class TestSJMusicSpider(unittest.TestCase):
|
class TestSJMusicSpider(unittest.TestCase):
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
@@ -56,6 +100,51 @@ class TestSJMusicSpider(unittest.TestCase):
|
|||||||
result = self.spider.homeVideoContent()
|
result = self.spider.homeVideoContent()
|
||||||
self.assertEqual([item["vod_id"] for item in result["list"]], ["song:123", "mv:456"])
|
self.assertEqual([item["vod_id"] for item in result["list"]], ["song:123", "mv:456"])
|
||||||
|
|
||||||
|
@patch.object(Spider, "fetch")
|
||||||
|
def test_category_content_supports_rank_playlist_singer_and_mv(self, mock_fetch):
|
||||||
|
mock_fetch.side_effect = [
|
||||||
|
SimpleNamespace(status_code=200, text=PLAYLIST_HTML),
|
||||||
|
SimpleNamespace(status_code=200, text=SINGER_HTML),
|
||||||
|
SimpleNamespace(status_code=200, text=MV_HTML),
|
||||||
|
]
|
||||||
|
rank_result = self.spider.categoryContent("rank_list", "1", False, {})
|
||||||
|
playlist_result = self.spider.categoryContent("playlist", "1", False, {"lang": "index"})
|
||||||
|
singer_result = self.spider.categoryContent(
|
||||||
|
"singer",
|
||||||
|
"1",
|
||||||
|
False,
|
||||||
|
{"sex": "girl", "area": "huayu", "char": "index"},
|
||||||
|
)
|
||||||
|
mv_result = self.spider.categoryContent(
|
||||||
|
"mv",
|
||||||
|
"1",
|
||||||
|
False,
|
||||||
|
{"area": "index", "type": "index", "sort": "new"},
|
||||||
|
)
|
||||||
|
self.assertTrue(rank_result["list"][0]["vod_id"].startswith("rank:"))
|
||||||
|
self.assertEqual(playlist_result["list"][0]["vod_id"], "playlist:gedan001")
|
||||||
|
self.assertEqual(singer_result["list"][0]["vod_id"], "singer:zhoujielun")
|
||||||
|
self.assertEqual(mv_result["list"][0]["vod_id"], "mv:999")
|
||||||
|
|
||||||
|
@patch.object(Spider, "fetch")
|
||||||
|
def test_search_content_maps_link_types_and_blank_keyword(self, mock_fetch):
|
||||||
|
mock_fetch.return_value = SimpleNamespace(status_code=200, text=SEARCH_HTML)
|
||||||
|
result = self.spider.searchContent("周杰伦", False, "1")
|
||||||
|
self.assertEqual(
|
||||||
|
[item["vod_id"] for item in result["list"]],
|
||||||
|
["song:777", "mv:778", "playlist:top777", "singer:jay"],
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
self.spider.searchContent("", False, "1"),
|
||||||
|
{"page": 1, "limit": 0, "total": 0, "list": []},
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_category_content_returns_empty_for_unknown_tid(self):
|
||||||
|
self.assertEqual(
|
||||||
|
self.spider.categoryContent("unknown", "1", False, {}),
|
||||||
|
{"page": 1, "limit": 0, "total": 0, "list": []},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
+90
-2
@@ -27,6 +27,11 @@ class Spider(BaseSpider):
|
|||||||
{"type_id": "singer", "type_name": "歌手"},
|
{"type_id": "singer", "type_name": "歌手"},
|
||||||
{"type_id": "mv", "type_name": "MV"},
|
{"type_id": "mv", "type_name": "MV"},
|
||||||
]
|
]
|
||||||
|
self.rank_list = [
|
||||||
|
("rise", "音乐飙升榜"),
|
||||||
|
("new", "新歌排行榜"),
|
||||||
|
("top", "Top热歌榜"),
|
||||||
|
]
|
||||||
|
|
||||||
def init(self, extend=""):
|
def init(self, extend=""):
|
||||||
return None
|
return None
|
||||||
@@ -50,11 +55,22 @@ class Spider(BaseSpider):
|
|||||||
def _encode_vod_id(self, href):
|
def _encode_vod_id(self, href):
|
||||||
href = str(href or "")
|
href = str(href or "")
|
||||||
if "/mp3/" in href:
|
if "/mp3/" in href:
|
||||||
return "song:" + href.rsplit("/", 1)[-1].replace(".html", "")
|
return "song:" + self._extract_site_id(href, "mp3")
|
||||||
if "/mp4/" in href:
|
if "/mp4/" in href:
|
||||||
return "mv:" + href.rsplit("/", 1)[-1].replace(".html", "")
|
return "mv:" + self._extract_site_id(href, "mp4")
|
||||||
|
if "/playlist/" in href:
|
||||||
|
return "playlist:" + self._extract_site_id(href, "playlist")
|
||||||
|
if "/singer/" in href:
|
||||||
|
return "singer:" + self._extract_site_id(href, "singer")
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
|
def _extract_site_id(self, href, prefix):
|
||||||
|
text = str(href or "")
|
||||||
|
marker = f"/{prefix}/"
|
||||||
|
if marker not in text:
|
||||||
|
return ""
|
||||||
|
return text.split(marker, 1)[1].split(".html", 1)[0].split("/", 1)[0]
|
||||||
|
|
||||||
def _build_filters(self):
|
def _build_filters(self):
|
||||||
return {"singer": [], "mv": [], "playlist": []}
|
return {"singer": [], "mv": [], "playlist": []}
|
||||||
|
|
||||||
@@ -81,9 +97,81 @@ class Spider(BaseSpider):
|
|||||||
)
|
)
|
||||||
return items
|
return items
|
||||||
|
|
||||||
|
def _page_result(self, items, pg):
|
||||||
|
page = int(pg)
|
||||||
|
return {"page": page, "limit": len(items), "total": len(items), "list": items}
|
||||||
|
|
||||||
|
def _parse_list_cards(self, html, expected_prefixes):
|
||||||
|
root = self._load_html(html)
|
||||||
|
items = []
|
||||||
|
seen = set()
|
||||||
|
for node in root.xpath("//li"):
|
||||||
|
href = "".join(node.xpath(".//a[1]/@href")).strip()
|
||||||
|
vod_id = self._encode_vod_id(href)
|
||||||
|
if not vod_id or vod_id in seen:
|
||||||
|
continue
|
||||||
|
if expected_prefixes and not any(vod_id.startswith(prefix) for prefix in expected_prefixes):
|
||||||
|
continue
|
||||||
|
seen.add(vod_id)
|
||||||
|
items.append(
|
||||||
|
{
|
||||||
|
"vod_id": vod_id,
|
||||||
|
"vod_name": self._clean_text(
|
||||||
|
"".join(node.xpath(".//div[contains(@class,'name')]//text()"))
|
||||||
|
or "".join(node.xpath(".//a[1]//text()"))
|
||||||
|
),
|
||||||
|
"vod_pic": self._build_url("".join(node.xpath(".//img[1]/@src"))),
|
||||||
|
"vod_remarks": "",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return items
|
||||||
|
|
||||||
def homeContent(self, filter):
|
def homeContent(self, filter):
|
||||||
items = self._parse_home_items(self._fetch_html("/"))
|
items = self._parse_home_items(self._fetch_html("/"))
|
||||||
return {"class": list(self.classes), "filters": self._build_filters(), "list": items}
|
return {"class": list(self.classes), "filters": self._build_filters(), "list": items}
|
||||||
|
|
||||||
def homeVideoContent(self):
|
def homeVideoContent(self):
|
||||||
return {"list": self.homeContent(False).get("list", [])}
|
return {"list": self.homeContent(False).get("list", [])}
|
||||||
|
|
||||||
|
def categoryContent(self, tid, pg, filter, extend):
|
||||||
|
if tid == "home":
|
||||||
|
return self._page_result(self.homeContent(False).get("list", []), 1)
|
||||||
|
if tid == "rank_list":
|
||||||
|
return self._page_result(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"vod_id": f"rank:{rank_id}",
|
||||||
|
"vod_name": title,
|
||||||
|
"vod_pic": "",
|
||||||
|
"vod_remarks": "排行榜",
|
||||||
|
}
|
||||||
|
for rank_id, title in self.rank_list
|
||||||
|
],
|
||||||
|
pg,
|
||||||
|
)
|
||||||
|
if tid == "playlist":
|
||||||
|
return self._page_result(
|
||||||
|
self._parse_list_cards(self._fetch_html("/playlists/index.html"), ["playlist:"]),
|
||||||
|
pg,
|
||||||
|
)
|
||||||
|
if tid == "singer":
|
||||||
|
return self._page_result(
|
||||||
|
self._parse_list_cards(self._fetch_html("/singerlist/huayu/girl/index.html"), ["singer:"]),
|
||||||
|
pg,
|
||||||
|
)
|
||||||
|
if tid == "mv":
|
||||||
|
return self._page_result(
|
||||||
|
self._parse_list_cards(self._fetch_html("/mvlist/index/index/new.html"), ["mv:"]),
|
||||||
|
pg,
|
||||||
|
)
|
||||||
|
return {"page": 1, "limit": 0, "total": 0, "list": []}
|
||||||
|
|
||||||
|
def searchContent(self, key, quick, pg="1"):
|
||||||
|
keyword = self._clean_text(key)
|
||||||
|
if not keyword:
|
||||||
|
return {"page": 1, "limit": 0, "total": 0, "list": []}
|
||||||
|
items = self._parse_list_cards(
|
||||||
|
self._fetch_html(f"/so.php?wd={keyword}&page={pg}"),
|
||||||
|
["song:", "mv:", "playlist:", "singer:"],
|
||||||
|
)
|
||||||
|
return self._page_result(items, pg)
|
||||||
|
|||||||
Reference in New Issue
Block a user