From d4c98c8b7d786b6760363ab7435ccb6e1f4ff4a7 Mon Sep 17 00:00:00 2001 From: XC-Pro <19840+xc-pro@noreply.example.org> Date: Fri, 14 Aug 2026 13:23:53 +0200 Subject: [PATCH] =?UTF-8?q?=E5=88=A0=E9=99=A4=20py/=E8=A5=BF=E7=93=9C?= =?UTF-8?q?=E5=BD=B1=E8=A7=86.py?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- py/西瓜影视.py | 336 ------------------------------------------------- 1 file changed, 336 deletions(-) delete mode 100644 py/西瓜影视.py diff --git a/py/西瓜影视.py b/py/西瓜影视.py deleted file mode 100644 index f9db284..0000000 --- a/py/西瓜影视.py +++ /dev/null @@ -1,336 +0,0 @@ -# -*- coding: utf-8 -*- -# by @PyramidStore AutoGen -import re -import sys -sys.path.append('..') -import json -from urllib.parse import quote -from base.spider import Spider - - -class Spider(Spider): - - def init(self, extend=""): - self.nav_host = 'https://www.xiguadh.com' - self.headers = { - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', - 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', - 'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8', - } - self.host = self._get_host() - - def _get_host(self): - """获取视频站点 URL,失败时从导航页获取""" - default_host = 'https://www.bzzdyy.com' - try: - r = self.fetch(default_host, headers=self.headers, timeout=10, verify=False) - if r.status_code == 200: - return default_host - except Exception: - pass - try: - r = self.fetch(self.nav_host, headers=self.headers, timeout=15, verify=False) - html = r.text - urls = re.findall(r'url:\s*["\']([^"\']+)["\']', html) - for url in urls: - if url.startswith('http') and 'xiguadh' not in url: - return url.rstrip('/') - except Exception: - pass - return default_host - - def getName(self): - return '西瓜影院' - - def isVideoFormat(self, url): - return False - - def manualVideoCheck(self): - return True - - def homeContent(self, filter): - try: - r = self.fetch(self.host, headers=self.headers, timeout=15, verify=False) - html = r.text - # 提取主要分类 - nav_match = re.search(r'', html, re.DOTALL) - if nav_match: - nav_html = nav_match.group(1) - categories = re.findall(r']*>([^<]+)', nav_html) - else: - categories = [] - seen = set() - classes = [] - for tid, name in categories: - if tid not in seen: - seen.add(tid) - classes.append({'type_name': name, 'type_id': tid}) - if not classes: - raise Exception('No categories found') - # 提取首页推荐视频 - videos = self._parse_vodlist(html) - except Exception: - classes = [ - {'type_name': '电影', 'type_id': '20'}, - {'type_name': '连续剧', 'type_id': '37'}, - {'type_name': '动漫', 'type_id': '43'}, - {'type_name': '综艺', 'type_id': '45'}, - {'type_name': 'B站', 'type_id': '47'}, - {'type_name': '人人专区', 'type_id': '60'}, - ] - videos = [] - return {"class": classes, "list": videos} - - def _parse_vodlist(self, html): - """解析视频列表""" - items = re.findall( - r']*data-original="([^"]*)"', - html - ) - videos = [] - for href, title, pic in items: - vod_id = re.search(r'/id/(\d+)\.html', href) - if vod_id: - vod_id = vod_id.group(1) - else: - continue - remark_match = re.search(r'([^<]+)', - html[html.find(href):html.find(href)+500] if href in html else '') - remark = remark_match.group(1) if remark_match else '' - if pic.startswith('/'): - pic = self.host + pic - videos.append({ - 'vod_id': vod_id, - 'vod_name': title, - 'vod_pic': pic, - 'vod_remarks': remark, - }) - return videos - - def homeVideoContent(self): - return '' - - def categoryContent(self, tid, pg, filter, extend): - pg = int(pg) - url = f'{self.host}/index.php/vod/type/id/{tid}/page/{pg}.html' - try: - r = self.fetch(url, headers=self.headers, timeout=15, verify=False) - html = r.text - items = re.findall( - r']*data-original="([^"]*)"', - html - ) - videos = [] - for href, title, pic in items: - vod_id = re.search(r'/id/(\d+)\.html', href) - if vod_id: - vod_id = vod_id.group(1) - else: - continue - remark_match = re.search(r'([^<]+)', - html[html.find(href):html.find(href)+500] if href in html else '') - remark = remark_match.group(1) if remark_match else '' - if pic.startswith('/'): - pic = self.host + pic - videos.append({ - 'vod_id': vod_id, - 'vod_name': title, - 'vod_pic': pic, - 'vod_remarks': remark, - }) - return { - "list": videos, - "page": pg, - "pagecount": 9999, - "limit": 90, - "total": len(videos), - } - except Exception as e: - return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0} - - def detailContent(self, ids): - try: - vod_id = ids[0] if isinstance(ids, list) else ids - url = f'{self.host}/index.php/vod/detail/id/{vod_id}.html' - r = self.fetch(url, headers=self.headers, timeout=15, verify=False) - html = r.text - title_match = re.search(r'

([^<]+)

', html) - title = title_match.group(1).strip() if title_match else '' - self._vod_name = title - pic_match = re.search(r']*>([^<]+)', html) - remark = remark_match.group(1).strip() if remark_match else '' - director_match = re.search(r'导演:(.*?)

', html, re.DOTALL) - director = '' - if director_match: - director = re.sub(r'<[^>]+>', '', director_match.group(1)).strip() - actor_match = re.search(r'主演:([^<]+)', html) - actor = actor_match.group(1).strip() if actor_match else '' - desc_match = re.search(r']*>([^<]+)', html) - desc = desc_match.group(1).strip() if desc_match else '' - play_from = [] - play_url = [] - source_tabs = re.findall(r'
  • ]*>([^<]+)
  • ', html) - for idx, source_name in enumerate(source_tabs): - source_id = idx + 1 - episodes_match = re.search( - f'
    ]*>.*?', - html, re.DOTALL - ) - if episodes_match: - episodes = re.findall(r'([^<]+)', episodes_match.group(1)) - episode_list = [] - for ep_url, ep_name in episodes: - episode_list.append(f'{ep_name}${self.host}{ep_url}') - play_from.append(source_name) - play_url.append('#'.join(episode_list)) - vod_play_from = '$$$'.join(play_from) if play_from else '默认' - vod_play_url = '$$$'.join(play_url) if play_url else '' - vod = { - 'vod_id': vod_id, - 'vod_name': title, - 'vod_pic': pic, - 'vod_year': year, - 'vod_area': area, - 'vod_remarks': remark, - 'vod_director': director, - 'vod_actor': actor, - 'vod_content': desc, - 'vod_play_from': vod_play_from, - 'vod_play_url': vod_play_url, - } - return {"list": [vod]} - except Exception as e: - return {"list": []} - - def searchContent(self, key, quick, pg="1"): - pg = int(pg) - url = f'{self.host}/index.php/vod/search/wd/{quote(key)}.html' - try: - r = self.fetch(url, headers=self.headers, timeout=15, verify=False) - html = r.text - items = re.findall( - r']*data-original="([^"]*)"', - html - ) - videos = [] - for href, title, pic in items: - vod_id = re.search(r'/id/(\d+)\.html', href) - if vod_id: - vod_id = vod_id.group(1) - else: - continue - remark_match = re.search(r'([^<]+)', - html[html.find(href):html.find(href)+500] if href in html else '') - remark = remark_match.group(1) if remark_match else '' - if pic.startswith('/'): - pic = self.host + pic - videos.append({ - 'vod_id': vod_id, - 'vod_name': title, - 'vod_pic': pic, - 'vod_remarks': remark, - }) - return { - "list": videos, - "page": pg, - "pagecount": 9999, - "limit": 90, - "total": len(videos), - } - except Exception as e: - return {"list": [], "page": pg, "pagecount": 1, "limit": 90, "total": 0} - - def _clean_vod_name(self, name): - import re - if not name: - return '' - cleaned = re.sub(r'第\s*\d+\s*[集話话章部期]', '', name) - cleaned = re.sub(r'EP\s*\d+', '', cleaned, flags=re.IGNORECASE) - cleaned = re.sub(r'全\d+集', '', cleaned) - cleaned = re.sub(r'更新至\d+集', '', cleaned) - cleaned = re.sub(r'\d+集全', '', cleaned) - cleaned = re.sub(r'[((].*?[))]', '', cleaned) - cleaned = re.sub(r'\s*-\s*.*$', '', cleaned) - cleaned = re.sub(r'\s+', ' ', cleaned) - cleaned = re.sub(r'^[\s\-_,.,。、]+|[\s\-_,.,。、]+$', '', cleaned) - return cleaned.strip() - - def _build_danmaku_url(self, vod_name, vod_index=''): - import re - idx = 0 - if vod_index: - s = str(vod_index).strip() - m = re.search(r'第\s*(\d+)\s*[集話话章部期]', s) - if m: - idx = int(m.group(1)) - else: - m = re.search(r'(\d+)', s) - if m: - idx = int(m.group(1)) - cleaned_name = self._clean_vod_name(vod_name) - params = [] - if cleaned_name: - params.append(f'vodName={quote(cleaned_name)}') - params.append(f'vodIndex={idx}') - query = '&'.join(params) - return f'http://127.0.0.1:9978/proxy?do=appdanmu&{query}' - - def playerContent(self, flag, id, vipFlags): - try: - ep_name = '' - vod_index = '' - if '$' in id: - parts = id.split('$', 1) - ep_name = parts[0] - url = parts[1] if len(parts) > 1 else '' - else: - url = id if id.startswith('http') else f'{self.host}{id}' - # 从 URL 中提取集数 (nid 参数) - nid_match = re.search(r'nid/(\d+)\.html', url) - if nid_match: - vod_index = nid_match.group(1) - danmaku_url = self._build_danmaku_url(self._vod_name, vod_index) - r = self.fetch(url, headers=self.headers, timeout=15, verify=False) - html = r.text - iframe_match = re.search(r']+src="([^"]+)"', html) - if iframe_match: - iframe_url = iframe_match.group(1) - if not iframe_url.startswith('http'): - iframe_url = self.host + iframe_url - return { - "parse": 1, - "url": iframe_url, - "header": self.headers, - "danmaku": danmaku_url - } - src_match = re.search(r'(https?://[^"\'<>\s]+\.m3u8[^"\'<>\s]*)', html) - if src_match: - return { - "parse": 0, - "url": src_match.group(1), - "header": self.headers, - "danmaku": danmaku_url - } - return { - "parse": 1, - "url": url, - "header": self.headers, - "danmaku": danmaku_url - } - except Exception as e: - danmaku_url = self._build_danmaku_url(self._vod_name, '') - return {"parse": 1, "url": id, "header": {}, "danmaku": danmaku_url} - - def localProxy(self, param): - return [200, {}, ""] - - def destroy(self): - pass