From dbee419c56fffa519a2f247ed681a684c37f486c Mon Sep 17 00:00:00 2001 From: yigedashu <123646671+yigedashu@users.noreply.github.com> Date: Sat, 7 Feb 2026 16:39:52 +0800 Subject: [PATCH] Add files via upload --- lib/a好色TV[密].py | 591 +++++++++++++++++++++++++++++++++++++++++ lib/b好色™ Tv[密].py | 543 +++++++++++++++++++++++++++++++++++++ lib/c色播聚合[密].py | 82 ++++++ lib/d随机小姐姐[密].py | 154 +++++++++++ lib/e小红薯App[密].py | 196 ++++++++++++++ lib/f妲己[密].py | 260 ++++++++++++++++++ lib/g瑟瑟直播[密].py | 267 +++++++++++++++++++ lib/h免翻好色TV[密].py | 543 +++++++++++++++++++++++++++++++++++++ lib/i推特APP[密].py | 256 ++++++++++++++++++ lib/j黄色仓库[密].py | 455 +++++++++++++++++++++++++++++++ lib/k神秘影院[密].py | 271 +++++++++++++++++++ 11 files changed, 3618 insertions(+) create mode 100644 lib/a好色TV[密].py create mode 100644 lib/b好色™ Tv[密].py create mode 100644 lib/c色播聚合[密].py create mode 100644 lib/d随机小姐姐[密].py create mode 100644 lib/e小红薯App[密].py create mode 100644 lib/f妲己[密].py create mode 100644 lib/g瑟瑟直播[密].py create mode 100644 lib/h免翻好色TV[密].py create mode 100644 lib/i推特APP[密].py create mode 100644 lib/j黄色仓库[密].py create mode 100644 lib/k神秘影院[密].py diff --git a/lib/a好色TV[密].py b/lib/a好色TV[密].py new file mode 100644 index 0000000..7fbf76d --- /dev/null +++ b/lib/a好色TV[密].py @@ -0,0 +1,591 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '好色TV[密]', + lang: 'hipy' +}) +""" + +import re +import sys +import urllib.parse +import threading +import time +import requests +from pyquery import PyQuery as pq + +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + def __init__(self): + # 基础配置 + self.name = '好色TV(优)' + self.host = 'https://m.ml0987.online/' + self.candidate_hosts = [ + "https://m.ml0987.online/" + ] + self.headers = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36', + 'Referer': self.host + } + self.timeout = 5000 + + # 分类映射 + self.class_map = { + '视频': {'type_id': 'list', 'url_suffix': ''}, + '周榜': {'type_id': 'top7', 'url_suffix': 'top7'}, + '月榜': {'type_id': 'top', 'url_suffix': 'top'}, + '5分钟+': {'type_id': '5min', 'url_suffix': '5min'}, + '10分钟+': {'type_id': 'long', 'url_suffix': 'long'} + } + + def getName(self): + return self.name + + def init(self, extend=""): + # 尝试获取最快可用域名 + self.host = self.get_fastest_host() + self.headers['Referer'] = self.host + + def isVideoFormat(self, url): + if not url: + return False + return any(fmt in url.lower() for fmt in ['.mp4', '.m3u8', '.flv', '.avi']) + + def manualVideoCheck(self): + def check(url): + if not self.isVideoFormat(url): + return False + try: + resp = self.fetch(url, headers=self.headers, method='HEAD', timeout=3) + return resp.status_code in (200, 302) and 'video' in resp.headers.get('Content-Type', '') + except: + return False + return check + + def get_fastest_host(self): + """测试候选域名,返回最快可用的""" + results = {} + threads = [] + + def test_host(url): + try: + start_time = time.time() + resp = requests.head(url, headers=self.headers, timeout=2, allow_redirects=False) + if resp.status_code in (200, 301, 302): + delay = (time.time() - start_time) * 1000 + results[url] = delay + else: + results[url] = float('inf') + except: + results[url] = float('inf') + + for host in self.candidate_hosts: + t = threading.Thread(target=test_host, args=(host,)) + threads.append(t) + t.start() + for t in threads: + t.join() + + valid_hosts = [(h, d) for h, d in results.items() if d != float('inf')] + return valid_hosts[0][0] if valid_hosts else self.candidate_hosts[0] + + def homeContent(self, filter): + result = {} + # 构造分类列表 + classes = [] + for name, info in self.class_map.items(): + classes.append({ + 'type_name': name, + 'type_id': info['type_id'] + }) + result['class'] = classes + + try: + # 获取首页内容 + html = self.fetch_with_retry(self.host, retry=2, timeout=5).text + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + # 提取图片URL + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + # 提取时长备注 + desc = item('.duration').text().strip() or '未知' + + # 提取视频ID + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析首页视频项失败: {e}") + continue + + result['list'] = vlist + except Exception as e: + print(f"首页解析失败: {e}") + result['list'] = [] + return result + + def homeVideoContent(self): + return [] + + def categoryContent(self, tid, pg, filter, extend): + result = {} + try: + # 匹配分类信息 + cate_info = None + for name, info in self.class_map.items(): + if info['type_id'] == tid: + cate_info = info + break + + if not cate_info: + result['list'] = [] + return result + + # 关键修复:区分视频分类与其他分类的URL格式 + if tid == 'list': # 视频分类(type_id为list) + url = f"{self.host}list-{pg}.htm" # 格式:list-1.htm、list-2.htm + else: # 其他分类(周榜/月榜等):xxx_list-{pg}.htm + url = f"{self.host}{cate_info['url_suffix']}_list-{pg}.htm" + + # 请求分类页 + html = self.fetch(url, headers=self.headers, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析分类视频项失败: {e}") + continue + + # 提取总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + except: + pagecount = 1 + + result['list'] = vlist + result['page'] = pg + result['pagecount'] = pagecount + result['limit'] = len(vlist) + result['total'] = 999999 + except Exception as e: + print(f"分类解析失败: {e}") + result['list'] = [] + result['page'] = pg + result['pagecount'] = 1 + result['limit'] = 0 + result['total'] = 0 + return result + + def detailContent(self, ids): + try: + if not ids or not ids[0]: + return {'list': []} + + vod_id = ids[0].strip() + if not vod_id.endswith('.htm'): + vod_id += '.htm' + url = f"{self.host}{vod_id.lstrip('/')}" + + html = self.fetch_with_retry(url, retry=2, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取标题 + title = data('.panel-title, .video-title, h1').text().strip() or '未知标题' + + # 提取封面图 + vod_pic = '' + poster_style = data('.vjs-poster').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', poster_style) + if pic_match: + vod_pic = pic_match.group(1) + if not vod_pic: + vod_pic = data('.video-pic img, .vjs-poster img, .thumbnail img').attr('src') or '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host}{vod_pic.lstrip('/')}" + + # 提取时长和观看量 + duration = '未知' + views = '未知' + info_items = data('.panel-body .col-md-3, .video-info .info-item, .info p') + for item in info_items.items(): + text = item.text().strip() + if '时长' in text or 'duration' in text.lower(): + duration = text.replace('时长:', '').replace('时长', '').strip() + elif '观看' in text or 'views' in text.lower(): + views_match = re.search(r'(\d+\.?\d*[kK]?)次观看', text) + if views_match: + views = views_match.group(1) + else: + views = text.replace('观看:', '').replace('观看', '').strip() + remarks = f"{duration} | {views}" + + # 简化版播放线路提取 - 直接基于找到的链接生成第二条线路 + video_urls = [] + + # 首先尝试提取任意一个m3u8链接 + found_url = None + + # 方法1: 从video标签提取 + video_element = data('video#video-play_html5_api') + if video_element: + video_src = video_element.attr('src') + if video_src and '.m3u8' in video_src: + found_url = video_src + print(f"从video标签找到链接: {found_url}") + + # 方法2: 从source标签提取 + if not found_url: + source_element = data('source#video-source') + if source_element: + source_src = source_element.attr('src') + if source_src and '.m3u8' in source_src: + found_url = source_src + print(f"从source标签找到链接: {found_url}") + + # 方法3: 正则搜索 + if not found_url: + m3u8_matches = re.findall(r'https?://[^\s"\']+\.m3u8[^\s"\']*', html) + if m3u8_matches: + found_url = m3u8_matches[0] + print(f"通过正则找到链接: {found_url}") + + # 清理找到的URL + if found_url: + found_url = found_url.replace('\\/', '/').replace('\\u002F', '/').replace('\\"', '') + if not found_url.startswith('http'): + found_url = f"https:{found_url}" if found_url.startswith('//') else f"https://{found_url}" + + # 关键修改:将HD线路放在前面 + if 'hdcdn.online' in found_url: + # 如果找到的是HD线路,直接添加,然后生成主线路 + video_urls.append(found_url) + second_url = found_url.replace('hdcdn.online', 'hsex.tv') + video_urls.append(second_url) + print(f"生成主线路: {second_url}") + elif 'hsex.tv' in found_url: + # 如果找到的是主线路,先生成HD线路,再添加主线路 + second_url = found_url.replace('hsex.tv', 'hdcdn.online') + video_urls.append(second_url) # HD线路在前 + video_urls.append(found_url) # 主线路在后 + print(f"生成HD线路: {second_url}") + else: + # 如果是其他域名,直接添加 + video_urls.append(found_url) + video_urls.append(found_url) # 复制一份作为备用 + print(f"复制备用线路: {found_url}") + + print(f"最终播放线路: {video_urls}") + + # 构建播放源信息 - 确保HD线路优先显示 + play_from = [] + play_url = [] + + for i, video_url in enumerate(video_urls): + if 'hdcdn.online' in video_url: + line_name = 'HD线路' # HD线路优先 + elif 'hsex.tv' in video_url: + line_name = '主线路' + else: + line_name = f'线路{i+1}' + + play_from.append(line_name) + play_url.append(f'正片${video_url}') + + # 如果没有找到任何线路 + if not play_from: + play_from = ['好色TV'] + play_url = ['正片$暂无播放地址'] + + # 确保有两条线路(即使只有一条也复制一份) + if len(play_from) == 1: + play_from.append(f'{play_from[0]}-备用') + play_url.append(play_url[0]) + + vod = { + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': remarks, + 'vod_play_from': '$$$'.join(play_from), + 'vod_play_url': '$$$'.join(play_url) + } + return {'list': [vod]} + except Exception as e: + print(f"详情解析失败: {e}") + import traceback + traceback.print_exc() + return {'list': []} + + def searchContent(self, key, quick, pg=1): + try: + # 关键词合法性校验 + if not key.strip(): + print("搜索关键词不能为空") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 编码关键词 + encoded_key = urllib.parse.quote(key.strip(), encoding='utf-8', errors='replace') + + # 修复搜索翻页:根据页码构造正确的搜索URL + if int(pg) == 1: + # 第一页:/search.htm?search=关键词&sort=new + search_url = f"{self.host}search.htm" + else: + # 第二页及以后:/search-页码.htm?search=关键词&sort=new + search_url = f"{self.host}search-{pg}.htm" + + # 搜索参数 - 添加 sort=new 参数 + params = { + 'search': encoded_key, + 'sort': 'new' # 新增排序参数 + } + + # 发起请求 + resp = self.fetch( + url=search_url, + headers=self.headers, + params=params, + timeout=8 + ) + + if resp.status_code not in (200, 302): + print(f"搜索页面请求失败,URL:{resp.url},状态码:{resp.status_code}") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 处理页面内容 + html = resp.text.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 检测无结果场景 + no_result_texts = ['没有找到相关视频', '无搜索结果', 'No results found', '未找到匹配内容'] + no_result = any(data(f'div:contains("{text}"), p:contains("{text}")').text() for text in no_result_texts) + if no_result: + print(f"搜索关键词「{key}」第{pg}页无结果") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 解析搜索结果 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith(('http://', 'https://')): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知时长' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析单条搜索结果失败:{e}(跳过该条)") + continue + + # 解析总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + print(f"搜索关键词「{key}」分页解析完成,共{pagecount}页") + except Exception as e: + print(f"解析分页失败(默认单页):{e}") + pagecount = 1 + + # 返回结果 + total = len(vlist) * pagecount + print(f"搜索关键词「{key}」第{pg}页处理完成,结果{len(vlist)}条,总页数{pagecount}") + return { + 'list': vlist, + 'page': int(pg), + 'pagecount': pagecount, + 'limit': len(vlist), + 'total': total + } + except Exception as e: + print(f"搜索功能整体异常:{e}") + return { + 'list': [], + 'page': int(pg), + 'pagecount': 1, + 'limit': 0, + 'total': 0 + } + + def playerContent(self, flag, id, vipFlags): + headers = self.headers.copy() + headers.update({ + 'Referer': self.host, + 'Origin': self.host.rstrip('/'), + 'Host': urllib.parse.urlparse(self.host).netloc, + }) + + return { + 'parse': 1, + 'url': id, + 'header': headers, + 'double': True + } + + def localProxy(self, param): + try: + url = param['url'] + if url and not url.startswith(('http://', 'https://')): + url = f"{self.host.rstrip('/')}/{url.lstrip('/')}" + + img_headers = self.headers.copy() + img_headers.update({'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8'}) + + res = self.fetch(url, headers=img_headers, timeout=10) + content_type = res.headers.get('Content-Type', 'image/jpeg') + + return [200, content_type, res.content] + except Exception as e: + print(f"图片代理失败: {e}") + return [200, 'image/jpeg', b''] + + def fetch_with_retry(self, url, retry=2, timeout=5): + for i in range(retry + 1): + try: + resp = self.fetch(url, headers=self.headers, timeout=timeout) + if resp.status_code in (200, 301, 302): + return resp + print(f"请求{url}返回状态码{resp.status_code},重试中...") + except Exception as e: + print(f"第{i+1}次请求{url}失败: {e}") + if i < retry: + time.sleep(0.5) + return type('obj', (object,), {'text': '', 'status_code': 404}) + + def fetch(self, url, headers=None, timeout=5, method='GET', params=None): + headers = headers or self.headers + params = params or {} + try: + # 直接请求目标URL + if method.upper() == 'GET': + resp = requests.get( + url, + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params + ) + elif method.upper() == 'HEAD': + resp = requests.head( + url, + headers=headers, + timeout=timeout, + allow_redirects=False, + params=params + ) + else: + resp = requests.get( + url, + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params + ) + + # 自动适配编码,避免中文乱码 + if 'charset' in resp.headers.get('Content-Type', '').lower(): + resp.encoding = resp.apparent_encoding + else: + resp.encoding = 'utf-8' + return resp + except Exception as e: + print(f"网络请求失败({url}): {e}") + return type('obj', (object,), { + 'text': '', + 'status_code': 500, + 'headers': {}, + 'url': url + }) \ No newline at end of file diff --git a/lib/b好色™ Tv[密].py b/lib/b好色™ Tv[密].py new file mode 100644 index 0000000..a7543c5 --- /dev/null +++ b/lib/b好色™ Tv[密].py @@ -0,0 +1,543 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '好色™ Tv', + lang: 'hipy' +}) +""" + +import re +import sys +import urllib.parse +import threading +import time +import requests +from pyquery import PyQuery as pq + +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + def __init__(self): + # 基础配置 + self.name = '好色TV(优)' + self.host = 'https://hsex.icu/' + self.candidate_hosts = [ + "https://hsex.icu/", + "https://hsex1.icu/", + "https://hsex.tv/" + ] + self.headers = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36', + 'Referer': self.host + } + self.timeout = 5000 + + # 分类映射(关键修复:视频分类url_suffix设为空,适配list-{pg}.htm格式) + self.class_map = { + '视频': {'type_id': 'list', 'url_suffix': ''}, # 修复点1:视频分类后缀为空 + '周榜': {'type_id': 'top7', 'url_suffix': 'top7'}, + '月榜': {'type_id': 'top', 'url_suffix': 'top'}, + '5分钟+': {'type_id': '5min', 'url_suffix': '5min'}, + '10分钟+': {'type_id': 'long', 'url_suffix': 'long'} + } + + def getName(self): + return self.name + + def init(self, extend=""): + # 尝试获取最快可用域名 + self.host = self.get_fastest_host() + self.headers['Referer'] = self.host + + def isVideoFormat(self, url): + if not url: + return False + return any(fmt in url.lower() for fmt in ['.mp4', '.m3u8', '.flv', '.avi']) + + def manualVideoCheck(self): + def check(url): + if not self.isVideoFormat(url): + return False + try: + resp = self.fetch(url, headers=self.headers, method='HEAD', timeout=3) + return resp.status_code in (200, 302) and 'video' in resp.headers.get('Content-Type', '') + except: + return False + return check + + def get_fastest_host(self): + """测试候选域名,返回最快可用的""" + results = {} + threads = [] + + def test_host(url): + try: + start_time = time.time() + resp = requests.head(url, headers=self.headers, timeout=2, allow_redirects=False) + if resp.status_code in (200, 301, 302): + delay = (time.time() - start_time) * 1000 + results[url] = delay + else: + results[url] = float('inf') + except: + results[url] = float('inf') + + for host in self.candidate_hosts: + t = threading.Thread(target=test_host, args=(host,)) + threads.append(t) + t.start() + for t in threads: + t.join() + + valid_hosts = [(h, d) for h, d in results.items() if d != float('inf')] + return valid_hosts[0][0] if valid_hosts else self.candidate_hosts[0] + + def homeContent(self, filter): + result = {} + # 构造分类列表 + classes = [] + for name, info in self.class_map.items(): + classes.append({ + 'type_name': name, + 'type_id': info['type_id'] + }) + result['class'] = classes + + try: + # 获取首页内容 + html = self.fetch_with_retry(self.host, retry=2, timeout=5).text + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + # 提取图片URL + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + # 提取时长备注 + desc = item('.duration').text().strip() or '未知' + + # 提取视频ID + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析首页视频项失败: {e}") + continue + + result['list'] = vlist + except Exception as e: + print(f"首页解析失败: {e}") + result['list'] = [] + return result + + def homeVideoContent(self): + return [] + + def categoryContent(self, tid, pg, filter, extend): + result = {} + try: + # 匹配分类信息 + cate_info = None + for name, info in self.class_map.items(): + if info['type_id'] == tid: + cate_info = info + break + + if not cate_info: + result['list'] = [] + return result + + # 关键修复:区分视频分类与其他分类的URL格式 + if tid == 'list': # 视频分类(type_id为list) + url = f"{self.host}list-{pg}.htm" # 格式:list-1.htm、list-2.htm + else: # 其他分类(周榜/月榜等):xxx_list-{pg}.htm + url = f"{self.host}{cate_info['url_suffix']}_list-{pg}.htm" + + # 请求分类页 + html = self.fetch(url, headers=self.headers, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析分类视频项失败: {e}") + continue + + # 提取总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + except: + pagecount = 1 + + result['list'] = vlist + result['page'] = pg + result['pagecount'] = pagecount + result['limit'] = len(vlist) + result['total'] = 999999 + except Exception as e: + print(f"分类解析失败: {e}") + result['list'] = [] + result['page'] = pg + result['pagecount'] = 1 + result['limit'] = 0 + result['total'] = 0 + return result + + def detailContent(self, ids): + try: + if not ids or not ids[0]: + return {'list': []} + + vod_id = ids[0].strip() + if not vod_id.endswith('.htm'): + vod_id += '.htm' + url = f"{self.host}{vod_id.lstrip('/')}" + + html = self.fetch_with_retry(url, retry=2, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取标题 + title = data('.panel-title, .video-title, h1').text().strip() or '未知标题' + + # 提取封面图 + vod_pic = '' + poster_style = data('.vjs-poster').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', poster_style) + if pic_match: + vod_pic = pic_match.group(1) + if not vod_pic: + vod_pic = data('.video-pic img, .vjs-poster img, .thumbnail img').attr('src') or '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host}{vod_pic.lstrip('/')}" + + # 提取时长和观看量 + duration = '未知' + views = '未知' + info_items = data('.panel-body .col-md-3, .video-info .info-item, .info p') + for item in info_items.items(): + text = item.text().strip() + if '时长' in text or 'duration' in text.lower(): + duration = text.replace('时长:', '').replace('时长', '').strip() + elif '观看' in text or 'views' in text.lower(): + views_match = re.search(r'(\d+\.?\d*[kK]?)次观看', text) + if views_match: + views = views_match.group(1) + else: + views = text.replace('观看:', '').replace('观看', '').strip() + remarks = f"{duration} | {views}" + + # 提取播放地址 + video_url = '' + m3u8_match = re.search(r'videoUrl\s*=\s*["\']([^"\']+\.m3u8)["\']', html) + if m3u8_match: + video_url = m3u8_match.group(1) + if not video_url: + source = data('source[src*=".m3u8"], source[src*=".mp4"]') + video_url = source.attr('src') or '' + if not video_url: + js_matches = re.findall(r'(https?://[^\s"\']+\.(?:m3u8|mp4))', html) + if js_matches: + video_url = js_matches[0] + + if video_url and not video_url.startswith('http'): + video_url = f"{self.host}{video_url.lstrip('/')}" + + vod = { + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': remarks, + 'vod_play_from': '好色TV(优)', + 'vod_play_url': f'正片${video_url}' if video_url else '正片$暂无地址' + } + return {'list': [vod]} + except Exception as e: + print(f"详情解析失败: {e}") + return {'list': []} + + def searchContent(self, key, quick, pg=1): + try: + # 关键词合法性校验 + if not key.strip(): + print("搜索关键词不能为空") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 编码关键词 + encoded_key = urllib.parse.quote(key.strip(), encoding='utf-8', errors='replace') + + # 构造搜索URL + search_url = f"{self.host}search.htm" + params = { + 'search': encoded_key, + 'page': int(pg) + } + + # 发起请求 + resp = self.fetch( + url=search_url, + headers=self.headers, + params=params, + timeout=8 + ) + if resp.status_code not in (200, 302): + print(f"搜索页面请求失败,URL:{resp.url},状态码:{resp.status_code}") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 处理页面内容 + html = resp.text.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 检测无结果场景 + no_result_texts = ['没有找到相关视频', '无搜索结果', 'No results found', '未找到匹配内容'] + no_result = any(data(f'div:contains("{text}"), p:contains("{text}")').text() for text in no_result_texts) + if no_result: + print(f"搜索关键词「{key}」第{pg}页无结果") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 解析搜索结果 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith(('http://', 'https://')): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知时长' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析单条搜索结果失败:{e}(跳过该条)") + continue + + # 解析总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + print(f"搜索关键词「{key}」分页解析完成,共{pagecount}页") + except Exception as e: + print(f"解析分页失败(默认单页):{e}") + pagecount = 1 + + # 返回结果(修复点2:补全page键的引号,修正语法错误) + total = len(vlist) * pagecount + print(f"搜索关键词「{key}」第{pg}页处理完成,结果{len(vlist)}条,总页数{pagecount}") + return { + 'list': vlist, + 'page': int(pg), # 原代码此处缺少引号,导致语法错误 + 'pagecount': pagecount, + 'limit': len(vlist), + 'total': total + } + except Exception as e: + print(f"搜索功能整体异常:{e}") + return { + 'list': [], + 'page': int(pg), 'pagecount': 1, + 'limit': 0, + 'total': 0 + } + + def playerContent(self, flag, id, vipFlags): + headers = self.headers.copy() + headers.update({ + 'Referer': self.host, + 'Origin': self.host.rstrip('/'), + 'Host': urllib.parse.urlparse(self.host).netloc, + }) + + # 根据rule中的double设置 + return { + 'parse': 1, # 根据rule中的play_parse设置 + 'url': id, + 'header': headers, + 'double': True # 根据rule中的double设置 + } + + def localProxy(self, param): + try: + url = param['url'] + if url and not url.startswith(('http://', 'https://')): + url = f"{self.host.rstrip('/')}/{url.lstrip('/')}" + + img_headers = self.headers.copy() + img_headers.update({'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8'}) + + res = self.fetch(url, headers=img_headers, timeout=10) + content_type = res.headers.get('Content-Type', 'image/jpeg') + + return [200, content_type, res.content] + except Exception as e: + print(f"图片代理失败: {e}") + return [200, 'image/jpeg', b''] + + def fetch_with_retry(self, url, retry=2, timeout=5): + for i in range(retry + 1): + try: + resp = self.fetch(f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}',headers=self.headers, timeout=timeout) + if resp.status_code in (200, 301, 302): + return resp + print(f"请求{url}返回状态码{resp.status_code},重试中...") + except Exception as e: + print(f"第{i+1}次请求{url}失败: {e}") + if i < retry: + time.sleep(0.5) + return type('obj', (object,), {'text': '', 'status_code': 404}) + + def fetch(self, url, headers=None, timeout=5, method='GET', params=None): + headers = headers or self.headers + params = params or {} + try: + if method.upper() == 'GET': + resp = requests.get( + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params # 支持GET请求带参数,适配搜索分页 + ) + elif method.upper() == 'HEAD': + resp = requests.head( + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=False, + params=params + ) + else: + resp = requests.get( # 默认GET请求,兼容其他方法调用 + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params + ) + # 自动适配编码,避免中文乱码 + if 'charset' in resp.headers.get('Content-Type', '').lower(): + resp.encoding = resp.apparent_encoding + else: + resp.encoding = 'utf-8' + return resp + except Exception as e: + print(f"网络请求失败({url}): {e}") + # 返回统一格式空响应,避免后续逻辑崩溃 + return type('obj', (object,), { + 'text': '', + 'status_code': 500, + 'headers': {}, + 'url': url + }) + + +# ------------------------------ +# 可选测试代码(运行时注释或删除,用于验证功能) +# ------------------------------ +if __name__ == "__main__": + # 初始化爬虫 + spider = Spider() + spider.init() + + # 测试首页内容 + print("=== 测试首页 ===") + home_data = spider.homeContent(filter='') + print(f"首页分类数:{len(home_data['class'])}") + print(f"首页视频数:{len(home_data['list'])}") + + # 测试视频分类(修复后的数据获取) + print("\n=== 测试视频分类(第1页) ===") + cate_data = spider.categoryContent(tid='list', pg=1, filter='', extend='') + print(f"视频分类第1页视频数:{len(cate_data['list'])}") + print(f"视频分类总页数:{cate_data['pagecount']}") + + # 测试搜索功能(修复语法错误后) + print("\n=== 测试搜索(关键词:测试) ===") + search_data = spider.searchContent(key="测试", quick=False, pg=1) + print(f"搜索结果数:{len(search_data['list'])}") + print(f"搜索总页数:{search_data['pagecount']}") diff --git a/lib/c色播聚合[密].py b/lib/c色播聚合[密].py new file mode 100644 index 0000000..d5adee7 --- /dev/null +++ b/lib/c色播聚合[密].py @@ -0,0 +1,82 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '色播聚合', + lang: 'hipy' +}) +""" + +#coding=utf-8 +#!/usr/bin/python +import sys +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + def init(self,extend=""): + self.base_url='http://api.hclyz.com:81/mf' + + def homeContent(self,filter): + classes = [{"type_name": "色播聚合","type_id":"/json.txt"}] + result = {"class": classes} + return result + + def categoryContent(self,tid,pg,filter,extend): + home = self.fetch(f'{self.base_url}/json.txt').json() + data = home.get("pingtai")[1:] + videos = [ + { + "vod_id": "/" + item['address'], + "vod_name": item['title'], + "vod_pic": item['xinimg'].replace("http://cdn.gcufbd.top/img/", + "https://slink.ltd/https://raw.githubusercontent.com/fish2018/lib/refs/heads/main/imgs/"), + "vod_remarks": item['Number'], + "style": {"type": "rect", "ratio": 1.33} + } for item in sorted(data, key=lambda x: int(x['Number']), reverse=True) + ] + result = { + "page": pg, + "pagecount": 1, + "limit": len(videos), + "total": len(videos), + "list": videos + } + return result + + def detailContent(self,array): + id = array[0] + data = self.fetch(f'{self.base_url}/{id}').json() + zhubo = data['zhubo'] + playUrls = '#'.join([f"{vod['title']}${vod['address']}" for vod in zhubo]) + vod = [{ + "vod_play_from": 'sebo', + "vod_play_url": playUrls, + "vod_content": 'https://github.com/fish2018', + }] + result = {"list": vod} + return result + + def playerContent(self,flag,id,vipFlags): + result = { + 'parse': 0, + 'url': id + } + return result + + def getName(self): + return '色播聚合' + + def homeVideoContent(self): + pass + def isVideoFormat(self,url): + pass + def manualVideoCheck(self): + pass + def searchContent(self,key,quick): + pass + def destroy(self): + pass + def localProxy(self, param): + pass \ No newline at end of file diff --git a/lib/d随机小姐姐[密].py b/lib/d随机小姐姐[密].py new file mode 100644 index 0000000..5c56d29 --- /dev/null +++ b/lib/d随机小姐姐[密].py @@ -0,0 +1,154 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '首页', + lang: 'hipy' +}) +""" + +# coding=utf-8 +# !/usr/bin/python +import sys +import requests +import datetime +from bs4 import BeautifulSoup +import re +import base64 +from base.spider import Spider +import json + +sys.path.append('..') +xurl = "http://xjj2.716888.xyz" +headerx = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.87 Safari/537.36', + 'Cookie':'mk_encrypt_c21f969b5f03d33d43e04f8f136e7682=390e11f0d5ae13b2787e6a72db11527f' +} + + +class Spider(Spider): + global xurl + global headerx + + def getName(self): + return "首页" + + def init(self, extend): + pass + + def isVideoFormat(self, url): + pass + + def manualVideoCheck(self): + pass + + def homeContent(self, filter): + pass + + def homeVideoContent(self): + id = ['4k/4k.php', 'djxjj/dj1.php', 'zj/jipinyz/jipinyz.php', 'zj/xuejie/xuejie.php', 'zj/kawayi/kawayi.php', + 'zj/nennen/nennen.php', 'zj/heji1/heji1.php', 'zj/sihuawd/sihuawd.php', 'zj/wanmeisc/wanmeisc.php', + 'zj/manyao/manyao.php', 'zj/sihuadd/sihuadd.php', 'zj/qingchun/qingchun.php', 'zj/cos/cos.php', + 'zj/jingpinbz/jingpinbz.php', 'zj/jipinll/jipinll.php', 'zj/nideym/nideym.php', 'zj/tianmei/tianmei.php', + 'zj/yusi/yusi.php', 'zj/shuaige/shuaige.php', 'zj/rewu/rewu.php', 'zj/jingpinsc/jingpinsc.php'] + name = ['随机', 'DJ姐姐', '极品钰足', '学姐系列', '卡哇伊', '嫩嫩系列', '美女舞蹈', '丝滑舞蹈', '完美身材', + '慢摇系列', '丝滑吊带', '清纯系列', 'COS系列', '精品变装', '极品罗丽', '你的裕梦', '甜妹系列', + '御丝系列', '帅哥哥', '热舞系列', '精品收藏'] + pic = ['https://img0.baidu.com/it/u=2236794495,926227820&fm=253&fmt=auto&app=138&f=JPEG?w=1091&h=500', + 'https://pic.rmb.bdstatic.com/mvideo/e17d86ce4489a02870ace9a25a804c3e', + 'https://img1.baidu.com/it/u=4087009209,613234683&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=364', + 'https://img1.baidu.com/it/u=2347706654,3055017263&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=750', + 'https://img2.baidu.com/it/u=3715511725,1094436549&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=1083', + 'https://img2.baidu.com/it/u=2560410906,3760952489&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=750', + 'https://img0.baidu.com/it/u=4119328645,2294770712&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=750', + 'https://img1.baidu.com/it/u=3167365498,4156845177&fm=253&fmt=auto&app=120&f=JPEG?w=355&h=631', + 'https://img2.baidu.com/it/u=2214691242,2295609938&fm=253&fmt=auto&app=120&f=JPEG?w=800&h=973', + 'https://img1.baidu.com/it/u=3930123826,1131807820&fm=253&fmt=auto&app=138&f=JPEG?w=889&h=500', + 'https://img2.baidu.com/it/u=3998619741,1128428746&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=594', + 'https://img2.baidu.com/it/u=1507871502,2316279678&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=768', + 'https://img0.baidu.com/it/u=2245878765,4037513957&fm=253&fmt=auto&app=138&f=JPEG?w=617&h=411', + 'https://img1.baidu.com/it/u=3623293272,829752126&fm=253&fmt=auto&app=138&f=JPEG?w=285&h=285', + 'https://img2.baidu.com/it/u=1922261112,3647796435&fm=253&fmt=auto&app=120&f=JPEG?w=500&h=542', + 'https://img1.baidu.com/it/u=3970043028,2042301564&fm=253&fmt=auto&app=120&f=JPEG?w=500&h=889', + 'https://img2.baidu.com/it/u=3229384329,3046902124&fm=253&fmt=auto&app=120&f=JPEG?w=800&h=800', + 'https://img1.baidu.com/it/u=3113661564,2558849413&fm=253&fmt=auto&app=138&f=JPEG?w=500&h=500', + 'https://img1.baidu.com/it/u=2361496550,3302335162&fm=253&fmt=auto&app=138&f=JPEG?w=333&h=500', + 'https://img1.baidu.com/it/u=270105183,1595166255&fm=253&fmt=auto&app=120&f=JPEG?w=800&h=500', + 'https://img1.baidu.com/it/u=4071105902,825241031&fm=253&fmt=auto&app=138&f=JPEG?w=235&h=340'] + list_length = len(id) + videos = [] + for i in range(list_length): + print(id[i]) + video = { + "vod_id": id[i], + "vod_name": name[i], + "vod_pic": pic[i], + "vod_remarks": '播放20个', + } + videos.append(video) + + result = {'list': videos} + + return result + + def categoryContent(self, cid, pg, filter, ext): + pass + + def detailContent(self, ids): + videos = [] + result = {} + did = ids[0] + for i in range(1, 21): + playurl = "" + for j in range(1, i + 1): + playurl += f"{j}$/fenlei/{did}#" + playurl = playurl[:-1] + + videos.append({ + "vod_id": '', + "vod_name": '', + "vod_pic": "", + "type_name": '', + "vod_year": "", + "vod_area": "", + "vod_remarks": "", + "vod_actor": "", + "vod_director": "", + "vod_content": "", + "vod_play_from": "GK推荐", + "vod_play_url": playurl + }) + + result['list'] = videos + return result + + def playerContent(self, flag, id, vipFlags): + result = {} + response = requests.get(url=xurl + id, headers=headerx, allow_redirects=False) + + location_header = response.headers.get('Location') + if 'http' in location_header: + purl = location_header + else: + purl = 'http:' + location_header + result["parse"] = 0 + result["playUrl"] = '' + result["url"] = purl + result["header"] = headerx + return result + + def searchContentPage(self, key, quick, page): + pass + + def searchContent(self, key, quick): + return self.searchContentPage(key, quick, '1') + + def localProxy(self, params): + if params['type'] == "m3u8": + return self.proxyM3u8(params) + elif params['type'] == "media": + return self.proxyMedia(params) + elif params['type'] == "ts": + return self.proxyTs(params) + return None diff --git a/lib/e小红薯App[密].py b/lib/e小红薯App[密].py new file mode 100644 index 0000000..1cccf93 --- /dev/null +++ b/lib/e小红薯App[密].py @@ -0,0 +1,196 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '小红薯APP', + lang: 'hipy' +}) +""" + +# -*- coding: utf-8 -*- +# by @嗷呜 +import json +import random +import string +import sys +import time +from base64 import b64decode +from Crypto.Cipher import AES +from Crypto.Hash import MD5 +from Crypto.Util.Padding import unpad +sys.path.append('..') +from base.spider import Spider + + +class Spider(Spider): + + def init(self, extend=""): + self.did = self.getdid() + self.token,self.phost,self.host = self.gettoken() + pass + + def isVideoFormat(self, url): + pass + + def manualVideoCheck(self): + pass + + def destroy(self): + pass + + hs = ['fhoumpjjih', 'dyfcbkggxn', 'rggwiyhqtg', 'bpbbmplfxc'] + + def homeContent(self, filter): + data = self.fetch(f'{self.host}/api/video/queryClassifyList?mark=4', headers=self.headers()).json()['encData'] + data1 = self.aes(data) + result = {} + classes = [] + for k in data1['data']: + classes.append({'type_name': k['classifyTitle'], 'type_id': k['classifyId']}) + result['class'] = classes + return result + + def homeVideoContent(self): + pass + + def categoryContent(self, tid, pg, filter, extend): + path=f'/api/short/video/getShortVideos?classifyId={tid}&videoMark=4&page={pg}&pageSize=20' + result = {} + videos = [] + data=self.fetch(f'{self.host}{path}', headers=self.headers()).json()['encData'] + vdata=self.aes(data) + for k in vdata['data']: + videos.append({"vod_id": k['videoId'], 'vod_name': k.get('title'), 'vod_pic': self.getProxyUrl() + '&url=' + k['coverImg'], + 'vod_remarks': self.dtim(k.get('playTime')),'style': {"type": "rect", "ratio": 1.33}}) + result["list"] = videos + result["page"] = pg + result["pagecount"] = 9999 + result["limit"] = 90 + result["total"] = 999999 + return result + + def detailContent(self, ids): + path = f'/api/video/getVideoById?videoId={ids[0]}' + data = self.fetch(f'{self.host}{path}', headers=self.headers()).json()['encData'] + v = self.aes(data) + d=f'{v["title"]}$auth_key={v["authKey"]}&path={v["videoUrl"]}' + vod = {'vod_name': v["title"], 'type_name': ''.join(v.get('tagTitles',[])),'vod_play_from': v.get('nickName') or "小红书官方", 'vod_play_url': d} + result = {"list": [vod]} + return result + + def searchContent(self, key, quick, pg='1'): + pass + + def playerContent(self, flag, id, vipFlags): + h=self.headers() + h['Authorization'] = h.pop('aut') + del h['deviceid'] + result = {"parse": 0, "url": f"{self.host}/api/m3u8/decode/authPath?{id}", "header": h} + return result + + def localProxy(self, param): + return self.action(param) + + def md5(self, text): + h = MD5.new() + h.update(text.encode('utf-8')) + return h.hexdigest() + + def aes(self, word): + key = b64decode("SmhiR2NpT2lKSVV6STFOaQ==") + iv = key + cipher = AES.new(key, AES.MODE_CBC, iv) + decrypted = unpad(cipher.decrypt(b64decode(word)), AES.block_size) + return json.loads(decrypted.decode('utf-8')) + + def dtim(self, seconds): + try: + seconds = int(seconds) + hours = seconds // 3600 + remaining_seconds = seconds % 3600 + minutes = remaining_seconds // 60 + remaining_seconds = remaining_seconds % 60 + + formatted_minutes = str(minutes).zfill(2) + formatted_seconds = str(remaining_seconds).zfill(2) + + if hours > 0: + formatted_hours = str(hours).zfill(2) + return f"{formatted_hours}:{formatted_minutes}:{formatted_seconds}" + else: + return f"{formatted_minutes}:{formatted_seconds}" + except: + return '' + + def getdid(self): + did = self.getCache('did') + if not did: + t = str(int(time.time())) + did = self.md5(t) + self.setCache('did', did) + return did + + def getsign(self): + t=str(int(time.time() * 1000)) + return self.md5(t[3:8]),t + + def gettoken(self, i=0, max_attempts=10): + if i >= len(self.hs) or i >= max_attempts: + return '' + current_domain = f"https://{''.join(random.choices(string.ascii_lowercase + string.digits, k=random.randint(5, 10)))}.{self.hs[i]}.work" + try: + sign,t=self.getsign() + url = f'{current_domain}/api/user/traveler' + headers = { + 'User-Agent': 'Mozilla/5.0 (Linux; Android 11; M2012K10C Build/RP1A.200720.011; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/87.0.4280.141 Mobile Safari/537.36;SuiRui/xhs/ver=1.2.6', + 'deviceid': self.did, 't': t, 's': sign, } + data = {'deviceId': self.did, 'tt': 'U', 'code': '', 'chCode': 'dafe13'} + data1 = self.post(url, json=data, headers=headers) + data1.raise_for_status() + data2 = data1.json()['data'] + return data2['token'], data2['imgDomain'],current_domain + except: + return self.gettoken(i+1, max_attempts) + + def headers(self): + sign,t=self.getsign() + henda = { + 'User-Agent': 'Mozilla/5.0 (Linux; Android 11; M2012K10C Build/RP1A.200720.011; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/87.0.4280.141 Mobile Safari/537.36;SuiRui/xhs/ver=1.2.6', + 'deviceid': self.did, 't': t, 's': sign, 'aut': self.token} + return henda + + def action(self, param): + headers = { + 'User-Agent': 'Dalvik/2.1.0 (Linux; U; Android 11; M2012K10C Build/RP1A.200720.011)'} + data = self.fetch(f'{self.phost}{param["url"]}', headers=headers) + type=data.headers.get('Content-Type').split(';')[0] + base64_data = self.img(data.content, 100, '2020-zq3-888') + return [200, type, base64_data] + + def img(self, data: bytes, length: int, key: str): + GIF = b'\x47\x49\x46' + JPG = b'\xFF\xD8\xFF' + PNG = b'\x89\x50\x4E\x47\x0D\x0A\x1A\x0A' + + def is_dont_need_decode_for_gif(data): + return len(data) > 2 and data[:3] == GIF + + def is_dont_need_decode_for_jpg(data): + return len(data) > 7 and data[:3] == JPG + + def is_dont_need_decode_for_png(data): + return len(data) > 7 and data[1:8] == PNG[1:8] + + if is_dont_need_decode_for_png(data): + return data + elif is_dont_need_decode_for_gif(data): + return data + elif is_dont_need_decode_for_jpg(data): + return data + else: + key_bytes = key.encode('utf-8') + result = bytearray(data) + for i in range(length): + result[i] ^= key_bytes[i % len(key_bytes)] + return bytes(result) diff --git a/lib/f妲己[密].py b/lib/f妲己[密].py new file mode 100644 index 0000000..5c8919c --- /dev/null +++ b/lib/f妲己[密].py @@ -0,0 +1,260 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '首页', + lang: 'hipy' +}) +""" + +import requests +from bs4 import BeautifulSoup +import re +from base.spider import Spider +import sys +import json +import base64 +import urllib.parse + +sys.path.append('..') + +murl = "https://3642.7rnr.com/web/index.html" +headerx = { + 'User-Agent': "Mozilla/5.0 (Linux; Android 13; M2102J2SC Build/TKQ1.221114.001; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/143.0.7499.3 Mobile Safari/537.36", + 'Accept-Encoding': "gzip, deflate, br, zstd" +} +response = requests.get(murl, allow_redirects=True) +xurl = response.url +nurl = xurl + '/web/abcdefg.ashx' +pm = '' + +class Spider(Spider): + global xurl + global headerx + + def getName(self): + return "首页" + + def init(self, extend): + pass + + def isVideoFormat(self, url): + pass + + def manualVideoCheck(self): + pass + + def homeContent(self, filter): + result = {} + result = {"class": [{"type_id": "1003", "type_name": "亚洲无码"}, + {"type_id": "3022", "type_name": "欧美无码"}, + {"type_id": "3026", "type_name": "中文字幕"}, + {"type_id": "3025", "type_name": "经典三级"}, + {"type_id": "5", "type_name": "国产主播"}, + {"type_id": "134", "type_name": "韩国主播"}, + {"type_id": "3137", "type_name": "ASMR"}, + {"type_id": "3138", "type_name": "恐怖色情"}, + {"type_id": "131", "type_name": "网红视频"}, + {"type_id": "132", "type_name": "国产视频"}, + {"type_id": "3023", "type_name": "人妖伪娘"}, + {"type_id": "130", "type_name": "动漫卡通"}, + {"type_id": "3088", "type_name": "华人原创"}, + {"type_id": "3135", "type_name": "JVID"}, + {"type_id": "3136", "type_name": "SWAG"}, + {"type_id": "3134", "type_name": "明星换脸"}]} + return result + + def homeVideoContent(self): + videos = [] + try: + payload = { + 'action': "getindexdata", + 't': "1762753963537691", + 's': "5ad87a586f5aae9c2ca4f913d45f8958"} + detail = requests.post(url=nurl, data=payload, headers=headerx) + detail.encoding = "utf-8" + res = detail.json() + video_list = res.get("videos", []) + + for video in video_list: + name = video.get("title", "") + id = video.get("id", "") + pic = video.get("coverimg", "") + remarks = video.get("updatedate", "") + video = { + "vod_id": id, + "vod_name": name, + "vod_pic": pic, + "vod_remarks": remarks + } + videos.append(video) + + result = {'list': videos} + return result + except: + pass + + def categoryContent(self, cid, pg, filter, ext): + result = {} + videos = [] + if pg: + page = int(pg) + else: + page = 1 + + if page == '1': + payload1 = { + 'action': "getvideos", + 'vtype': {cid}, + 'pageindex': "1", + 'pagesize': "12", + 'tags': "全部", + 'sortindex': "1", + 't': "176275570014518", + 's': "ff4218e4cafd552c4d0c93eb935c14f1"} + else: + payload1 = { + 'action': "getvideos", + 'vtype': {cid}, + 'pageindex': {str(page)}, + 'pagesize': "12", + 'tags': "全部", + 'sortindex': "1", + 't': "176275570014518", + 's': "ff4218e4cafd552c4d0c93eb935c14f1"} + try: + detail = requests.post(url=nurl, data=payload1, headers=headerx) + detail.encoding = "utf-8" + res = detail.json() + video_list = res.get("videos", []) + + for video in video_list: + name = video.get("title", "") + id = video.get("id", "") + pic = video.get("coverimg", "") + remarks = video.get("updatedate", "") + video = { + "vod_id": id, + "vod_name": name, + "vod_pic": pic, + "vod_remarks": remarks + } + videos.append(video) + + except: + pass + result = {'list': videos} + result['page'] = pg + result['pagecount'] = 99 + result['limit'] = 90 + result['total'] = 99 + return result + + def detailContent(self, ids): + did = ids[0] + result = {} + videos = [] + playurl = '' + payload2 = { + 'action': "getvideo", + 'vid': did, + 't': "1762756475175265", + 's': "656d5b40c2122f86bc35895dc58fd113"} + res1 = requests.post(url=nurl, data=payload2, headers=headerx) + res1.encoding = "utf-8" + res = res1.json() + node = res.get("data", {}).get("Table", [{}])[0] + + vod_id = node.get("id", "") + vod_name = node.get("title", "") + vod_pic = node.get("coverimg", "") + vod_remarks = node.get("updatedate", "") + vod_content = node.get("title", "") + + playFrom = [] + playList = [] + if node.get("vurl"): + base_url1 = res.get("xldata", {}).get("value", "") + base_url2 = res.get("xldata", {}).get("value1", "") + if base_url1: + full_vurl1 = base_url1 + node.get("vurl", "") + playFrom.append("播放源1") + playList.append(full_vurl1) + if base_url2: + full_vurl2 = base_url2 + node.get("vurl", "") + playFrom.append("播放源2") + playList.append(full_vurl2) + + videos.append({ + "vod_id": vod_id, + "vod_name": vod_name, + "vod_pic": vod_pic, + "vod_remarks": vod_remarks, + "vod_content": vod_content, + "vod_play_from": "$$$".join(playFrom), + "vod_play_url": "$$$".join(playList) + }) + + result['list'] = videos + return result + + def playerContent(self, flag, id, vipFlags): + parts = id.split("http") + xiutan = 1 + if xiutan == 1: + if len(parts) > 1: + before_https, after_https = parts[0], 'http' + parts[1] + result = {} + result["parse"] = xiutan + result["playUrl"] = '' + result["url"] = after_https + result["header"] = headerx + return result + + def searchContentPage(self, key, quick, page): + result = {} + videos = [] + payload3 = { + 'action': "search", + 'p': {key}, + 'pageindex': {str(page)}, + 'pagesize': "12", + 'channelid': "0", + 't': "1762756927087982", + 's': "2392bd117b4e6e35b5ec1fa9bc380b6f"} + detail = requests.post(url=nurl, data=payload3, headers=headerx) + detail.encoding = "utf-8" + res = detail.json() + video_list = res.get("data", []) + for video in video_list: + name = video.get("title", "") + id = video.get("id", "") + pic = video.get("imgurl", "") + remarks = video.get("updatedate", "") + video = { + "vod_id": id, + "vod_name": name, + "vod_pic": pic, + "vod_remarks": remarks + } + videos.append(video) + + result['list'] = videos + result['page'] = page + result['pagecount'] = 60 + result['limit'] = 30 + result['total'] = 999999 + return result + + def searchContent(self, key, quick): + return self.searchContentPage(key, quick, '1') + + def localProxy(self, params): + if params['type'] == "m3u8": + return self.proxyM3u8(params) + elif params['type'] == "media": + return self.proxyMedia(params) + elif params['type'] == "ts": + return self.proxyTs(params) + return None \ No newline at end of file diff --git a/lib/g瑟瑟直播[密].py b/lib/g瑟瑟直播[密].py new file mode 100644 index 0000000..306dec5 --- /dev/null +++ b/lib/g瑟瑟直播[密].py @@ -0,0 +1,267 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: 'Leospring直播', + lang: 'hipy' +}) +""" + +# -*- coding: utf-8 -*- +import json +import sys +import traceback +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + primary_host = 'http://api.hclyz.com:81/mf/' + backup_host = 'http://api.maiyoux.com:81/mf/' + host = primary_host + platforms = [] + + def init(self, extend=""): + if extend: + h = extend.strip() + if not h.endswith('/'): + h += '/' + self.host = h + else: + self.host = self.primary_host + + content = None + for base in [self.host, self.backup_host]: + try: + txt = self.fetch(base + 'json.txt').text + data = None + try: + data = json.loads(txt) + except Exception: + data = None + + if data is not None: + self.host = base + self.platforms = self._parse_catalog_json(data) + if self.platforms: + return + else: + plats = self._parse_catalog_text(txt) + if plats: + self.host = base + self.platforms = plats + return + except Exception: + continue + + def _parse_catalog_json(self, data): + items = [] + if isinstance(data, dict): + if 'pingtai' in data and isinstance(data['pingtai'], list): + items = data['pingtai'] + else: + if all(isinstance(v, dict) for v in data.values()): + items = list(data.values()) + elif isinstance(data, list): + items = data + + platforms = [] + for it in items: + name = it.get('mc') or it.get('title') or it.get('name') or '' + img = it.get('tp1') or it.get('xinimg') or it.get('img') or '' + file = it.get('dz') or it.get('address') or it.get('file') or '' + count = it.get('sl') or it.get('Number') or it.get('count') or 0 + try: + count = int(count) + except Exception: + pass + if name and file: + if not file.endswith('.txt'): + file += '.txt' + if not file.startswith('json'): + file = 'json' + file + platforms.append({ + 'name': name, + 'img': img, + 'file': file, + 'count': count + }) + return platforms + + def _parse_catalog_text(self, txt): + plats = [] + block = txt.strip() + if not block: + return plats + if block.startswith('{') and block.endswith('}'): + block = block[1:-1] + chunks = [c for c in block.split('|') if c.strip()] + curr = {'mc': '', 'tp1': '', 'dz': '', 'sl': 0} + def flush(): + if curr.get('mc') and curr.get('dz'): + try: + sl = int(curr.get('sl') or 0) + except Exception: + sl = curr.get('sl') or 0 + fname = curr.get('dz') + if not fname.endswith('.txt'): + fname += '.txt' + if not fname.startswith('json'): + fname = 'json' + fname + plats.append({ + 'name': curr.get('mc'), + 'img': curr.get('tp1'), + 'file': fname, + 'count': sl + }) + for c in chunks: + s = c.strip() + if s.startswith('@mc'): + if curr.get('mc') or curr.get('dz'): + flush() + curr = {'mc': '', 'tp1': '', 'dz': '', 'sl': 0} + curr['mc'] = s.replace('@mc', '', 1) + elif s.startswith('@tp1'): + curr['tp1'] = s.replace('@tp1', '', 1) + elif s.startswith('@dz'): + curr['dz'] = s.replace('@dz', '', 1) + elif s.startswith('@sl'): + curr['sl'] = s.replace('@sl', '', 1) + if curr.get('mc') or curr.get('dz'): + flush() + return plats + + def _load_platform_rooms(self, file_path, pg=1): + url_candidates = [ + self.host + file_path, + ] + alt_path = file_path.replace('json', '', 1) + if alt_path != file_path: + url_candidates.insert(0, self.host + alt_path) + + data = None + raw = None + for u in url_candidates: + try: + raw = self.fetch(u).text + data = json.loads(raw) + break + except Exception: + data = None + continue + + rooms = [] + if isinstance(data, dict): + if 'zhubo' in data and isinstance(data['zhubo'], list): + rooms = data['zhubo'] + elif 'list' in data and isinstance(data['list'], list): + rooms = data['list'] + elif 'data' in data and isinstance(data['data'], list): + rooms = data['data'] + elif 'rooms' in data and isinstance(data['rooms'], list): + rooms = data['rooms'] + elif isinstance(data, list): + rooms = data + + videos = [] + for idx, r in enumerate(rooms or [], 1): + title = r.get('title') or r.get('name') or r.get('nickname') or f'主播{idx}' + address = r.get('address') or r.get('url') or r.get('stream') or '' + img = r.get('img') or r.get('pic') or r.get('avatar') or '' + remarks = r.get('Number') or r.get('online') or r.get('viewers') or '' + if title and address: + videos.append({ + 'vod_id': address, + 'vod_name': title, + 'vod_pic': img, + 'vod_remarks': str(remarks) + }) + total = len(videos) + if pg > 1: + start = (pg - 1) * 20 + videos = videos[start:start + 20] + return videos, total + + def isVideoFormat(self, url): + return False + + def manualVideoCheck(self): + pass + + def getName(self): + return 'Leospring直播' + + def homeContent(self, filter): + classes = [] + for p in self.platforms: + classes.append({ + 'type_id': p['file'], + 'type_name': p['name'], + }) + return {'class': classes} + + def homeVideoContent(self): + return {'list': []} + + def _find_platform(self, tid): + for p in self.platforms: + if p['file'] == tid or p['name'] == tid: + return p + return None + + def categoryContent(self, tid, pg, filter, extend): + p = self._find_platform(tid) + if not p: + return {'list': [], 'page': pg, 'pagecount': 0, 'limit': 0, 'total': 0} + + videos, total = self._load_platform_rooms(p['file'], int(pg)) + pagecount = (total + 19) // 20 + result = { + 'list': videos, + 'page': pg, + 'pagecount': pagecount, + 'limit': min(len(videos), 20), + 'total': total + } + return result + + def searchContent(self, key, quick, pg="1"): + key = (key or '').strip().lower() + out = [] + if not key: + return {'list': out} + for p in self.platforms: + videos, _ = self._load_platform_rooms(p['file']) + for v in videos: + if key in v['vod_name'].lower(): + out.append(v) + if len(out) >= 50: + break + return {'list': out[:50]} + + def detailContent(self, ids): + try: + address = ids[0] + if not address: + return {'list': []} + + vod = { + 'vod_id': address, + 'vod_name': '直播源', + 'vod_play_from': '瑟瑟站大佬张佬', + 'vod_play_url': address, + 'vod_content': '多看少打卡', + } + return {'list': [vod]} + except Exception as e: + traceback.print_exc() + return {'list': []} + + def playerContent(self, flag, id, vipFlags): + return { + 'parse': 0, + 'url': id + } + + def localProxy(self, param): + pass diff --git a/lib/h免翻好色TV[密].py b/lib/h免翻好色TV[密].py new file mode 100644 index 0000000..b09e50d --- /dev/null +++ b/lib/h免翻好色TV[密].py @@ -0,0 +1,543 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '月佬免翻版好色TV[密]', + lang: 'hipy' +}) +""" + +import re +import sys +import urllib.parse +import threading +import time +import requests +from pyquery import PyQuery as pq + +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + def __init__(self): + # 基础配置 + self.name = '好色TV(优)' + self.host = 'https://hsex.icu/' + self.candidate_hosts = [ + "https://hsex.icu/", + "https://hsex1.icu/", + "https://hsex.tv/" + ] + self.headers = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36', + 'Referer': self.host + } + self.timeout = 5000 + + # 分类映射(关键修复:视频分类url_suffix设为空,适配list-{pg}.htm格式) + self.class_map = { + '视频': {'type_id': 'list', 'url_suffix': ''}, # 修复点1:视频分类后缀为空 + '周榜': {'type_id': 'top7', 'url_suffix': 'top7'}, + '月榜': {'type_id': 'top', 'url_suffix': 'top'}, + '5分钟+': {'type_id': '5min', 'url_suffix': '5min'}, + '10分钟+': {'type_id': 'long', 'url_suffix': 'long'} + } + + def getName(self): + return self.name + + def init(self, extend=""): + # 尝试获取最快可用域名 + self.host = self.get_fastest_host() + self.headers['Referer'] = self.host + + def isVideoFormat(self, url): + if not url: + return False + return any(fmt in url.lower() for fmt in ['.mp4', '.m3u8', '.flv', '.avi']) + + def manualVideoCheck(self): + def check(url): + if not self.isVideoFormat(url): + return False + try: + resp = self.fetch(url, headers=self.headers, method='HEAD', timeout=3) + return resp.status_code in (200, 302) and 'video' in resp.headers.get('Content-Type', '') + except: + return False + return check + + def get_fastest_host(self): + """测试候选域名,返回最快可用的""" + results = {} + threads = [] + + def test_host(url): + try: + start_time = time.time() + resp = requests.head(url, headers=self.headers, timeout=2, allow_redirects=False) + if resp.status_code in (200, 301, 302): + delay = (time.time() - start_time) * 1000 + results[url] = delay + else: + results[url] = float('inf') + except: + results[url] = float('inf') + + for host in self.candidate_hosts: + t = threading.Thread(target=test_host, args=(host,)) + threads.append(t) + t.start() + for t in threads: + t.join() + + valid_hosts = [(h, d) for h, d in results.items() if d != float('inf')] + return valid_hosts[0][0] if valid_hosts else self.candidate_hosts[0] + + def homeContent(self, filter): + result = {} + # 构造分类列表 + classes = [] + for name, info in self.class_map.items(): + classes.append({ + 'type_name': name, + 'type_id': info['type_id'] + }) + result['class'] = classes + + try: + # 获取首页内容 + html = self.fetch_with_retry(self.host, retry=2, timeout=5).text + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + # 提取图片URL + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + # 提取时长备注 + desc = item('.duration').text().strip() or '未知' + + # 提取视频ID + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析首页视频项失败: {e}") + continue + + result['list'] = vlist + except Exception as e: + print(f"首页解析失败: {e}") + result['list'] = [] + return result + + def homeVideoContent(self): + return [] + + def categoryContent(self, tid, pg, filter, extend): + result = {} + try: + # 匹配分类信息 + cate_info = None + for name, info in self.class_map.items(): + if info['type_id'] == tid: + cate_info = info + break + + if not cate_info: + result['list'] = [] + return result + + # 关键修复:区分视频分类与其他分类的URL格式 + if tid == 'list': # 视频分类(type_id为list) + url = f"{self.host}list-{pg}.htm" # 格式:list-1.htm、list-2.htm + else: # 其他分类(周榜/月榜等):xxx_list-{pg}.htm + url = f"{self.host}{cate_info['url_suffix']}_list-{pg}.htm" + + # 请求分类页 + html = self.fetch(url, headers=self.headers, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取视频列表 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析分类视频项失败: {e}") + continue + + # 提取总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + except: + pagecount = 1 + + result['list'] = vlist + result['page'] = pg + result['pagecount'] = pagecount + result['limit'] = len(vlist) + result['total'] = 999999 + except Exception as e: + print(f"分类解析失败: {e}") + result['list'] = [] + result['page'] = pg + result['pagecount'] = 1 + result['limit'] = 0 + result['total'] = 0 + return result + + def detailContent(self, ids): + try: + if not ids or not ids[0]: + return {'list': []} + + vod_id = ids[0].strip() + if not vod_id.endswith('.htm'): + vod_id += '.htm' + url = f"{self.host}{vod_id.lstrip('/')}" + + html = self.fetch_with_retry(url, retry=2, timeout=8).text + html = html.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 提取标题 + title = data('.panel-title, .video-title, h1').text().strip() or '未知标题' + + # 提取封面图 + vod_pic = '' + poster_style = data('.vjs-poster').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', poster_style) + if pic_match: + vod_pic = pic_match.group(1) + if not vod_pic: + vod_pic = data('.video-pic img, .vjs-poster img, .thumbnail img').attr('src') or '' + if vod_pic and not vod_pic.startswith('http'): + vod_pic = f"{self.host}{vod_pic.lstrip('/')}" + + # 提取时长和观看量 + duration = '未知' + views = '未知' + info_items = data('.panel-body .col-md-3, .video-info .info-item, .info p') + for item in info_items.items(): + text = item.text().strip() + if '时长' in text or 'duration' in text.lower(): + duration = text.replace('时长:', '').replace('时长', '').strip() + elif '观看' in text or 'views' in text.lower(): + views_match = re.search(r'(\d+\.?\d*[kK]?)次观看', text) + if views_match: + views = views_match.group(1) + else: + views = text.replace('观看:', '').replace('观看', '').strip() + remarks = f"{duration} | {views}" + + # 提取播放地址 + video_url = '' + m3u8_match = re.search(r'videoUrl\s*=\s*["\']([^"\']+\.m3u8)["\']', html) + if m3u8_match: + video_url = m3u8_match.group(1) + if not video_url: + source = data('source[src*=".m3u8"], source[src*=".mp4"]') + video_url = source.attr('src') or '' + if not video_url: + js_matches = re.findall(r'(https?://[^\s"\']+\.(?:m3u8|mp4))', html) + if js_matches: + video_url = js_matches[0] + + if video_url and not video_url.startswith('http'): + video_url = f"{self.host}{video_url.lstrip('/')}" + + vod = { + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': remarks, + 'vod_play_from': '好色TV(优)', + 'vod_play_url': f'正片${video_url}' if video_url else '正片$暂无地址' + } + return {'list': [vod]} + except Exception as e: + print(f"详情解析失败: {e}") + return {'list': []} + + def searchContent(self, key, quick, pg=1): + try: + # 关键词合法性校验 + if not key.strip(): + print("搜索关键词不能为空") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 编码关键词 + encoded_key = urllib.parse.quote(key.strip(), encoding='utf-8', errors='replace') + + # 构造搜索URL + search_url = f"{self.host}search.htm" + params = { + 'search': encoded_key, + 'page': int(pg) + } + + # 发起请求 + resp = self.fetch( + url=search_url, + headers=self.headers, + params=params, + timeout=8 + ) + if resp.status_code not in (200, 302): + print(f"搜索页面请求失败,URL:{resp.url},状态码:{resp.status_code}") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 处理页面内容 + html = resp.text.encode('utf-8', errors='ignore').decode('utf-8') + data = pq(html) + + # 检测无结果场景 + no_result_texts = ['没有找到相关视频', '无搜索结果', 'No results found', '未找到匹配内容'] + no_result = any(data(f'div:contains("{text}"), p:contains("{text}")').text() for text in no_result_texts) + if no_result: + print(f"搜索关键词「{key}」第{pg}页无结果") + return {'list': [], 'page': int(pg), 'pagecount': 1, 'limit': 0, 'total': 0} + + # 解析搜索结果 + vlist = [] + items = data('.row .col-xs-6.col-md-3') + for item in items.items(): + try: + title = item('h5').text().strip() + if not title: + continue + + style = item('.image').attr('style') or '' + pic_match = re.search(r'url\(["\']?([^"\']+)["\']?\)', style) + vod_pic = pic_match.group(1) if pic_match else '' + if vod_pic and not vod_pic.startswith(('http://', 'https://')): + vod_pic = f"{self.host.rstrip('/')}/{vod_pic.lstrip('/')}" + + desc = item('.duration').text().strip() or '未知时长' + + href = item('a').attr('href') or '' + if not href: + continue + vod_id = href.split('/')[-1] + if not vod_id.endswith('.htm'): + vod_id += '.htm' + + vlist.append({ + 'vod_id': vod_id, + 'vod_name': title, + 'vod_pic': vod_pic, + 'vod_remarks': desc + }) + except Exception as e: + print(f"解析单条搜索结果失败:{e}(跳过该条)") + continue + + # 解析总页数 + pagecount = 1 + try: + pagination = data('.pagination1 li a') + page_nums = [] + for a in pagination.items(): + text = a.text().strip() + if text.isdigit(): + page_nums.append(int(text)) + if page_nums: + pagecount = max(page_nums) + print(f"搜索关键词「{key}」分页解析完成,共{pagecount}页") + except Exception as e: + print(f"解析分页失败(默认单页):{e}") + pagecount = 1 + + # 返回结果(修复点2:补全page键的引号,修正语法错误) + total = len(vlist) * pagecount + print(f"搜索关键词「{key}」第{pg}页处理完成,结果{len(vlist)}条,总页数{pagecount}") + return { + 'list': vlist, + 'page': int(pg), # 原代码此处缺少引号,导致语法错误 + 'pagecount': pagecount, + 'limit': len(vlist), + 'total': total + } + except Exception as e: + print(f"搜索功能整体异常:{e}") + return { + 'list': [], + 'page': int(pg), 'pagecount': 1, + 'limit': 0, + 'total': 0 + } + + def playerContent(self, flag, id, vipFlags): + headers = self.headers.copy() + headers.update({ + 'Referer': self.host, + 'Origin': self.host.rstrip('/'), + 'Host': urllib.parse.urlparse(self.host).netloc, + }) + + # 根据rule中的double设置 + return { + 'parse': 1, # 根据rule中的play_parse设置 + 'url': id, + 'header': headers, + 'double': True # 根据rule中的double设置 + } + + def localProxy(self, param): + try: + url = param['url'] + if url and not url.startswith(('http://', 'https://')): + url = f"{self.host.rstrip('/')}/{url.lstrip('/')}" + + img_headers = self.headers.copy() + img_headers.update({'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8'}) + + res = self.fetch(url, headers=img_headers, timeout=10) + content_type = res.headers.get('Content-Type', 'image/jpeg') + + return [200, content_type, res.content] + except Exception as e: + print(f"图片代理失败: {e}") + return [200, 'image/jpeg', b''] + + def fetch_with_retry(self, url, retry=2, timeout=5): + for i in range(retry + 1): + try: + resp = self.fetch(f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}',headers=self.headers, timeout=timeout) + if resp.status_code in (200, 301, 302): + return resp + print(f"请求{url}返回状态码{resp.status_code},重试中...") + except Exception as e: + print(f"第{i+1}次请求{url}失败: {e}") + if i < retry: + time.sleep(0.5) + return type('obj', (object,), {'text': '', 'status_code': 404}) + + def fetch(self, url, headers=None, timeout=5, method='GET', params=None): + headers = headers or self.headers + params = params or {} + try: + if method.upper() == 'GET': + resp = requests.get( + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params # 支持GET请求带参数,适配搜索分页 + ) + elif method.upper() == 'HEAD': + resp = requests.head( + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=False, + params=params + ) + else: + resp = requests.get( # 默认GET请求,兼容其他方法调用 + f'https://vpsdn.leuse.top/proxy?single=true&url={urllib.parse.quote(url)}', + headers=headers, + timeout=timeout, + allow_redirects=True, + params=params + ) + # 自动适配编码,避免中文乱码 + if 'charset' in resp.headers.get('Content-Type', '').lower(): + resp.encoding = resp.apparent_encoding + else: + resp.encoding = 'utf-8' + return resp + except Exception as e: + print(f"网络请求失败({url}): {e}") + # 返回统一格式空响应,避免后续逻辑崩溃 + return type('obj', (object,), { + 'text': '', + 'status_code': 500, + 'headers': {}, + 'url': url + }) + + +# ------------------------------ +# 可选测试代码(运行时注释或删除,用于验证功能) +# ------------------------------ +if __name__ == "__main__": + # 初始化爬虫 + spider = Spider() + spider.init() + + # 测试首页内容 + print("=== 测试首页 ===") + home_data = spider.homeContent(filter='') + print(f"首页分类数:{len(home_data['class'])}") + print(f"首页视频数:{len(home_data['list'])}") + + # 测试视频分类(修复后的数据获取) + print("\n=== 测试视频分类(第1页) ===") + cate_data = spider.categoryContent(tid='list', pg=1, filter='', extend='') + print(f"视频分类第1页视频数:{len(cate_data['list'])}") + print(f"视频分类总页数:{cate_data['pagecount']}") + + # 测试搜索功能(修复语法错误后) + print("\n=== 测试搜索(关键词:测试) ===") + search_data = spider.searchContent(key="测试", quick=False, pg=1) + print(f"搜索结果数:{len(search_data['list'])}") + print(f"搜索总页数:{search_data['pagecount']}") diff --git a/lib/i推特APP[密].py b/lib/i推特APP[密].py new file mode 100644 index 0000000..0f363a5 --- /dev/null +++ b/lib/i推特APP[密].py @@ -0,0 +1,256 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '推特APP', + lang: 'hipy' +}) +""" + +# -*- coding: utf-8 -*- +# by @嗷呜 +import json +import random +import string +import sys +import time +from base64 import b64decode +from urllib.parse import quote +from Crypto.Cipher import AES +from Crypto.Hash import MD5 +from Crypto.Util.Padding import unpad +sys.path.append('..') +from base.spider import Spider + + +class Spider(Spider): + + def init(self, extend=""): + self.did = self.getdid() + self.token,self.phost,self.host = self.gettoken() + pass + + def isVideoFormat(self, url): + pass + + def manualVideoCheck(self): + pass + + def action(self, action): + pass + + def destroy(self): + pass + + hs=['wcyfhknomg','pdcqllfomw','alxhzjvean','bqeaaxzplt','hfbtpixjso'] + + ua='Mozilla/5.0 (Linux; Android 11; M2012K10C Build/RP1A.200720.011; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/87.0.4280.141 Mobile Safari/537.36;SuiRui/twitter/ver=1.4.4' + + def homeContent(self, filter): + data = self.fetch(f'{self.host}/api/video/classifyList', headers=self.headers()).json()['encData'] + data1 = self.aes(data) + result = {'filters': {"1": [{"key": "fl", "name": "分类", + "value": [{"n": "最近更新", "v": "1"}, {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], "2": [{"key": "fl", "name": "分类", + "value": [ + {"n": "最近更新", "v": "1"}, + {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], + "3": [{"key": "fl", "name": "分类", + "value": [{"n": "最近更新", "v": "1"}, {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], "4": [{"key": "fl", "name": "分类", + "value": [ + {"n": "最近更新", "v": "1"}, + {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], + "5": [{"key": "fl", "name": "分类", + "value": [{"n": "最近更新", "v": "1"}, {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], "6": [{"key": "fl", "name": "分类", + "value": [ + {"n": "最近更新", "v": "1"}, + {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], + "7": [{"key": "fl", "name": "分类", + "value": [{"n": "最近更新", "v": "1"}, {"n": "最多播放", "v": "2"}, + {"n": "好评榜", "v": "3"}]}], "jx": [{"key": "type", "name": "精选", + "value": [{"n": "日榜", "v": "1"}, + {"n": "周榜", "v": "2"}, + {"n": "月榜", "v": "3"}, + {"n": "总榜", + "v": "4"}]}]}} + classes = [{'type_name': "精选", 'type_id': "jx"}] + for k in data1['data']: + classes.append({'type_name': k['classifyTitle'], 'type_id': k['classifyId']}) + result['class'] = classes + return result + + def homeVideoContent(self): + pass + + def categoryContent(self, tid, pg, filter, extend): + path = f'/api/video/queryVideoByClassifyId?pageSize=20&page={pg}&classifyId={tid}&sortType={extend.get("fl", "1")}' + if 'click' in tid: + path = f'/api/video/queryPersonVideoByType?pageSize=20&page={pg}&userId={tid.replace("click", "")}' + if tid == 'jx': + path = f'/api/video/getRankVideos?pageSize=20&page={pg}&type={extend.get("type", "1")}' + data = self.fetch(f'{self.host}{path}', headers=self.headers()).json()['encData'] + data1 = self.aes(data)['data'] + result = {} + videos = [] + for k in data1: + id = f'{k.get("videoId")}?{k.get("userId")}?{k.get("nickName")}' + if 'click' in tid: + id = id + 'click' + videos.append({"vod_id": id, 'vod_name': k.get('title'), 'vod_pic': self.getProxyUrl() + f"&url={k.get('coverImg')[0]}", + 'vod_remarks': self.dtim(k.get('playTime')),'style': {"type": "rect", "ratio": 1.33}}) + result["list"] = videos + result["page"] = pg + result["pagecount"] = 9999 + result["limit"] = 90 + result["total"] = 999999 + return result + + def detailContent(self, ids): + vid = ids[0].replace('click', '').split('?') + path = f'/api/video/can/watch?videoId={vid[0]}' + data = self.fetch(f'{self.host}{path}', headers=self.headers()).json()['encData'] + data1 = self.aes(data)['playPath'] + clj = '[a=cr:' + json.dumps({'id': vid[1] + 'click', 'name': vid[2]}) + '/]' + vid[2] + '[/a]' + if 'click' in ids[0]: + clj = vid[2] + vod = {'vod_director': clj, 'vod_play_from': "推特", 'vod_play_url': vid[2] + "$" + data1} + result = {"list": [vod]} + return result + + def searchContent(self, key, quick, pg='1'): + path = f'/api/search/keyWord?pageSize=20&page={pg}&searchWord={quote(key)}&searchType=1' + data = self.fetch(f'{self.host}{path}', headers=self.headers()).json()['encData'] + data1 = self.aes(data)['videoList'] + result = {} + videos = [] + for k in data1: + id = f'{k.get("videoId")}?{k.get("userId")}?{k.get("nickName")}' + videos.append({"vod_id": id, 'vod_name': k.get('title'), 'vod_pic': self.getProxyUrl() + f"&url={k.get('coverImg')[0]}", + 'vod_remarks': self.dtim(k.get('playTime')), 'style': {"type": "rect", "ratio": 1.33}}) + result["list"] = videos + result["page"] = pg + result["pagecount"] = 9999 + result["limit"] = 90 + result["total"] = 999999 + return result + + def playerContent(self, flag, id, vipFlags): + return {"parse": 0, "url": id, "header": self.headers()} + + def localProxy(self, param): + return self.imgs(param) + + def getsign(self): + t = str(int(time.time() * 1000)) + sign = self.md5(t) + return sign, t + + def headers(self): + sign, t = self.getsign() + return {'User-Agent': self.ua,'deviceid': self.did, 't': t, 's': sign, 'aut': self.token} + + def aes(self, word): + key = b64decode("SmhiR2NpT2lKSVV6STFOaQ==") + iv = key + cipher = AES.new(key, AES.MODE_CBC, iv) + decrypted = unpad(cipher.decrypt(b64decode(word)), AES.block_size) + return json.loads(decrypted.decode('utf-8')) + + def dtim(self, seconds): + try: + seconds = int(seconds) + hours = seconds // 3600 + remaining_seconds = seconds % 3600 + minutes = remaining_seconds // 60 + remaining_seconds = remaining_seconds % 60 + + formatted_minutes = str(minutes).zfill(2) + formatted_seconds = str(remaining_seconds).zfill(2) + + if hours > 0: + formatted_hours = str(hours).zfill(2) + return f"{formatted_hours}:{formatted_minutes}:{formatted_seconds}" + else: + return f"{formatted_minutes}:{formatted_seconds}" + except: + return "666" + + def gettoken(self, i=0, max_attempts=10): + if i >= len(self.hs) or i >= max_attempts: + return '' + current_domain = f"https://{''.join(random.choices(string.ascii_lowercase + string.digits, k=random.randint(5, 10)))}.{self.hs[i]}.work" + try: + url = f'{current_domain}/api/user/traveler' + sign, t = self.getsign() + headers = { + 'User-Agent': self.ua, + 'Accept': 'application/json', + 'deviceid': self.did, + 't': t, + 's': sign, + } + data = { + 'deviceId': self.did, + 'tt': 'U', + 'code': '##X-4m6Goo4zzPi1hF##', + 'chCode': 'tt09' + } + response = self.post(url, json=data, headers=headers) + response.raise_for_status() + data1 = response.json()['data'] + return data1['token'], data1['imgDomain'], current_domain + except Exception as e: + return self.gettoken(i + 1, max_attempts) + + def getdid(self): + did = self.getCache('did') + if not did: + t = str(int(time.time())) + did = self.md5(t) + self.setCache('did', did) + return did + + def md5(self, text): + h = MD5.new() + h.update(text.encode('utf-8')) + return h.hexdigest() + + def imgs(self, param): + headers = {'User-Agent': self.ua} + url = param['url'] + data = self.fetch(f"{self.phost}{url}",headers=headers) + bdata = self.img(data.content, 100, '2020-zq3-888') + return [200, data.headers.get('Content-Type'), bdata] + + def img(self, data: bytes, length: int, key: str): + GIF = b'\x47\x49\x46' + JPG = b'\xFF\xD8\xFF' + PNG = b'\x89\x50\x4E\x47\x0D\x0A\x1A\x0A' + + def is_dont_need_decode_for_gif(data): + return len(data) > 2 and data[:3] == GIF + + def is_dont_need_decode_for_jpg(data): + return len(data) > 7 and data[:3] == JPG + + def is_dont_need_decode_for_png(data): + return len(data) > 7 and data[1:8] == PNG[1:8] + + if is_dont_need_decode_for_png(data): + return data + elif is_dont_need_decode_for_gif(data): + return data + elif is_dont_need_decode_for_jpg(data): + return data + else: + key_bytes = key.encode('utf-8') + result = bytearray(data) + for i in range(length): + result[i] ^= key_bytes[i % len(key_bytes)] + return bytes(result) diff --git a/lib/j黄色仓库[密].py b/lib/j黄色仓库[密].py new file mode 100644 index 0000000..248b2a8 --- /dev/null +++ b/lib/j黄色仓库[密].py @@ -0,0 +1,455 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: 'http://6590ck.cc/', + lang: 'hipy' +}) +""" + +import re +import sys +import urllib.parse +import threading +import time +import requests +import base64 +import gzip +import json +from io import BytesIO +from pyquery import PyQuery as pq + +sys.path.append('..') +from base.spider import Spider + +class Spider(Spider): + def __init__(self): + self.name = "黄色仓库" + self.host = self.getDynamicHost() + self.classes = self.preprocessClasses() + + def getName(self): + return self.name + + def getDynamicHost(self): + """动态获取主机地址""" + try: + # 解码base64获取初始主机 + initial_host = base64.b64decode('aHR0cDovL2hzY2submV0').decode('utf-8') + + # 获取初始页面 + response = requests.get(initial_host, headers=self.header) + html = response.text + + # 匹配strU参数 + strU_match = re.search(r'strU="(.*?)"', html) + if not strU_match: + return initial_host + + strU = strU_match.group(1) + locationU = strU + initial_host.rstrip('/') + '/&p=/' + + # 获取重定向地址 + redirect_response = requests.get(locationU, headers=self.header, allow_redirects=False) + if 'location' in redirect_response.headers: + return redirect_response.headers['location'] + else: + # 尝试从JSON响应中获取 + try: + data = redirect_response.json() + return data.get('location', initial_host) + except: + return initial_host + + except Exception as e: + print(f"获取动态主机失败: {e}") + return "http://6590ck.cc/" + + def preprocessClasses(self): + """预处理分类数据""" + return [ + {"type_name": "日韩AV", "type_id": "1"}, + {"type_name": "国产系列", "type_id": "2"}, + {"type_name": "欧美", "type_id": "3"}, + {"type_name": "成人动漫", "type_id": "4"}, + {"type_name": "日本有码", "type_id": "7"}, + {"type_name": "一本道高清无码", "type_id": "8"}, + {"type_name": "有码中文字幕", "type_id": "9"}, + {"type_name": "日本无码", "type_id": "10"}, + {"type_name": "国产视频", "type_id": "15"}, + {"type_name": "欧美高清", "type_id": "21"}, + {"type_name": "动漫剧情", "type_id": "22"} + ] + + def init(self, extend): + pass + + def isVideoFormat(self, url): + pass + + def manualVideoCheck(self): + pass + + def homeContent(self, filter): + result = {} + result['class'] = self.classes + return result + + def homeVideoContent(self): + """推荐内容""" + result = {} + try: + url = f"{self.host.rstrip('/')}/" + rsp = self.fetch(url) + root = pq(rsp.text) + + videos = [] + list_items = root('.stui-vodlist li') + + for item in list_items.items(): + vid = item.find('a').attr('href') + if not vid or not vid.startswith('/vodplay/'): + continue + + name = item.find('h4').text() + img = item.find('a').attr('data-original') + remark = item.find('.pic-text').text() + + if not name or not img: + continue + + videos.append({ + "vod_id": vid, # 只保存相对路径 + "vod_name": name, + "vod_pic": self.getFullUrl(img), + "vod_remarks": remark + }) + + result['list'] = videos + except Exception as e: + print(f"获取推荐内容失败: {e}") + result['list'] = [] + + return result + + def categoryContent(self, tid, pg, filter, extend): + result = {} + try: + url = f"{self.host.rstrip('/')}/vodtype/{tid}-{pg}.html" + rsp = self.fetch(url) + root = pq(rsp.text) + + videos = [] + list_items = root('.stui-vodlist li') + + for item in list_items.items(): + vid = item.find('a').attr('href') + if not vid or not vid.startswith('/vodplay/'): + continue + + name = item.find('h4').text() + img = item.find('a').attr('data-original') + remark = item.find('.pic-text').text() + + if not name or not img: + continue + + videos.append({ + "vod_id": vid, # 只保存相对路径 + "vod_name": name, + "vod_pic": self.getFullUrl(img), + "vod_remarks": remark + }) + + result['list'] = videos + result['page'] = int(pg) + result['pagecount'] = 9999 + result['limit'] = 6 + result['total'] = 999999 + + except Exception as e: + print(f"获取分类内容失败: {e}") + result['list'] = [] + result['page'] = 1 + result['pagecount'] = 1 + result['limit'] = 6 + result['total'] = 0 + + return result + + def extractM3U8Url(self, script_text): + """专门提取m3u8播放链接的方法""" + m3u8_urls = [] + + print("开始提取m3u8链接...") + + # 方法1: 从player_aaaa JavaScript变量中提取 + player_patterns = [ + r'var\s+player_aaaa\s*=\s*({.*?});', + r'player_aaaa\s*=\s*({.*?});', + r'var\s+player_aaaa\s*=\s*({.*?})\s*<\/script>', + r'player_aaaa\s*=\s*({.*?})\s*<\/script>' + ] + + for pattern in player_patterns: + player_match = re.search(pattern, script_text, re.DOTALL) + if player_match: + try: + player_data_str = player_match.group(1) + print(f"找到player_aaaa数据: {player_data_str[:200]}...") + + # 修复JSON字符串 + player_data_str = player_data_str.replace('\\/', '/') + player_data = json.loads(player_data_str) + + m3u8_url = player_data.get('url') + if m3u8_url and '.m3u8' in m3u8_url: + print(f"从player_aaaa提取到m3u8: {m3u8_url}") + # 确保URL完整 + if not m3u8_url.startswith('http'): + if m3u8_url.startswith('//'): + m3u8_url = 'https:' + m3u8_url + else: + m3u8_url = self.getFullUrl(m3u8_url) + m3u8_urls.append(m3u8_url) + return m3u8_urls # 找到就返回 + + except Exception as e: + print(f"解析player_aaaa失败: {e}") + + # 方法2: 直接搜索m3u8链接 + m3u8_patterns = [ + r'"url"\s*:\s*"([^"]+\.m3u8[^"]*)"', + r'url\s*:\s*"([^"]+\.m3u8[^"]*)"', + r'src\s*:\s*"([^"]+\.m3u8[^"]*)"', + r'file\s*:\s*"([^"]+\.m3u8[^"]*)"', + r'https?://[^\s"\'<>]+\.m3u8[^\s"\'<>]*' + ] + + for pattern in m3u8_patterns: + matches = re.findall(pattern, script_text) + for match in matches: + if '.m3u8' in match and match not in m3u8_urls: + print(f"从正则匹配提取到m3u8: {match}") + # 确保URL完整 + if not match.startswith('http'): + if match.startswith('//'): + match = 'https:' + match + else: + match = self.getFullUrl(match) + m3u8_urls.append(match) + return m3u8_urls # 找到就返回 + + print("未找到m3u8播放链接") + return m3u8_urls + + def detailContent(self, array): + """二级详情页面解析 - 修复播放链接提取及简介使用标题""" + result = {} + try: + vid = array[0] + # 确保vid是完整URL + if not vid.startswith('http'): + vid = self.getFullUrl(vid) + + print(f"开始解析详情页面: {vid}") + rsp = self.fetch(vid) + root = pq(rsp.text) + + # 提取基本信息 + title = root('.stui-pannel__head .title').text() + if not title: + title = root('title').text().split(' - ')[0] + + # 提取封面图 + pic = root('.stui-vodlist__thumb').attr('data-original') or root('.stui-vodlist__thumb').attr('src') + if not pic: + pic = root('img').attr('src') + + # 获取所有script内容 + script_text = root('script').text() + + # 提取m3u8播放链接 + m3u8_urls = self.extractM3U8Url(script_text) + + # 构建播放链接 + play_urls = [] + if m3u8_urls: + for i, m3u8_url in enumerate(m3u8_urls): + play_urls.append(f"线路{i+1}${m3u8_url}") + else: + # 如果没有找到m3u8链接,尝试从页面其他位置提取 + print("尝试从页面其他位置提取播放链接...") + # 从iframe中提取 + iframe_src = root('iframe').attr('src') + if iframe_src and 'm3u8' in iframe_src: + if not iframe_src.startswith('http'): + iframe_src = self.getFullUrl(iframe_src) + play_urls.append(f"iframe线路${iframe_src}") + else: + # 最后使用详情页URL + play_urls.append(f"详情页线路${vid}") + + # 用视频标题作为简介,避免没有简介内容 + vod = { + "vod_id": array[0], # 保持原始ID + "vod_name": title, + "vod_pic": self.getFullUrl(pic) if pic else "", + "vod_content": title, # 用标题当简介 + "vod_play_from": "黄色仓库", + "vod_play_url": "#".join(play_urls) # 使用#分隔多个播放源 + } + + result['list'] = [vod] + print(f"详情页解析完成,播放链接: {vod['vod_play_url']}") + + except Exception as e: + print(f"解析详情页面失败: {e}") + import traceback + traceback.print_exc() + # 返回基础信息 + result['list'] = [{ + "vod_id": array[0], + "vod_name": "未知标题", + "vod_pic": "", + "vod_content": "", + "vod_play_from": "默认线路", + "vod_play_url": f"详情页线路${array[0]}" + }] + + return result + + def searchContent(self, key, quick): + result = {} + try: + # 使用搜索URL + search_url = f"{self.host.rstrip('/')}/vodsearch/-------------.html?wd={urllib.parse.quote(key)}" + rsp = self.fetch(search_url) + root = pq(rsp.text) + + videos = [] + list_items = root('.stui-vodlist li') + + for item in list_items.items(): + vid = item.find('a').attr('href') + if not vid or not vid.startswith('/vodplay/'): + continue + + name = item.find('h4').text() + img = item.find('a').attr('data-original') + remark = item.find('.pic-text').text() + + if not name or not img: + continue + + videos.append({ + "vod_id": vid, # 只保存相对路径 + "vod_name": name, + "vod_pic": self.getFullUrl(img), + "vod_remarks": remark + }) + + result['list'] = videos + except Exception as e: + print(f"搜索失败: {e}") + result['list'] = [] + + return result + + def playerContent(self, flag, id, vipFlags): + """播放页面解析 - 修复数组越界问题""" + result = {} + try: + print(f"playerContent被调用: flag={flag}, id={id}") + + # 如果id已经是m3u8链接,直接返回 + if id.startswith('http') and '.m3u8' in id: + result["parse"] = 0 + result["playUrl"] = "" + result["url"] = id + result["header"] = self.header + print(f"直接返回m3u8链接: {id}") + return result + + # 如果id是播放线路格式,提取m3u8链接 + if '#' in id: + play_sources = id.split('#') + for source in play_sources: + if '$' in source: + _, url = source.split('$', 1) + if '.m3u8' in url: + result["parse"] = 0 + result["playUrl"] = "" + result["url"] = url + result["header"] = self.header + print(f"从播放线路提取到m3u8: {url}") + return result + + # 如果id是详情页链接,重新解析详情页 + print(f"重新解析详情页获取m3u8: {id}") + detail_result = self.detailContent([id]) + + if detail_result and 'list' in detail_result and detail_result['list']: + vod = detail_result['list'][0] + play_url = vod.get('vod_play_url', '') + print(f"从详情页获取的播放链接: {play_url}") + + # 解析播放链接 + if '#' in play_url: + play_sources = play_url.split('#') + for source in play_sources: + if '$' in source: + _, url = source.split('$', 1) + if '.m3u8' in url: + result["parse"] = 0 + result["playUrl"] = "" + result["url"] = url + result["header"] = self.header + print(f"最终提取到m3u8: {url}") + return result + + # 如果没有找到m3u8,使用第一个播放源 + if play_sources: + first_source = play_sources[0] + if '$' in first_source: + _, url = first_source.split('$', 1) + result["parse"] = 0 + result["playUrl"] = "" + result["url"] = url + result["header"] = self.header + print(f"使用第一个播放源: {url}") + return result + + # 如果所有方法都失败,返回空结果 + print("无法提取播放链接,返回空结果") + return {} + + except Exception as e: + print(f"解析播放页面失败: {e}") + import traceback + traceback.print_exc() + return {} + + def getFullUrl(self, url): + """获取完整的URL""" + if not url: + return "" + if url.startswith('http'): + return url + if url.startswith('//'): + return f"https:{url}" + return f"{self.host.rstrip('/')}{url}" + + config = { + "player": {}, + "filter": {} + } + header = { + "User-Agent": "Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 Mobile/15E148 Safari/604.1", + "Referer": "https://hsck123.com/" + } + + def localProxy(self, param): + action = {} + return action \ No newline at end of file diff --git a/lib/k神秘影院[密].py b/lib/k神秘影院[密].py new file mode 100644 index 0000000..c7152c4 --- /dev/null +++ b/lib/k神秘影院[密].py @@ -0,0 +1,271 @@ +""" +@header({ + searchable: 1, + filterable: 1, + quickSearch: 1, + title: '神秘影院[密]', + lang: 'hipy' +}) +""" + +# -*- coding: utf-8 -*- +#恰逢 +import re +import urllib.parse +from base.spider import Spider as BaseSpile +import requests +from bs4 import BeautifulSoup + + +class VideoDecryptor: + """XOR 128 解密""" + @staticmethod + def decrypt(text: str) -> str: + if not text: + return "" + try: + return ''.join(chr(128 ^ ord(c)) for c in text) + except: + return text + + @staticmethod + def from_js(js: str) -> str: + return VideoDecryptor.decrypt(m.group(1)) if (m := re.search(r"document\.write\(l\('([^']+)'\)\)", js)) else "" + + +class Spider(BaseSpile): + + def init(self, extend=""): + self.host = "https://h4ivs.sm431.vip" + self.video_host = "https://m3u8.nl:88" + self.image_host = "https://3334.nl:33" + self.headers = { + "User-Agent": "Mozilla/5.0 (Linux; Android 13; 22127RK46C Build/TKQ1.220905.001; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/104.0.5112.97 Mobile Safari/537.36", + "Referer": self.host, + "Accept-Language": "zh-CN,zh;q=0.9", + } + self.cache = {} + + def get(self, url): + try: + r = requests.get(url, headers=self.headers, timeout=15) + r.raise_for_status() + r.encoding = "utf-8" + return r.text + except: + return "" + + def img_url(self, url): + """格式化图片URL""" + if not url: + return "" + if url.startswith("//"): + url = "https:" + url + elif url.startswith("/"): + url = self.image_host + url + return f"{url}@User-Agent={self.headers['User-Agent']}@Referer={self.host}/" + + def parse(self, el): + """解析卡片""" + a = el if el.name == 'a' else el.find('a') + if not a or not (href := a.get("href", "")): + return None + + href = self.host + href if href.startswith("/") else href + if not (vid := re.search(r"/vid/(\d+)", href)): + return None + + vid = vid.group(1) + title = "" + + # 解密标题 + if p := el.find('p'): + if s := p.find('script'): + if s.string: + title = VideoDecryptor.from_js(s.string) + title = title or p.get_text(strip=True) + + if not title: + for attr in ['data-title', 'data-name', 'title']: + if el.has_attr(attr) and (val := el[attr]): + if (de := VideoDecryptor.decrypt(val)) and len(de) > 3: + title = de + break + + title = title or "未知标题" + if title != "未知标题": + self.cache[vid] = title + + # 图片 + img = "" + if node := el.select_one("img"): + img = node.get("data-src") or node.get("src") or "" + img = img or f"{self.image_host}/{vid}.jpg" + + return { + "vod_id": vid, + "vod_name": title, + "vod_pic": self.img_url(img), + "vod_remarks": "", + } + + def get_title(self, vid): + """从缓存或首页获取标题""" + if vid in self.cache: + return self.cache[vid] + + if html := self.get(self.host): + soup = BeautifulSoup(html, "html.parser") + for link in soup.select('a[href*="/vid/"]'): + if f'/vid/{vid}' in link.get('href', ''): + if p := link.find('p'): + if s := p.find('script'): + if s.string and (t := VideoDecryptor.from_js(s.string)): + self.cache[vid] = t + return t + if t := p.get_text(strip=True): + self.cache[vid] = t + return t + return None + + def homeContent(self, filter): + return { + "class": [ + {"type_name": "国产", "type_id": "1"}, + {"type_name": "日本", "type_id": "2"}, + {"type_name": "韩国", "type_id": "3"}, + {"type_name": "欧美", "type_id": "4"}, + {"type_name": "三级", "type_id": "5"}, + {"type_name": "动漫", "type_id": "6"}, + ] + } + + def homeVideoContent(self): + if not (html := self.get(self.host)): + return {"list": []} + + soup = BeautifulSoup(html, "html.parser") + videos = [v for v in (self.parse(el) for el in soup.select(".vodbox, .stui-vodlist__box, .vodlist__box, .video-card, .item")) if v] + + if not videos: + videos = [{"vod_id": v, "vod_name": "未知标题", "vod_pic": self.img_url(f"{self.image_host}/{v}.jpg"), "vod_remarks": ""} + for v in re.findall(r'\[]\(/vid/(\d+)\.html\)', html)] + + return {"list": videos} + + def categoryContent(self, tid, pg, filter, extend): + if tid == "0": + url = self.host if int(pg) == 1 else f"{self.host}/page/{pg}.html" + else: + url = f"{self.host}/list/{tid}.html" if int(pg) == 1 else f"{self.host}/list/{tid}/{pg}.html" + + if not (html := self.get(url)): + return {"list": [], "page": pg, "pagecount": 1, "limit": 30, "total": 0} + + soup = BeautifulSoup(html, "html.parser") + videos = [v for v in (self.parse(el) for el in soup.select(".vodbox, .stui-vodlist__box, .vodlist__box, .video-card, .item")) if v] + + if not videos: + videos = [{"vod_id": v, "vod_name": "未知标题", "vod_pic": self.img_url(f"{self.image_host}/{v}.jpg"), "vod_remarks": ""} + for v in re.findall(r'\[]\(/vid/(\d+)\.html\)', html)] + + last = max([int(m.group(1)) for a in soup.select("a[href*='list/']") if (m := re.search(r"/list/\d+/(\d+)\.html", a.get("href", "")))], default=int(pg)) + + return {"list": videos, "page": pg, "pagecount": max(last, 1), "limit": 30, "total": 99999} + + def searchContent(self, key, quick, pg="1"): + url = f"{self.host}/so.html" + params = {"wd": key} + if int(pg) > 1: + params["page"] = pg + + html = "" + for method in [requests.get, requests.post]: + try: + r = method(url, params=params if method == requests.get else None, + data=params if method == requests.post else None, + headers=self.headers, timeout=15) + r.raise_for_status() + r.encoding = "utf-8" + html = r.text + break + except: + continue + + if not html: + return {"list": []} + + soup = BeautifulSoup(html, "html.parser") + videos = [v for v in (self.parse(el) for el in soup.select(".vodbox, .stui-vodlist__box, .vodlist__box, .video-card, .item")) if v] + + if not videos: + videos = [{"vod_id": v, "vod_name": "未知标题", "vod_pic": self.img_url(f"{self.image_host}/{v}.jpg"), "vod_remarks": ""} + for v in re.findall(r'\[]\(/vid/(\d+)\.html\)', html)] + + last = max([int(m.group(1)) for a in soup.select("a[href*='so.html'], .pagination a, .page-link") + if (m := re.search(r"[?&]page=(\d+)", a.get("href", "")))], default=int(pg)) + + return {"list": videos, "page": pg, "pagecount": max(last, 1), "limit": 30, "total": 99999} + + def detailContent(self, ids): + vid = ids[0] + if not (html := self.get(f"{self.host}/vid/{vid}.html")): + return {"list": []} + + soup = BeautifulSoup(html, "html.parser") + + # 标题 + title = self.get_title(vid) + if not title: + if t := soup.find('title'): + title = re.sub(r'\s*[-_|]\s*.{0,20}$', '', t.get_text(strip=True)).strip() + + if not title or len(title) < 5: + for sel in ['h1', 'h2', '.video-title', '.title']: + if (el := soup.select_one(sel)) and (txt := el.get_text(strip=True)) and len(txt) > 5: + title = txt + break + + title = title or f"视频{vid}" + + # 图片 + pic = "" + for sel in ['.picbox img', '.vodimg img', '.video-pic img', '.poster img', 'img[data-id]']: + if (node := soup.select_one(sel)) and (p := node.get("data-src") or node.get("src")) and 'favicon' not in p.lower(): + pic = p + break + + if not pic or 'favicon' in pic.lower(): + if meta := soup.select_one('meta[property="og:image"]'): + pic = meta.get('content', '') + + pic = pic or f"{self.image_host}/{vid}.jpg" + + # 简介 + desc = soup.select_one(".vodinfo, .video-info, .content, .intro, .description") + desc = desc.get_text(strip=True) if desc else "" + + return {"list": [{ + "vod_id": vid, + "vod_name": title, + "vod_pic": self.img_url(pic), + "vod_content": desc, + "vod_play_from": "七哥比较瑟", + "vod_play_url": f"狗哥特别瑟${vid}@@0@@1", + }]} + + def playerContent(self, flag, id, vipFlags): + vid = id.split("@@")[0] + return {"parse": 0, "url": f"{self.video_host}/{vid}/hls/index.m3u8", "header": self.headers} + + def localProxy(self, param): + return {"code": 404, "content": ""} + + def isVideoFormat(self, url): + return ".m3u8" in url.lower() + + def manualVideoCheck(self): + pass + + def destroy(self): + pass \ No newline at end of file