删除 py/奈飞.py
This commit is contained in:
@@ -1,373 +0,0 @@
|
|||||||
# -*- coding: utf-8 -*-
|
|
||||||
"""
|
|
||||||
奈飞影视 - naifei.im
|
|
||||||
"""
|
|
||||||
import re
|
|
||||||
import json
|
|
||||||
import sys
|
|
||||||
import time
|
|
||||||
from urllib.parse import quote, urljoin
|
|
||||||
from base.spider import Spider
|
|
||||||
|
|
||||||
|
|
||||||
class Spider(Spider):
|
|
||||||
def __init__(self):
|
|
||||||
super(Spider, self).__init__()
|
|
||||||
self.host = "https://naifei.im"
|
|
||||||
self.name = "奈飞影视"
|
|
||||||
self.headers = {
|
|
||||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
|
||||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
|
|
||||||
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
|
|
||||||
'Accept-Encoding': 'gzip, deflate, br',
|
|
||||||
'Connection': 'keep-alive',
|
|
||||||
'Upgrade-Insecure-Requests': '1',
|
|
||||||
'Sec-Fetch-Dest': 'document',
|
|
||||||
'Sec-Fetch-Mode': 'navigate',
|
|
||||||
'Sec-Fetch-Site': 'none',
|
|
||||||
'Sec-Fetch-User': '?1',
|
|
||||||
'Cache-Control': 'max-age=0',
|
|
||||||
'Referer': self.host
|
|
||||||
}
|
|
||||||
self.categories = {
|
|
||||||
'1': '电影',
|
|
||||||
'2': '剧集',
|
|
||||||
'3': '综艺',
|
|
||||||
'4': '动漫',
|
|
||||||
'5': '短剧'
|
|
||||||
}
|
|
||||||
self._detail_cache = {}
|
|
||||||
|
|
||||||
def getName(self):
|
|
||||||
return "奈飞影视"
|
|
||||||
|
|
||||||
def init(self, extend=""):
|
|
||||||
pass
|
|
||||||
|
|
||||||
def homeContent(self, filter):
|
|
||||||
classes = [
|
|
||||||
{"type_id": "1", "type_name": "电影"},
|
|
||||||
{"type_id": "2", "type_name": "剧集"},
|
|
||||||
{"type_id": "3", "type_name": "综艺"},
|
|
||||||
{"type_id": "4", "type_name": "动漫"},
|
|
||||||
{"type_id": "5", "type_name": "短剧"},
|
|
||||||
]
|
|
||||||
return {'class': classes, 'filters': {}, 'list': []}
|
|
||||||
|
|
||||||
def homeVideoContent(self):
|
|
||||||
try:
|
|
||||||
videos = self._fetch_home()
|
|
||||||
return {'list': videos}
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 首页爬取失败: {e}')
|
|
||||||
return {'list': []}
|
|
||||||
|
|
||||||
def categoryContent(self, tid, pg, filter, extend):
|
|
||||||
try:
|
|
||||||
page = int(pg) if pg and str(pg).isdigit() else 1
|
|
||||||
videos = self._fetch_category(tid, page)
|
|
||||||
return {
|
|
||||||
'page': page,
|
|
||||||
'pagecount': 9999,
|
|
||||||
'limit': 20,
|
|
||||||
'total': 99999,
|
|
||||||
'list': videos
|
|
||||||
}
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 分类爬取失败: {e}')
|
|
||||||
return {'page': int(pg), 'pagecount': 0, 'limit': 20, 'total': 0, 'list': []}
|
|
||||||
|
|
||||||
def detailContent(self, ids):
|
|
||||||
try:
|
|
||||||
vod_id = ids[0] if isinstance(ids, list) else ids
|
|
||||||
detail = self._fetch_detail(vod_id)
|
|
||||||
if detail:
|
|
||||||
return {'list': [detail]}
|
|
||||||
return {'list': []}
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 详情爬取失败: {e}')
|
|
||||||
return {'list': []}
|
|
||||||
|
|
||||||
def playerContent(self, flag, id, vipFlags):
|
|
||||||
try:
|
|
||||||
play_url = ''
|
|
||||||
if id and id.startswith('http'):
|
|
||||||
play_url = id
|
|
||||||
elif '$' in str(id):
|
|
||||||
parts = str(id).split('$', 1)
|
|
||||||
if len(parts) == 2:
|
|
||||||
play_url = parts[1]
|
|
||||||
else:
|
|
||||||
play_url = id
|
|
||||||
|
|
||||||
# 如果是网页链接,需要解析获取真实播放地址
|
|
||||||
if play_url and 'naifei.im' in play_url:
|
|
||||||
real_url = self._parse_play_url(play_url)
|
|
||||||
if real_url:
|
|
||||||
play_url = real_url
|
|
||||||
|
|
||||||
return {
|
|
||||||
'parse': 0,
|
|
||||||
'playUrl': '',
|
|
||||||
'url': play_url,
|
|
||||||
}
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 播放失败: {e}')
|
|
||||||
return {
|
|
||||||
'parse': 1,
|
|
||||||
'playUrl': '',
|
|
||||||
'url': str(id),
|
|
||||||
}
|
|
||||||
|
|
||||||
def _parse_play_url(self, url):
|
|
||||||
"""解析播放页面获取真实播放地址"""
|
|
||||||
import re
|
|
||||||
|
|
||||||
html = self._fetch_page(url)
|
|
||||||
if not html:
|
|
||||||
return None
|
|
||||||
|
|
||||||
# 直接提取url字段
|
|
||||||
url_match = re.search(r'"url"\s*:\s*"(https?:[^"]+)"', html)
|
|
||||||
if url_match:
|
|
||||||
video_url = url_match.group(1).replace('\\/', '/')
|
|
||||||
if video_url and video_url.startswith('http'):
|
|
||||||
return video_url
|
|
||||||
|
|
||||||
# 备用:直接匹配m3u8地址
|
|
||||||
m3u8_match = re.search(r'(https?://[^\s"\'\\]+\.m3u8[^\s"\'\\]*)', html)
|
|
||||||
if m3u8_match:
|
|
||||||
return m3u8_match.group(1).replace('\\/', '/')
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
def searchContent(self, key, quick, pg="1"):
|
|
||||||
try:
|
|
||||||
page = int(pg) if pg and str(pg).isdigit() else 1
|
|
||||||
videos = self._fetch_search(key, page)
|
|
||||||
return {'list': videos}
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 搜索失败: {e}')
|
|
||||||
return {'list': []}
|
|
||||||
|
|
||||||
def _fetch_page(self, url, retries=2):
|
|
||||||
"""获取页面内容"""
|
|
||||||
import requests
|
|
||||||
session = requests.Session()
|
|
||||||
session.headers.update(self.headers)
|
|
||||||
|
|
||||||
for attempt in range(retries + 1):
|
|
||||||
try:
|
|
||||||
resp = session.get(url, timeout=15)
|
|
||||||
|
|
||||||
if resp.status_code == 403:
|
|
||||||
redirect_match = re.search(r'window\.location\.href\s*=\s*"([^"]+)"', resp.text)
|
|
||||||
if redirect_match:
|
|
||||||
redirect_path = redirect_match.group(1)
|
|
||||||
if redirect_path.startswith('/'):
|
|
||||||
new_url = self.host + redirect_path
|
|
||||||
else:
|
|
||||||
new_url = redirect_path
|
|
||||||
resp = session.get(new_url, timeout=15)
|
|
||||||
elif attempt < retries:
|
|
||||||
time.sleep(1)
|
|
||||||
session.get(self.host, timeout=10)
|
|
||||||
continue
|
|
||||||
|
|
||||||
resp.raise_for_status()
|
|
||||||
resp.encoding = 'utf-8'
|
|
||||||
return resp.text
|
|
||||||
except Exception as e:
|
|
||||||
if attempt < retries:
|
|
||||||
time.sleep(1)
|
|
||||||
continue
|
|
||||||
print(f'[{self.name}] 请求失败: {url}, 错误: {e}')
|
|
||||||
return ''
|
|
||||||
return ''
|
|
||||||
|
|
||||||
def _fetch_home(self):
|
|
||||||
"""获取首页视频"""
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
|
|
||||||
html = self._fetch_page(self.host)
|
|
||||||
if not html:
|
|
||||||
return []
|
|
||||||
|
|
||||||
soup = BeautifulSoup(html, 'html.parser')
|
|
||||||
videos = []
|
|
||||||
|
|
||||||
items_containers = soup.find_all('div', class_='module-items')
|
|
||||||
for container in items_containers:
|
|
||||||
items = container.find_all('a', class_='module-poster-item')
|
|
||||||
for item in items[:20]:
|
|
||||||
vod = self._parse_video_item(item)
|
|
||||||
if vod:
|
|
||||||
videos.append(vod)
|
|
||||||
|
|
||||||
return videos[:50]
|
|
||||||
|
|
||||||
def _fetch_category(self, tid, page=1):
|
|
||||||
"""获取分类视频"""
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
|
|
||||||
if page <= 1:
|
|
||||||
url = f"{self.host}/vodtype/{tid}.html"
|
|
||||||
else:
|
|
||||||
url = f"{self.host}/vodtype/{tid}-{page}.html"
|
|
||||||
|
|
||||||
html = self._fetch_page(url)
|
|
||||||
if not html:
|
|
||||||
return []
|
|
||||||
|
|
||||||
soup = BeautifulSoup(html, 'html.parser')
|
|
||||||
videos = []
|
|
||||||
|
|
||||||
items = soup.find_all('a', class_='module-poster-item')
|
|
||||||
for item in items:
|
|
||||||
vod = self._parse_video_item(item)
|
|
||||||
if vod:
|
|
||||||
videos.append(vod)
|
|
||||||
|
|
||||||
return videos
|
|
||||||
|
|
||||||
def _fetch_detail(self, vid):
|
|
||||||
"""获取视频详情"""
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
|
|
||||||
if vid in self._detail_cache:
|
|
||||||
return self._detail_cache[vid]
|
|
||||||
|
|
||||||
url = f"{self.host}/voddetail/{vid}.html"
|
|
||||||
html = self._fetch_page(url)
|
|
||||||
if not html:
|
|
||||||
return None
|
|
||||||
|
|
||||||
soup = BeautifulSoup(html, 'html.parser')
|
|
||||||
|
|
||||||
result = {"vod_id": vid}
|
|
||||||
|
|
||||||
title = soup.find('h1', class_='video-info-heading')
|
|
||||||
result['vod_name'] = title.text.strip() if title else ''
|
|
||||||
|
|
||||||
cover = soup.find('img', class_='lazy lazyload')
|
|
||||||
if cover:
|
|
||||||
pic = cover.get('data-original', '') or cover.get('src', '')
|
|
||||||
if pic and pic.startswith('//'):
|
|
||||||
pic = 'https:' + pic
|
|
||||||
result['vod_pic'] = pic
|
|
||||||
else:
|
|
||||||
result['vod_pic'] = ''
|
|
||||||
|
|
||||||
info_items = soup.find_all('li', class_='list-item')
|
|
||||||
for item in info_items:
|
|
||||||
text = item.text.strip()
|
|
||||||
if '主演' in text:
|
|
||||||
result['vod_actor'] = text.split(':', 1)[-1] if ':' in text else ''
|
|
||||||
elif '导演' in text:
|
|
||||||
result['vod_director'] = text.split(':', 1)[-1] if ':' in text else ''
|
|
||||||
elif '地区' in text or '语言' in text:
|
|
||||||
result['vod_area'] = text.split(':', 1)[-1] if ':' in text else ''
|
|
||||||
elif '年份' in text:
|
|
||||||
result['vod_year'] = text.split(':', 1)[-1] if ':' in text else ''
|
|
||||||
elif '更新' in text or '集数' in text:
|
|
||||||
result['vod_remarks'] = text.split(':', 1)[-1] if ':' in text else ''
|
|
||||||
|
|
||||||
desc = soup.find('div', class_='video-info-content')
|
|
||||||
result['vod_content'] = desc.text.strip() if desc else ''
|
|
||||||
|
|
||||||
episodes = []
|
|
||||||
episode_list = soup.find('div', class_='module-play-list')
|
|
||||||
if episode_list:
|
|
||||||
ep_items = episode_list.find_all('a')
|
|
||||||
for ep in ep_items:
|
|
||||||
ep_link = ep.get('href', '')
|
|
||||||
ep_title = ep.text.strip()
|
|
||||||
if ep_title and ep_link:
|
|
||||||
full_url = urljoin(self.host, ep_link) if ep_link.startswith('/') else ep_link
|
|
||||||
episodes.append(f'{ep_title}${full_url}')
|
|
||||||
|
|
||||||
if episodes:
|
|
||||||
result['vod_play_from'] = '奈飞影视'
|
|
||||||
result['vod_play_url'] = '#'.join(episodes)
|
|
||||||
else:
|
|
||||||
result['vod_play_from'] = ''
|
|
||||||
result['vod_play_url'] = ''
|
|
||||||
|
|
||||||
self._detail_cache[vid] = result
|
|
||||||
return result
|
|
||||||
|
|
||||||
def _fetch_search(self, keyword, page=1):
|
|
||||||
"""搜索视频"""
|
|
||||||
import json
|
|
||||||
|
|
||||||
url = f"{self.host}/index.php/ajax/suggest?mid=1&limit=20&wd={quote(keyword)}"
|
|
||||||
html = self._fetch_page(url)
|
|
||||||
if not html:
|
|
||||||
return []
|
|
||||||
|
|
||||||
videos = []
|
|
||||||
try:
|
|
||||||
data = json.loads(html)
|
|
||||||
if data.get('code') == 1 and data.get('list'):
|
|
||||||
for item in data['list']:
|
|
||||||
vod = self._parse_search_item(item)
|
|
||||||
if vod:
|
|
||||||
videos.append(vod)
|
|
||||||
except Exception as e:
|
|
||||||
print(f'[{self.name}] 解析搜索结果失败: {e}')
|
|
||||||
|
|
||||||
return videos
|
|
||||||
|
|
||||||
def _parse_search_item(self, item):
|
|
||||||
"""解析搜索结果项"""
|
|
||||||
try:
|
|
||||||
vid = str(item.get('id', ''))
|
|
||||||
name = item.get('name', '')
|
|
||||||
if not vid or not name:
|
|
||||||
return None
|
|
||||||
|
|
||||||
pic = item.get('pic', '')
|
|
||||||
if pic and pic.startswith('//'):
|
|
||||||
pic = 'https:' + pic
|
|
||||||
|
|
||||||
return {
|
|
||||||
'vod_id': vid,
|
|
||||||
'vod_name': name,
|
|
||||||
'vod_pic': pic,
|
|
||||||
'vod_remarks': '',
|
|
||||||
}
|
|
||||||
except Exception as e:
|
|
||||||
return None
|
|
||||||
|
|
||||||
def _parse_video_item(self, item):
|
|
||||||
"""解析视频项"""
|
|
||||||
try:
|
|
||||||
link = item.get('href', '')
|
|
||||||
title = item.get('title', '')
|
|
||||||
|
|
||||||
img = item.find('img')
|
|
||||||
cover = ''
|
|
||||||
if img:
|
|
||||||
cover = img.get('data-original', '') or img.get('src', '')
|
|
||||||
if cover and cover.startswith('//'):
|
|
||||||
cover = 'https:' + cover
|
|
||||||
|
|
||||||
note = item.find('div', class_='module-item-note')
|
|
||||||
quality = note.text.strip() if note else ''
|
|
||||||
|
|
||||||
vid = ''
|
|
||||||
match = re.search(r'/voddetail/(\d+)\.html', link)
|
|
||||||
if match:
|
|
||||||
vid = match.group(1)
|
|
||||||
|
|
||||||
if not vid or not title:
|
|
||||||
return None
|
|
||||||
|
|
||||||
return {
|
|
||||||
'vod_id': vid,
|
|
||||||
'vod_name': title,
|
|
||||||
'vod_pic': cover,
|
|
||||||
'vod_remarks': quality,
|
|
||||||
}
|
|
||||||
except Exception as e:
|
|
||||||
return None
|
|
||||||
Reference in New Issue
Block a user