# -*- coding: utf-8 -*-
# by @Qist
"""
ITalkBB TV - 海外华人影视
"""
import re
import requests
from base.spider import Spider
class Spider(Spider):
def getName(self):
return 'ITalkBB TV'
def init(self, extend=""):
pass
def isVideoFormat(self, url):
pass
def manualVideoCheck(self):
pass
def destroy(self):
pass
def __init__(self):
self.name = 'ITalkBB TV'
self.host = 'https://www.italkbbtv.com'
self.header = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36 Edg/136.0.0.0',
'Referer': 'https://www.italkbbtv.com/'
}
self.timeout = 20
self.class_names = '电视剧&直播频道&短剧&综艺&电影&动画'.split('&')
self.class_urls = 'drama/62c670dc1dca2d424404499c&live/62ac4e2e4beefe535864769d&shorts/66b1d25cf2dde82c215f9b59&variety/62ce7417c7daaa4a5d3fea14&movie/62ac4ef36e0b5a13ed291544&cartoon/62ac4e6e4beefe53586478ca'.split('&')
def parse_list_page(self, html):
if not html:
return []
cards = re.findall(r']*href="(/(?:play|shortsPlay)/[a-f0-9]+)"[^>]*>(.*?)', html, re.DOTALL)
vods = []
seen = set()
for href, content in cards:
sid = href.split('/')[-1]
if sid in seen:
continue
seen.add(sid)
name = ''
title_match = re.search(r'title="([^"]+)"', content)
if title_match:
name = title_match.group(1).strip()
if not name:
info_match = re.search(r'info-title[^>]*>([^<]+)', content)
if info_match:
name = info_match.group(1).strip()
if not name:
alt_match = re.search(r'alt="[^"]*[《]([^》]+)[》]', content)
if alt_match:
name = alt_match.group(1).strip()
img_match = re.search(r']*src="([^"]+)"', content)
pic = img_match.group(1) if img_match else ''
remarks = ''
ep_match = re.search(r'(全\d+集|更新至\d+集)', content)
if ep_match:
remarks = ep_match.group(1)
if name:
route = 'shortsPlay' if '/shortsPlay/' in href else 'play'
vods.append({
'vod_id': route + '$' + sid,
'vod_name': name,
'vod_pic': pic,
'vod_remarks': remarks
})
return vods
def parse_live_page(self, html):
if not html:
return []
m = re.search(r'window\.__NUXT__=([\s\S]*?);', html)
if not m:
return []
js = m.group(1)
channels = re.findall(r'\{[^{}]*id:"([a-f0-9]+)"[^{}]*name:"([^"]+)"', js)
vods = []
seen = set()
for ch_id, name in channels:
if ch_id in seen:
continue
seen.add(ch_id)
vods.append({
'vod_id': 'live@' + ch_id + '@' + name,
'vod_name': name,
'vod_pic': '',
'vod_remarks': '直播'
})
return vods
def _get_live_name(self, ch_id):
"""从直播页获取频道名称"""
html = self.fetch(self.host + '/live/62ac4e2e4beefe535864769d')
if not html:
return ch_id
m = re.search(r'window\.__NUXT__=([\s\S]*?);', html)
if not m:
return ch_id
js = m.group(1)
match = re.search(r'\{[^{}]*id:"' + ch_id + r'"[^{}]*name:"([^"]+)"', js)
return match.group(1) if match else ch_id
def homeContent(self, filter):
result = {'class': [], 'list': []}
for name, cid in zip(self.class_names, self.class_urls):
result['class'].append({'type_name': name, 'type_id': cid})
html = self.fetch(self.host + '/drama/62c670dc1dca2d424404499c')
result['list'] = self.parse_list_page(html)
return result
def homeVideoContent(self):
return {}
def categoryContent(self, tid, pg, filter, extend):
result = {'list': [], 'page': int(pg), 'pagecount': 999, 'limit': 24, 'total': 999999}
alias = tid.split('/')[0]
url = self.host + '/' + tid
if int(pg) > 1:
url += '?page=' + str(pg)
html = self.fetch(url)
if alias == 'live':
result['list'] = self.parse_live_page(html)
result['total'] = len(result['list'])
result['pagecount'] = 1
else:
result['list'] = self.parse_list_page(html)
return result
def detailContent(self, ids):
if not ids or not ids[0]:
return {'list': []}
vid = ids[0]
# 直播频道: live@ch_id 或 live@ch_id@name
if vid.startswith('live@'):
parts = vid.split('@', 2)
ch_id = parts[1] if len(parts) > 1 else ''
ch_name = parts[2] if len(parts) > 2 else self._get_live_name(ch_id)
return {'list': [{
'vod_id': vid,
'vod_name': ch_name,
'vod_play_from': 'ITalkBB直播',
'vod_play_url': '直播$' + ch_id
}]}
parts = vid.split('$')
route = parts[0] if len(parts) > 1 else 'play'
sid = parts[1] if len(parts) > 1 else vid
html = self.fetch(self.host + '/' + route + '/' + sid)
if not html:
return {'list': []}
# 从