mirror of
https://github.com/JefferyHcool/BiliNote.git
synced 2026-09-06 16:16:56 +08:00
之前 BilibiliDownloader.download_subtitles 走的是 yt-dlp 的 writesubtitles 路径,对 B 站签名/Cookie 的兼容性差,常常空手而归,落到音频下载 + Whisper 转写的慢路径。 新增 bilibili_subtitle.BilibiliSubtitleFetcher: - /x/web-interface/view?bvid=... → 拿 cid - /x/player/wbi/v2?bvid=...&cid=... → 拿 subtitle 列表(subtitle_url 已带 auth_key) - 优先级:人工中文 > AI 中文 > 任意中文 > 任意非空 - fetch JSON body 解析为 TranscriptResult - 通过 CookieConfigManager 自动注入 SESSDATA cookie(AI 字幕必需) bilibili_downloader.download_subtitles 顺序改为:先试新 fetcher,失败再回退到原 yt-dlp 路径。NoteGenerator 的字幕优先逻辑无需改动——它本来就调 download_subtitles。 效果: - B 站视频如果有字幕(人工或 AI),直接秒拿,跳过音频下载 + 转写 - 完全绕开 MLX Whisper 不可用 / 模型未下载 等转写器问题 - 拿不到字幕时仍可走原音频转写路径 Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
343 lines
12 KiB
Python
343 lines
12 KiB
Python
import os
|
||
import json
|
||
import logging
|
||
import tempfile
|
||
from abc import ABC
|
||
from typing import Union, Optional, List
|
||
|
||
import yt_dlp
|
||
|
||
from app.downloaders.base import Downloader, DownloadQuality, QUALITY_MAP
|
||
from app.downloaders.bilibili_subtitle import BilibiliSubtitleFetcher
|
||
from app.models.notes_model import AudioDownloadResult
|
||
from app.models.transcriber_model import TranscriptResult, TranscriptSegment
|
||
from app.utils.path_helper import get_data_dir
|
||
from app.utils.url_parser import extract_video_id
|
||
from app.services.cookie_manager import CookieConfigManager
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
|
||
class BilibiliDownloader(Downloader, ABC):
|
||
def __init__(self):
|
||
super().__init__()
|
||
self._cookie_mgr = CookieConfigManager()
|
||
self._cookie = self._cookie_mgr.get('bilibili')
|
||
self._cookiefile = self._write_netscape_cookie_file()
|
||
|
||
def _write_netscape_cookie_file(self) -> Optional[str]:
|
||
"""将 Cookie 写入 Netscape 格式临时文件,返回文件路径(供 yt-dlp cookiefile 使用)"""
|
||
if not self._cookie:
|
||
logger.warning("B站 Cookie 未配置,下载可能失败")
|
||
return None
|
||
lines = ["# Netscape HTTP Cookie File\n"]
|
||
for pair in self._cookie.split("; "):
|
||
if "=" in pair:
|
||
key, value = pair.split("=", 1)
|
||
lines.append(f".bilibili.com\tTRUE\t/\tFALSE\t0\t{key}\t{value}\n")
|
||
tmp = tempfile.NamedTemporaryFile(mode='w', suffix='.txt', delete=False, encoding='utf-8')
|
||
tmp.writelines(lines)
|
||
tmp.close()
|
||
logger.info("已生成 B站 Netscape Cookie 文件: %s (条目: %d)", tmp.name, len(lines) - 1)
|
||
return tmp.name
|
||
|
||
def download(
|
||
self,
|
||
video_url: str,
|
||
output_dir: Union[str, None] = None,
|
||
quality: DownloadQuality = "fast",
|
||
need_video:Optional[bool]=False
|
||
) -> AudioDownloadResult:
|
||
if output_dir is None:
|
||
output_dir = get_data_dir()
|
||
if not output_dir:
|
||
output_dir=self.cache_data
|
||
os.makedirs(output_dir, exist_ok=True)
|
||
|
||
output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
|
||
|
||
ydl_opts = {
|
||
'format': 'bestaudio[ext=m4a]/bestaudio/best',
|
||
'outtmpl': output_path,
|
||
'http_headers': {'Referer': 'https://www.bilibili.com'},
|
||
'postprocessors': [
|
||
{
|
||
'key': 'FFmpegExtractAudio',
|
||
'preferredcodec': 'mp3',
|
||
'preferredquality': '64',
|
||
}
|
||
],
|
||
'noplaylist': True,
|
||
'quiet': False,
|
||
}
|
||
if self._cookiefile:
|
||
ydl_opts['cookiefile'] = self._cookiefile
|
||
|
||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||
info = ydl.extract_info(video_url, download=True)
|
||
video_id = info.get("id")
|
||
title = info.get("title")
|
||
duration = info.get("duration", 0)
|
||
cover_url = info.get("thumbnail")
|
||
audio_path = os.path.join(output_dir, f"{video_id}.mp3")
|
||
|
||
return AudioDownloadResult(
|
||
file_path=audio_path,
|
||
title=title,
|
||
duration=duration,
|
||
cover_url=cover_url,
|
||
platform="bilibili",
|
||
video_id=video_id,
|
||
raw_info=info,
|
||
video_path=None # ❗音频下载不包含视频路径
|
||
)
|
||
|
||
def download_video(
|
||
self,
|
||
video_url: str,
|
||
output_dir: Union[str, None] = None,
|
||
) -> str:
|
||
"""
|
||
下载视频,返回视频文件路径
|
||
"""
|
||
|
||
if output_dir is None:
|
||
output_dir = get_data_dir()
|
||
os.makedirs(output_dir, exist_ok=True)
|
||
print("video_url",video_url)
|
||
video_id=extract_video_id(video_url, "bilibili")
|
||
video_path = os.path.join(output_dir, f"{video_id}.mp4")
|
||
if os.path.exists(video_path):
|
||
return video_path
|
||
|
||
# 检查是否已经存在
|
||
|
||
|
||
output_path = os.path.join(output_dir, "%(id)s.%(ext)s")
|
||
|
||
ydl_opts = {
|
||
'format': 'bv*[ext=mp4]/bestvideo+bestaudio/best',
|
||
'outtmpl': output_path,
|
||
'http_headers': {'Referer': 'https://www.bilibili.com'},
|
||
'noplaylist': True,
|
||
'quiet': False,
|
||
'merge_output_format': 'mp4', # 确保合并成 mp4
|
||
}
|
||
if self._cookiefile:
|
||
ydl_opts['cookiefile'] = self._cookiefile
|
||
|
||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||
info = ydl.extract_info(video_url, download=True)
|
||
video_id = info.get("id")
|
||
video_path = os.path.join(output_dir, f"{video_id}.mp4")
|
||
|
||
if not os.path.exists(video_path):
|
||
raise FileNotFoundError(f"视频文件未找到: {video_path}")
|
||
|
||
return video_path
|
||
|
||
def delete_video(self, video_path: str) -> str:
|
||
"""
|
||
删除视频文件
|
||
"""
|
||
if os.path.exists(video_path):
|
||
os.remove(video_path)
|
||
return f"视频文件已删除: {video_path}"
|
||
else:
|
||
return f"视频文件未找到: {video_path}"
|
||
|
||
def download_subtitles(self, video_url: str, output_dir: str = None,
|
||
langs: List[str] = None) -> Optional[TranscriptResult]:
|
||
"""
|
||
尝试获取B站视频字幕
|
||
|
||
:param video_url: 视频链接
|
||
:param output_dir: 输出路径
|
||
:param langs: 优先语言列表
|
||
:return: TranscriptResult 或 None
|
||
"""
|
||
# 1) 优先走 B 站官方 player API(直拉,无需下视频;AI 字幕需 SESSDATA cookie)
|
||
try:
|
||
result = BilibiliSubtitleFetcher().fetch_subtitles(video_url)
|
||
if result and result.segments:
|
||
return result
|
||
except Exception as e:
|
||
logger.warning(f"player API 直拉字幕异常,回退到 yt-dlp: {e}")
|
||
|
||
# 2) Fallback:原 yt-dlp 路径(更脆弱,遇到签名/Cookie 问题失败概率较高)
|
||
if output_dir is None:
|
||
output_dir = get_data_dir()
|
||
if not output_dir:
|
||
output_dir = self.cache_data
|
||
os.makedirs(output_dir, exist_ok=True)
|
||
|
||
if langs is None:
|
||
langs = ['zh-Hans', 'zh', 'zh-CN', 'ai-zh', 'en', 'en-US']
|
||
|
||
video_id = extract_video_id(video_url, "bilibili")
|
||
|
||
ydl_opts = {
|
||
'writesubtitles': True,
|
||
'writeautomaticsub': True,
|
||
'subtitleslangs': langs,
|
||
'subtitlesformat': 'srt/json3/best', # 支持多种格式
|
||
'skip_download': True,
|
||
'outtmpl': os.path.join(output_dir, f'{video_id}.%(ext)s'),
|
||
'quiet': True,
|
||
}
|
||
|
||
# 通过 CookieConfigManager 注入 B站 Cookie(Netscape cookiefile)
|
||
if self._cookiefile:
|
||
ydl_opts['cookiefile'] = self._cookiefile
|
||
ydl_opts['http_headers'] = {'Referer': 'https://www.bilibili.com'}
|
||
|
||
try:
|
||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||
info = ydl.extract_info(video_url, download=True)
|
||
|
||
# 查找下载的字幕文件
|
||
subtitles = info.get('requested_subtitles') or {}
|
||
if not subtitles:
|
||
logger.info(f"B站视频 {video_id} 没有可用字幕")
|
||
return None
|
||
|
||
# 按优先级查找字幕
|
||
detected_lang = None
|
||
sub_info = None
|
||
for lang in langs:
|
||
if lang in subtitles:
|
||
detected_lang = lang
|
||
sub_info = subtitles[lang]
|
||
break
|
||
|
||
# 如果按优先级没找到,取第一个可用的(排除弹幕)
|
||
if not detected_lang:
|
||
for lang, info_item in subtitles.items():
|
||
if lang != 'danmaku': # 排除弹幕
|
||
detected_lang = lang
|
||
sub_info = info_item
|
||
break
|
||
|
||
if not sub_info:
|
||
logger.info(f"B站视频 {video_id} 没有可用字幕(排除弹幕)")
|
||
return None
|
||
|
||
# 检查是否有内嵌数据(yt-dlp 有时直接返回字幕内容)
|
||
if 'data' in sub_info and sub_info['data']:
|
||
logger.info(f"直接从返回数据解析字幕: {detected_lang}")
|
||
return self._parse_srt_content(sub_info['data'], detected_lang)
|
||
|
||
# 查找字幕文件
|
||
ext = sub_info.get('ext', 'srt')
|
||
subtitle_file = os.path.join(output_dir, f"{video_id}.{detected_lang}.{ext}")
|
||
|
||
if not os.path.exists(subtitle_file):
|
||
logger.info(f"字幕文件不存在: {subtitle_file}")
|
||
return None
|
||
|
||
# 根据格式解析字幕文件
|
||
if ext == 'json3':
|
||
return self._parse_json3_subtitle(subtitle_file, detected_lang)
|
||
else:
|
||
with open(subtitle_file, 'r', encoding='utf-8') as f:
|
||
return self._parse_srt_content(f.read(), detected_lang)
|
||
|
||
except Exception as e:
|
||
logger.warning(f"获取B站字幕失败: {e}")
|
||
return None
|
||
|
||
def _parse_srt_content(self, srt_content: str, language: str) -> Optional[TranscriptResult]:
|
||
"""
|
||
解析 SRT 格式字幕内容
|
||
|
||
:param srt_content: SRT 字幕文本内容
|
||
:param language: 语言代码
|
||
:return: TranscriptResult
|
||
"""
|
||
import re
|
||
try:
|
||
segments = []
|
||
# SRT 格式: 序号\n时间戳\n文本\n\n
|
||
pattern = r'(\d+)\n(\d{2}:\d{2}:\d{2},\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2},\d{3})\n(.*?)(?=\n\n|\n\d+\n|$)'
|
||
matches = re.findall(pattern, srt_content, re.DOTALL)
|
||
|
||
for match in matches:
|
||
idx, start_time, end_time, text = match
|
||
text = text.strip()
|
||
if not text:
|
||
continue
|
||
|
||
# 转换时间格式 00:00:00,000 -> 秒
|
||
def time_to_seconds(t):
|
||
parts = t.replace(',', '.').split(':')
|
||
return float(parts[0]) * 3600 + float(parts[1]) * 60 + float(parts[2])
|
||
|
||
segments.append(TranscriptSegment(
|
||
start=time_to_seconds(start_time),
|
||
end=time_to_seconds(end_time),
|
||
text=text
|
||
))
|
||
|
||
if not segments:
|
||
return None
|
||
|
||
full_text = ' '.join(seg.text for seg in segments)
|
||
logger.info(f"成功解析B站SRT字幕,共 {len(segments)} 段")
|
||
return TranscriptResult(
|
||
language=language,
|
||
full_text=full_text,
|
||
segments=segments,
|
||
raw={'source': 'bilibili_subtitle', 'format': 'srt'}
|
||
)
|
||
|
||
except Exception as e:
|
||
logger.warning(f"解析SRT字幕失败: {e}")
|
||
return None
|
||
|
||
def _parse_json3_subtitle(self, subtitle_file: str, language: str) -> Optional[TranscriptResult]:
|
||
"""
|
||
解析 json3 格式字幕文件
|
||
|
||
:param subtitle_file: 字幕文件路径
|
||
:param language: 语言代码
|
||
:return: TranscriptResult
|
||
"""
|
||
try:
|
||
with open(subtitle_file, 'r', encoding='utf-8') as f:
|
||
data = json.load(f)
|
||
|
||
segments = []
|
||
events = data.get('events', [])
|
||
|
||
for event in events:
|
||
# json3 格式中时间单位是毫秒
|
||
start_ms = event.get('tStartMs', 0)
|
||
duration_ms = event.get('dDurationMs', 0)
|
||
|
||
# 提取文本
|
||
segs = event.get('segs', [])
|
||
text = ''.join(seg.get('utf8', '') for seg in segs).strip()
|
||
|
||
if text: # 只添加非空文本
|
||
segments.append(TranscriptSegment(
|
||
start=start_ms / 1000.0,
|
||
end=(start_ms + duration_ms) / 1000.0,
|
||
text=text
|
||
))
|
||
|
||
if not segments:
|
||
return None
|
||
|
||
full_text = ' '.join(seg.text for seg in segments)
|
||
|
||
logger.info(f"成功解析B站字幕,共 {len(segments)} 段")
|
||
return TranscriptResult(
|
||
language=language,
|
||
full_text=full_text,
|
||
segments=segments,
|
||
raw={'source': 'bilibili_subtitle', 'file': subtitle_file}
|
||
)
|
||
|
||
except Exception as e:
|
||
logger.warning(f"解析字幕文件失败: {e}")
|
||
return None |