mirror of
https://github.com/JefferyHcool/BiliNote.git
synced 2026-09-04 23:26:51 +08:00
## 问题
输入 YouTube 链接生成笔记时任务必定失败:
ERROR: [youtube] <id>: Requested format is not available
即使字幕已经抓取成功(日志里能看到"成功获取 YouTube 字幕,共 N 段"),
任务仍然在下载阶段崩掉。
## 原因
两个独立的问题叠加:
1. `requirements.txt` 把 yt-dlp 钉在 `2025.3.31`。YouTube 之后轮换过
player,旧版 yt-dlp 解不出 nsig 签名,所有音视频格式都被丢弃,只剩下
storyboard 图片(日志:`Only images are available for download`),
格式选择随即抛错。
2. 有字幕时 `NoteGenerator` 走的是"只取元信息"的路径
(`download(skip_download=True)`,只要 title/duration/cover),但
`YoutubeDownloader.download` 无条件设置了
`format='bestaudio[ext=m4a]/bestaudio/best'`。于是一个根本不需要媒体流的
调用,也会因为选不出格式而失败——把一个字幕已经到手的任务整个带崩。
## 改动
- `youtube_downloader.py`:`skip_download` 时设置
`ignore_no_formats_error=True`。yt-dlp 落后于 YouTube 时,只取元信息的路径
降级为"没有音频",而不是让整个笔记失败。
- `youtube_downloader.py`:`ext = info.get("ext") or "m4a"`。跳过下载时
yt-dlp 返回 `ext=None`,`dict.get` 的默认值对显式 None 不生效,
会拼出 `xxx.None` 这样的路径。
- `requirements.txt`:`yt-dlp==2025.3.31` → `>=2026.7.4`。yt-dlp 是对抗
YouTube 变化的滚动依赖,精确钉版本本身就是这个 bug 的成因;用 `>=` 与同文件
的 `youtube-transcript-api>=1.0.0` 保持一致。
- `bilibili_dm_patch.py`:wrapper 改为透传 `**kwargs`。升级 yt-dlp 后
`_real_extract` 会以 `fatal=False` 调用 `_download_playinfo`,而 wrapper
钉死了签名,导致 **所有 B 站下载** 抛
`TypeError: ... got an unexpected keyword argument 'fatal'`。
- 测试:新增 `test_youtube_metadata_only.py` 覆盖上面两条 YouTube 保证;
`test_bilibili_dm_patch.py` 新增未知 kwargs 透传用例,并让 fake 响应带上
`code` 字段(yt-dlp 2026.x 会先校验信封再返回 data)。
## 验证
- 真实跑通:YouTube(有字幕,走元信息路径)与 B 站(无字幕,走完整下载 +
转写)均能生成笔记。
- `pytest tests/` → 46 passed。唯一失败的
`test_task_serial_executor` 在升级前后表现一致,与本次改动无关,未作改动。
147 lines
4.9 KiB
Python
147 lines
4.9 KiB
Python
"""
|
|
Coverage for the YouTube "metadata only" download path.
|
|
|
|
Background: when a YouTube video already has subtitles, NoteGenerator skips the
|
|
audio download and calls `YoutubeDownloader.download(skip_download=True)` purely
|
|
to read title/duration/cover. That call used to still request
|
|
`format='bestaudio[ext=m4a]/bestaudio/best'`.
|
|
|
|
Whenever the installed yt-dlp lags behind YouTube's player, nsig extraction
|
|
fails, every audio/video format is dropped (only storyboard images remain) and
|
|
format selection raises "Requested format is not available" — killing a task
|
|
whose transcript had already been fetched successfully.
|
|
|
|
These tests pin the two guarantees of that path:
|
|
1. skip_download implies ignore_no_formats_error, so a formatless extraction
|
|
degrades to "no audio" instead of failing the whole note.
|
|
2. ext falls back to m4a, since yt-dlp reports ext=None when skipping the
|
|
download (dict.get's default does not fire on an explicit None).
|
|
"""
|
|
import importlib.util
|
|
import pathlib
|
|
import sys
|
|
import types
|
|
import unittest
|
|
|
|
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
|
MODULE_PATH = ROOT / "app" / "downloaders" / "youtube_downloader.py"
|
|
|
|
|
|
def _stub(name, **attrs):
|
|
module = types.ModuleType(name)
|
|
for key, value in attrs.items():
|
|
setattr(module, key, value)
|
|
sys.modules.setdefault(name, module)
|
|
return module
|
|
|
|
|
|
class _Downloader:
|
|
def __init__(self):
|
|
self.cache_data = "/tmp"
|
|
|
|
|
|
class _AudioDownloadResult:
|
|
def __init__(self, **kwargs):
|
|
self.__dict__.update(kwargs)
|
|
|
|
|
|
def _load_youtube_downloader():
|
|
"""Load the module with its app-level dependencies stubbed out."""
|
|
_stub("app")
|
|
_stub("app.downloaders")
|
|
_stub("app.models")
|
|
_stub("app.services")
|
|
_stub("app.utils")
|
|
_stub("app.downloaders.base", Downloader=_Downloader, DownloadQuality=str)
|
|
_stub("app.downloaders.youtube_subtitle", YouTubeSubtitleFetcher=object)
|
|
_stub("app.models.notes_model", AudioDownloadResult=_AudioDownloadResult)
|
|
_stub("app.models.transcriber_model", TranscriptResult=object)
|
|
_stub(
|
|
"app.services.proxy_config_manager",
|
|
ProxyConfigManager=type(
|
|
"ProxyConfigManager", (), {"get_proxy_url": lambda self: None}
|
|
),
|
|
)
|
|
_stub("app.utils.path_helper", get_data_dir=lambda: "/tmp")
|
|
_stub("app.utils.url_parser", extract_video_id=lambda url, platform: "vid")
|
|
|
|
spec = importlib.util.spec_from_file_location("youtube_downloader", MODULE_PATH)
|
|
if spec is None or spec.loader is None:
|
|
raise ImportError("youtube_downloader module spec not found")
|
|
module = importlib.util.module_from_spec(spec)
|
|
spec.loader.exec_module(module)
|
|
return module
|
|
|
|
|
|
class _FakeYoutubeDL:
|
|
"""Records the opts it was constructed with; mimics a formatless extraction."""
|
|
|
|
captured_opts = None
|
|
|
|
def __init__(self, opts):
|
|
type(self).captured_opts = opts
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *exc_info):
|
|
return False
|
|
|
|
def extract_info(self, url, download=True):
|
|
# What yt-dlp yields for a metadata-only extraction: no media, so no ext.
|
|
return {
|
|
"id": "CJ4ndXv3CkY",
|
|
"title": "example",
|
|
"duration": 2231,
|
|
"thumbnail": "https://example.invalid/t.jpg",
|
|
"ext": None,
|
|
"tags": [],
|
|
}
|
|
|
|
|
|
class YoutubeMetadataOnlyTest(unittest.TestCase):
|
|
@classmethod
|
|
def setUpClass(cls):
|
|
try:
|
|
cls.module = _load_youtube_downloader()
|
|
except Exception as exc: # pragma: no cover - env without yt-dlp
|
|
raise unittest.SkipTest(f"youtube_downloader not importable: {exc}")
|
|
|
|
def setUp(self):
|
|
self._real_ydl = self.module.yt_dlp.YoutubeDL
|
|
self.module.yt_dlp.YoutubeDL = _FakeYoutubeDL
|
|
_FakeYoutubeDL.captured_opts = None
|
|
|
|
def tearDown(self):
|
|
self.module.yt_dlp.YoutubeDL = self._real_ydl
|
|
|
|
def test_skip_download_tolerates_missing_formats(self):
|
|
self.module.YoutubeDownloader().download(
|
|
"https://www.youtube.com/watch?v=CJ4ndXv3CkY",
|
|
output_dir="/tmp",
|
|
skip_download=True,
|
|
)
|
|
self.assertTrue(_FakeYoutubeDL.captured_opts.get("ignore_no_formats_error"))
|
|
|
|
def test_missing_ext_falls_back_to_m4a(self):
|
|
result = self.module.YoutubeDownloader().download(
|
|
"https://www.youtube.com/watch?v=CJ4ndXv3CkY",
|
|
output_dir="/tmp",
|
|
skip_download=True,
|
|
)
|
|
self.assertTrue(result.file_path.endswith(".m4a"), result.file_path)
|
|
self.assertNotIn("None", result.file_path)
|
|
|
|
def test_full_download_still_selects_an_audio_format(self):
|
|
self.module.YoutubeDownloader().download(
|
|
"https://www.youtube.com/watch?v=CJ4ndXv3CkY",
|
|
output_dir="/tmp",
|
|
)
|
|
opts = _FakeYoutubeDL.captured_opts
|
|
self.assertIn("bestaudio", opts.get("format", ""))
|
|
self.assertNotIn("ignore_no_formats_error", opts)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|