Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 47 additions & 4 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,11 +18,33 @@ B 站视频自动字幕 Tampermonkey 脚本开发指南
- **格式处理**:优先下载 m4a 格式,与腾讯云 API 兼容性最好。
- **缓存机制**:使用 `CacheManager` (IndexedDB) 存储 Blob 数据,避免重复下载,设置 100MB 自动清理阈值。

### 2. AI 识别服务 (AISubtitleService)
### 2. 音频压缩 (AudioCompressor)
- **ffmpeg.wasm 集成**:使用 ffmpeg.wasm v0.12.x 实现浏览器端音频处理。
- **智能检测**:
- 自动检测音频文件大小是否超过 50MB(腾讯云 API 限制)。
- 提供 `needsCompression()` 方法供外部调用判断。
- **压缩引擎**:
- **懒加载**:首次调用时才加载 FFmpeg WASM 模块(约 30MB)。
- **复用机制**:加载后缓存实例,避免重复下载。
- **进度回调**:支持压缩进度监听(0-100%)。
- **压缩参数**:
- 输入格式:m4a
- 输出格式:opus(更高压缩率)
- 比特率:32kbps
- 采样率:16kHz
- 声道:单声道
- **错误处理**:完善的异常捕获和提示机制。

### 3. AI 识别服务 (AISubtitleService)
- **API 对接**:对接腾讯云录音文件识别极速版 (`/asr/flash/v1`)。
- **压缩集成**:
- 识别前自动检测文件大小。
- 超过 50MB 自动调用 AudioCompressor 压缩。
- 支持进度回调(压缩进度、上传状态)。
- **签名鉴权**:
- 内置纯 JavaScript 实现的 HMAC-SHA1 算法 (`HmacSha1`),不依赖 `Web Crypto API`,确保在所有 Tampermonkey 环境下的兼容性。
- 实现参数字典序排序和签名原文拼接。
- **多格式支持**:自动适配 m4a / opus 格式参数。
- **智能分段算法 (`_jsonToSrt`)**:
- **输入**:API 返回的 `sentence_list` (粗粒度) 和 `word_list` (词级)。
- **逻辑**:
Expand All @@ -34,7 +56,7 @@ B 站视频自动字幕 Tampermonkey 脚本开发指南
- 单词间停顿超过 500ms。
- **输出**:生成时间轴精准、长短适宜的 SRT 格式字幕。

### 3. 字幕渲染 (SubtitleRenderer)
### 4. 字幕渲染 (SubtitleRenderer)
- **DOM 注入**:自动侦测 B 站播放器容器 (`.bpx-player-video-area`) 并插入字幕层。
- **样式复刻**:
- 使用半透明黑底 (`rgba(0,0,0,0.6)`) 和白色文字。
Expand All @@ -61,7 +83,6 @@ B 站视频自动字幕 Tampermonkey 脚本开发指南
- **双语字幕**:利用翻译 API 实现实时双语字幕显示。

### 2. 性能与体验
- **长视频处理**:对于超过 2 小时或 100MB 的视频,实现前端音频分片上传或压缩(利用 ffmpeg.wasm)。
- **样式自定义**:提供字体大小、颜色、背景透明度的用户自定义选项。

## 已实现的优化
Expand All @@ -80,4 +101,26 @@ B 站视频自动字幕 Tampermonkey 脚本开发指南
- **缓存管理**:
- 字幕与音频文件关联存储(同一记录中)。
- UI 显示缓存状态("已缓存音频+字幕"或"已缓存音频")。
- 按钮文案智能切换("加载字幕 (使用缓存)"或"生成字幕 (腾讯云 AI)")。
- 按钮文案智能切换("加载字幕 (使用缓存)"或"生成字幕 (腾讯云 AI)")。

### ✅ 长视频智能压缩(v0.3.0)
- 集成 **ffmpeg.wasm** 实现浏览器端音频压缩,解决腾讯云 API 50MB 文件大小限制。
- **自动检测机制**:
- 实时检测音频文件大小。
- 当文件超过 50MB 时自动启动压缩流程。
- 小文件(≤50MB)直接上传,无性能损失。
- **压缩策略**:
- **编码器**:使用 Opus 编码(比 m4a 压缩率更高)。
- **参数优化**:
- 比特率:32kbps(语音识别场景足够)
- 声道:单声道(减少文件大小)
- 采样率:16kHz(匹配腾讯云 16k_zh 引擎)
- **压缩效果**:通常可将文件压缩至原大小的 10-20%,大幅降低 API 调用成本。
- **实时进度反馈**:
- 加载 FFmpeg 引擎进度提示。
- 压缩进度百分比显示。
- 压缩完成后显示节省空间比例。
- **技术细节**:
- 使用 CDN 加载 ffmpeg.wasm 和 ffmpeg-core(约 30MB)。
- 首次压缩会下载 WASM 文件,后续压缩无需重新加载。
- 支持 2 小时以上超长视频处理。
200 changes: 187 additions & 13 deletions bilibili-auto-subtitle.user.js
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
// ==UserScript==
// @name B站自动字幕
// @namespace http://tampermonkey.net/
// @version 0.2.1
// @description 为B站视频自动生成字幕,支持提取音频、AI识别、字幕缓存和字幕显示
// @version 0.3.0
// @description 为B站视频自动生成字幕,支持提取音频、AI识别、字幕缓存、长视频压缩和字幕显示
// @author You
// @match https://www.bilibili.com/video/*
// @icon https://www.bilibili.com/favicon.ico
Expand All @@ -11,6 +11,8 @@
// @grant GM_getValue
// @grant GM_registerMenuCommand
// @grant unsafeWindow
// @require https://cdn.jsdelivr.net/npm/@ffmpeg/ffmpeg@0.12.10/dist/umd/ffmpeg.min.js
// @require https://cdn.jsdelivr.net/npm/@ffmpeg/util@0.12.1/dist/umd/index.min.js
// @run-at document-end
// ==/UserScript==

Expand Down Expand Up @@ -137,7 +139,127 @@
})();

// ==========================================
// 模块 2: CacheManager (缓存管理模块)
// 模块 2: AudioCompressor (音频压缩模块)
// ==========================================
const AudioCompressor = (function() {
const MAX_SIZE = 50 * 1024 * 1024; // 50MB 腾讯云 API 限制
const MAX_DURATION = 2 * 3600; // 2 小时

let _ffmpeg = null;
let _isLoaded = false;

async function _loadFFmpeg(onProgress) {
if (_isLoaded && _ffmpeg) return _ffmpeg;

try {
// 使用 @ffmpeg/ffmpeg (v0.12.x)
const { FFmpeg } = FFmpegWASM;
const { toBlobURL, fetchFile } = FFmpegUtil;

_ffmpeg = new FFmpeg();

_ffmpeg.on('log', ({ message }) => {
console.log('[FFmpeg]', message);
});

if (onProgress) {
_ffmpeg.on('progress', ({ progress, time }) => {
onProgress(Math.round(progress * 100));
});
}

console.log('[AudioCompressor] 正在加载 FFmpeg WASM...');
const baseURL = 'https://cdn.jsdelivr.net/npm/@ffmpeg/core@0.12.6/dist/umd';

await _ffmpeg.load({
coreURL: await toBlobURL(`${baseURL}/ffmpeg-core.js`, 'text/javascript'),
wasmURL: await toBlobURL(`${baseURL}/ffmpeg-core.wasm`, 'application/wasm'),
});

_isLoaded = true;
console.log('[AudioCompressor] FFmpeg 加载完成');
return _ffmpeg;
} catch (e) {
console.error('[AudioCompressor] FFmpeg 加载失败:', e);
throw new Error('FFmpeg 加载失败: ' + e.message);
}
}

async function _compressAudio(audioBlob, filename, onProgress) {
const ffmpeg = await _loadFFmpeg(onProgress);

try {
console.log('[AudioCompressor] 开始压缩音频...');

// 写入输入文件
const { fetchFile } = FFmpegUtil;
await ffmpeg.writeFile('input.m4a', await fetchFile(audioBlob));

// 压缩参数:
// - 转换为 opus 编码(更高压缩率)
// - 降低比特率到 32kbps(语音识别足够)
// - 单声道
// - 采样率 16kHz(匹配 16k_zh 引擎)
await ffmpeg.exec([
'-i', 'input.m4a',
'-vn', // 去除视频流
'-c:a', 'libopus', // 使用 opus 编码
'-b:a', '32k', // 比特率 32kbps
'-ac', '1', // 单声道
'-ar', '16000', // 采样率 16kHz
'-f', 'opus', // 输出格式
'output.opus'
]);

// 读取输出文件
const data = await ffmpeg.readFile('output.opus');
const compressedBlob = new Blob([data.buffer], { type: 'audio/opus' });

// 清理临时文件
await ffmpeg.deleteFile('input.m4a');
await ffmpeg.deleteFile('output.opus');

const compressionRatio = ((1 - compressedBlob.size / audioBlob.size) * 100).toFixed(1);
console.log(`[AudioCompressor] 压缩完成: ${(audioBlob.size / 1024 / 1024).toFixed(1)}MB -> ${(compressedBlob.size / 1024 / 1024).toFixed(1)}MB (节省 ${compressionRatio}%)`);

return {
blob: compressedBlob,
filename: filename.replace('.m4a', '.opus'),
originalSize: audioBlob.size,
compressedSize: compressedBlob.size,
compressionRatio: compressionRatio
};
} catch (e) {
console.error('[AudioCompressor] 压缩失败:', e);
throw new Error('音频压缩失败: ' + e.message);
}
}

function _needsCompression(audioBlob) {
return audioBlob.size > MAX_SIZE;
}

return {
needsCompression: _needsCompression,

compress: async function(audioBlob, filename, onProgress) {
if (!this.needsCompression(audioBlob)) {
console.log('[AudioCompressor] 文件大小正常,无需压缩');
return { blob: audioBlob, filename: filename };
}

console.log(`[AudioCompressor] 文件过大 (${(audioBlob.size / 1024 / 1024).toFixed(1)}MB),开始压缩...`);
return await _compressAudio(audioBlob, filename, onProgress);
},

getMaxSize: () => MAX_SIZE,

isLoaded: () => _isLoaded
};
})();

// ==========================================
// 模块 3: CacheManager (缓存管理模块)
// ==========================================
const CacheManager = (function() {
const DB_NAME = 'BilibiliSubtitleCache';
Expand Down Expand Up @@ -266,7 +388,7 @@
})();

// ==========================================
// 模块 3: AISubtitleService (AI 接口服务)
// 模块 4: AISubtitleService (AI 接口服务)
// ==========================================
const AISubtitleService = (function() {

Expand Down Expand Up @@ -299,7 +421,7 @@
const TencentCloudProvider = {
name: 'tencent',

transcribe: async function(audioBlob) {
transcribe: async function(audioBlob, voiceFormat = 'm4a') {
// 从 ConfigManager 读取配置
const config = ConfigManager.get();

Expand All @@ -308,7 +430,7 @@
secretid: config.SECRET_ID,
engine_type: config.ENGINE_TYPE,
timestamp: timestamp,
voice_format: 'm4a',
voice_format: voiceFormat,
speaker_diarization: 0,
filter_dirty: 0,
filter_modal: 0,
Expand Down Expand Up @@ -437,18 +559,58 @@
let _currentProvider = TencentCloudProvider;

return {
transcribe: async function(audioBlob) {
transcribe: async function(audioBlob, onProgress) {
if (!_currentProvider) throw new Error('未设置 AI 提供者');
if (!ConfigManager.isConfigured()) {
throw new Error('请先配置腾讯云 API 密钥');
}
return _currentProvider.transcribe(audioBlob);

// 检查文件大小,如果超过限制则压缩
let finalBlob = audioBlob;
let voiceFormat = 'm4a';

if (AudioCompressor.needsCompression(audioBlob)) {
console.log('[AISubtitleService] 检测到大文件,启动压缩流程...');
if (onProgress) {
onProgress({ type: 'compress', message: '正在加载压缩引擎...' });
}

const result = await AudioCompressor.compress(
audioBlob,
'audio.m4a',
(percent) => {
if (onProgress) {
onProgress({
type: 'compress',
message: `正在压缩音频: ${percent}%`,
percent: percent
});
}
}
);

finalBlob = result.blob;
voiceFormat = result.filename.endsWith('.opus') ? 'opus' : 'm4a';

if (onProgress) {
onProgress({
type: 'compress_done',
message: `压缩完成,节省 ${result.compressionRatio}%`
});
}
}

if (onProgress) {
onProgress({ type: 'upload', message: '正在上传到腾讯云...' });
}

return _currentProvider.transcribe(finalBlob, voiceFormat);
}
};
})();

// ==========================================
// 模块 4: SRTParser (SRT 解析器)
// 模块 5: SRTParser (SRT 解析器)
// ==========================================
const SRTParser = {
parse: function(srtContent) {
Expand Down Expand Up @@ -485,7 +647,7 @@
};

// ==========================================
// 模块 5: SubtitleRenderer (字幕渲染模块)
// 模块 6: SubtitleRenderer (字幕渲染模块)
// ==========================================
const SubtitleRenderer = (function() {
let _container = null;
Expand Down Expand Up @@ -646,7 +808,7 @@
})();

// ==========================================
// 模块 6: UI Manager (界面管理)
// 模块 7: UI Manager (界面管理)
// ==========================================
const UIManager = (function() {
let _container = null;
Expand Down Expand Up @@ -937,8 +1099,20 @@
}

console.log('[UIManager] 字幕未缓存,调用 API 识别');
_updateStatus('上传腾讯云识别中...');
srt = await AISubtitleService.transcribe(cachedItem.blob);

// 显示文件大小信息
const fileSizeMB = (cachedItem.blob.size / 1024 / 1024).toFixed(1);
_updateStatus(`准备识别 (${fileSizeMB}MB)...`);

srt = await AISubtitleService.transcribe(cachedItem.blob, (progress) => {
if (progress.type === 'compress') {
_updateStatus(progress.message);
} else if (progress.type === 'compress_done') {
_updateStatus(progress.message);
} else if (progress.type === 'upload') {
_updateStatus(progress.message);
}
});

// 保存字幕到缓存
_updateStatus('正在保存字幕到缓存...');
Expand Down