refactor(transcript): drop Tonemark rewrite
Build & Push Docker Image / test (push) Successful in 10s
Build & Push Docker Image / build-and-push (push) Successful in 50s

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
2026-05-12 00:10:32 +02:00
co-authored by Copilot
parent df50e74939
commit 929c482497
10 changed files with 161 additions and 540 deletions
+37 -63
View File
@@ -1,8 +1,9 @@
import { execFile } from 'child_process';
import { promisify } from 'util';
import { existsSync } from 'fs';
import { mkdir, unlink, writeFile } from 'fs/promises';
import { mkdir, writeFile } from 'fs/promises';
import { join } from 'path';
import { fetchTranscript, type TranscriptResponse } from 'youtube-transcript';
const execFileAsync = promisify(execFile);
const TMP_DIR = join(process.env.DATA_DIR ?? '/tmp/.whisper-pwa', 'downloads');
@@ -26,43 +27,33 @@ export interface AudioResult {
export type DownloadResult = CaptionResult | AudioResult;
/** Try to get auto-generated captions from YouTube. Returns null if unavailable. */
async function tryGetCaptions(url: string, outDir: string): Promise<CaptionResult | null> {
const jsonPath = join(outDir, 'info.json');
async function tryGetCaptions(url: string, _outDir: string): Promise<CaptionResult | null> {
try {
await execFileAsync('yt-dlp', [
'--write-auto-subs',
'--sub-langs', 'en.*',
'--skip-download',
'--write-info-json',
'--no-playlist',
'-o', join(outDir, '%(title)s.%(ext)s'),
url
]);
// Find the VTT/SRT file
const { readdirSync } = await import('fs');
const files = readdirSync(outDir);
const vttFile = files.find((f) => f.endsWith('.vtt') || f.endsWith('.srt'));
if (!vttFile) return null;
let title = 'Untitled';
if (existsSync(jsonPath)) {
try {
const info = JSON.parse((await import('fs')).readFileSync(jsonPath, 'utf8'));
title = info.title ?? title;
} catch { /* ignore */ }
}
const content = (await import('fs')).readFileSync(join(outDir, vttFile), 'utf8');
const segments = parseVtt(content);
const transcript = await fetchTranscript(url, { lang: 'en' });
const segments = transcriptEntriesToSegments(transcript);
if (segments.length === 0) return null;
const title = await getYouTubeTitle(url);
return { type: 'captions', segments, title };
} catch {
return null;
}
}
async function getYouTubeTitle(url: string): Promise<string> {
try {
const { stdout } = await execFileAsync('yt-dlp', [
'--dump-single-json',
'--skip-download',
'--no-playlist',
url
]);
return JSON.parse(stdout).title ?? 'Untitled';
} catch {
return 'Untitled';
}
}
/** Download best audio from YouTube. Returns path to audio file. */
async function downloadAudio(url: string, outDir: string): Promise<{ audioPath: string; title: string }> {
await execFileAsync('yt-dlp', [
@@ -124,39 +115,22 @@ export async function cleanupJobTmp(jobId: string) {
} catch { /* ignore */ }
}
/** Parse a WebVTT string into segments. */
function parseVtt(
content: string
export function transcriptEntriesToSegments(
entries: TranscriptResponse[]
): Array<{ index: number; start: number; end: number; text: string; words: [] }> {
const segments: Array<{ index: number; start: number; end: number; text: string; words: [] }> = [];
const blocks = content.split(/\n\n+/);
let index = 0;
for (const block of blocks) {
const lines = block.trim().split('\n');
const timeLine = lines.find((l) => l.includes('-->'));
if (!timeLine) continue;
const [startStr, endStr] = timeLine.split('-->').map((s) => s.trim().split(' ')[0]);
const start = vttTimeToSec(startStr);
const end = vttTimeToSec(endStr);
const text = lines
.filter((l) => !l.includes('-->') && !/^\d+$/.test(l.trim()) && l.trim())
.join(' ')
.replace(/<[^>]+>/g, '')
.trim();
if (text) {
segments.push({ index: index++, start, end, text, words: [] });
}
}
return segments;
}
function vttTimeToSec(t: string): number {
const parts = t.split(':').map(Number);
if (parts.length === 3) return parts[0] * 3600 + parts[1] * 60 + parts[2];
if (parts.length === 2) return parts[0] * 60 + parts[1];
return parts[0];
const useMilliseconds = entries.some((entry) => entry.offset > 1000 || entry.duration > 1000);
return entries
.map((entry) => {
const start = useMilliseconds ? entry.offset / 1000 : entry.offset;
const duration = useMilliseconds ? entry.duration / 1000 : entry.duration;
return {
index: 0,
start,
end: start + duration,
text: entry.text.trim(),
words: [] as []
};
})
.filter((entry) => entry.text.length > 0)
.map((entry, index) => ({ ...entry, index }));
}