refactor(transcript): drop Tonemark rewrite
Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import { existsSync } from 'fs';
|
||||
import { mkdir, unlink, writeFile } from 'fs/promises';
|
||||
import { mkdir, writeFile } from 'fs/promises';
|
||||
import { join } from 'path';
|
||||
import { fetchTranscript, type TranscriptResponse } from 'youtube-transcript';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
const TMP_DIR = join(process.env.DATA_DIR ?? '/tmp/.whisper-pwa', 'downloads');
|
||||
@@ -26,43 +27,33 @@ export interface AudioResult {
|
||||
export type DownloadResult = CaptionResult | AudioResult;
|
||||
|
||||
/** Try to get auto-generated captions from YouTube. Returns null if unavailable. */
|
||||
async function tryGetCaptions(url: string, outDir: string): Promise<CaptionResult | null> {
|
||||
const jsonPath = join(outDir, 'info.json');
|
||||
async function tryGetCaptions(url: string, _outDir: string): Promise<CaptionResult | null> {
|
||||
try {
|
||||
await execFileAsync('yt-dlp', [
|
||||
'--write-auto-subs',
|
||||
'--sub-langs', 'en.*',
|
||||
'--skip-download',
|
||||
'--write-info-json',
|
||||
'--no-playlist',
|
||||
'-o', join(outDir, '%(title)s.%(ext)s'),
|
||||
url
|
||||
]);
|
||||
|
||||
// Find the VTT/SRT file
|
||||
const { readdirSync } = await import('fs');
|
||||
const files = readdirSync(outDir);
|
||||
const vttFile = files.find((f) => f.endsWith('.vtt') || f.endsWith('.srt'));
|
||||
if (!vttFile) return null;
|
||||
|
||||
let title = 'Untitled';
|
||||
if (existsSync(jsonPath)) {
|
||||
try {
|
||||
const info = JSON.parse((await import('fs')).readFileSync(jsonPath, 'utf8'));
|
||||
title = info.title ?? title;
|
||||
} catch { /* ignore */ }
|
||||
}
|
||||
|
||||
const content = (await import('fs')).readFileSync(join(outDir, vttFile), 'utf8');
|
||||
const segments = parseVtt(content);
|
||||
const transcript = await fetchTranscript(url, { lang: 'en' });
|
||||
const segments = transcriptEntriesToSegments(transcript);
|
||||
if (segments.length === 0) return null;
|
||||
|
||||
const title = await getYouTubeTitle(url);
|
||||
return { type: 'captions', segments, title };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function getYouTubeTitle(url: string): Promise<string> {
|
||||
try {
|
||||
const { stdout } = await execFileAsync('yt-dlp', [
|
||||
'--dump-single-json',
|
||||
'--skip-download',
|
||||
'--no-playlist',
|
||||
url
|
||||
]);
|
||||
return JSON.parse(stdout).title ?? 'Untitled';
|
||||
} catch {
|
||||
return 'Untitled';
|
||||
}
|
||||
}
|
||||
|
||||
/** Download best audio from YouTube. Returns path to audio file. */
|
||||
async function downloadAudio(url: string, outDir: string): Promise<{ audioPath: string; title: string }> {
|
||||
await execFileAsync('yt-dlp', [
|
||||
@@ -124,39 +115,22 @@ export async function cleanupJobTmp(jobId: string) {
|
||||
} catch { /* ignore */ }
|
||||
}
|
||||
|
||||
/** Parse a WebVTT string into segments. */
|
||||
function parseVtt(
|
||||
content: string
|
||||
export function transcriptEntriesToSegments(
|
||||
entries: TranscriptResponse[]
|
||||
): Array<{ index: number; start: number; end: number; text: string; words: [] }> {
|
||||
const segments: Array<{ index: number; start: number; end: number; text: string; words: [] }> = [];
|
||||
const blocks = content.split(/\n\n+/);
|
||||
let index = 0;
|
||||
|
||||
for (const block of blocks) {
|
||||
const lines = block.trim().split('\n');
|
||||
const timeLine = lines.find((l) => l.includes('-->'));
|
||||
if (!timeLine) continue;
|
||||
|
||||
const [startStr, endStr] = timeLine.split('-->').map((s) => s.trim().split(' ')[0]);
|
||||
const start = vttTimeToSec(startStr);
|
||||
const end = vttTimeToSec(endStr);
|
||||
const text = lines
|
||||
.filter((l) => !l.includes('-->') && !/^\d+$/.test(l.trim()) && l.trim())
|
||||
.join(' ')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
.trim();
|
||||
|
||||
if (text) {
|
||||
segments.push({ index: index++, start, end, text, words: [] });
|
||||
}
|
||||
}
|
||||
|
||||
return segments;
|
||||
}
|
||||
|
||||
function vttTimeToSec(t: string): number {
|
||||
const parts = t.split(':').map(Number);
|
||||
if (parts.length === 3) return parts[0] * 3600 + parts[1] * 60 + parts[2];
|
||||
if (parts.length === 2) return parts[0] * 60 + parts[1];
|
||||
return parts[0];
|
||||
const useMilliseconds = entries.some((entry) => entry.offset > 1000 || entry.duration > 1000);
|
||||
return entries
|
||||
.map((entry) => {
|
||||
const start = useMilliseconds ? entry.offset / 1000 : entry.offset;
|
||||
const duration = useMilliseconds ? entry.duration / 1000 : entry.duration;
|
||||
return {
|
||||
index: 0,
|
||||
start,
|
||||
end: start + duration,
|
||||
text: entry.text.trim(),
|
||||
words: [] as []
|
||||
};
|
||||
})
|
||||
.filter((entry) => entry.text.length > 0)
|
||||
.map((entry, index) => ({ ...entry, index }));
|
||||
}
|
||||
|
||||
@@ -96,15 +96,13 @@ async function runJob(
|
||||
|
||||
if (captionSegments) {
|
||||
// Caption fast path — skip whisper
|
||||
const { deduplicateSegments } = await import('./postprocess.js');
|
||||
const { writeOutputs } = await import('./formatter.js');
|
||||
const segments = deduplicateSegments(captionSegments);
|
||||
const paths = await writeOutputs(segments, title, jobId);
|
||||
const paths = await writeOutputs(captionSegments, title, jobId);
|
||||
updateJob({
|
||||
id: jobId,
|
||||
status: 'done',
|
||||
progress: 100,
|
||||
segmentsJson: JSON.stringify(segments),
|
||||
segmentsJson: JSON.stringify(captionSegments),
|
||||
outputDir: paths.srt.replace(/\/[^/]+$/, '')
|
||||
});
|
||||
emitProgress(jobId, { type: 'done' });
|
||||
|
||||
@@ -1,235 +0,0 @@
|
||||
import type { Segment } from '$lib/types.js';
|
||||
|
||||
// ── Collapse consecutive repeated phrases within a segment's text ────────────
|
||||
|
||||
function collapseRepeats(text: string): string {
|
||||
let prev = '';
|
||||
// Keep applying until stable
|
||||
while (true) {
|
||||
const next = collapseOnce(text);
|
||||
if (next === prev || next === text) return next;
|
||||
prev = text;
|
||||
text = next;
|
||||
}
|
||||
}
|
||||
|
||||
function collapseOnce(text: string): string {
|
||||
// Match any repeated phrase (2+ words) appearing consecutively
|
||||
return text.replace(/\b(.{10,}?)\s+\1\b/gi, '$1');
|
||||
}
|
||||
|
||||
// ── Merge consecutive segments with identical (or near-identical) text ───────
|
||||
|
||||
function normalise(s: string) {
|
||||
return s.toLowerCase().replace(/[^\w\s]/g, '').replace(/\s+/g, ' ').trim();
|
||||
}
|
||||
|
||||
function mergeConsecutive(segments: Segment[]): Segment[] {
|
||||
const out: Segment[] = [];
|
||||
for (const seg of segments) {
|
||||
const last = out[out.length - 1];
|
||||
if (last && normalise(last.text) === normalise(seg.text)) {
|
||||
last.end = seg.end;
|
||||
} else {
|
||||
out.push({ ...seg });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ── Collapse rolling prefix/suffix chains from backend segment hypotheses ──────
|
||||
|
||||
const MAX_CHAIN_GAP_SECS = 0.15;
|
||||
const MIN_MEANINGFUL_WORDS = 2;
|
||||
const MIN_MEANINGFUL_CHARS = 8;
|
||||
const MIN_OVERLAP_WORDS = 1;
|
||||
|
||||
function splitWords(text: string): string[] {
|
||||
return text.trim().split(/\s+/).filter(Boolean);
|
||||
}
|
||||
|
||||
function normaliseWords(text: string): string[] {
|
||||
return splitWords(text)
|
||||
.map((word) => word.toLowerCase().replace(/[^\w]/g, ''))
|
||||
.filter(Boolean);
|
||||
}
|
||||
|
||||
function arraysEqual(a: string[], b: string[]): boolean {
|
||||
return a.length === b.length && a.every((value, index) => value === b[index]);
|
||||
}
|
||||
|
||||
function startsWithWords(full: string[], prefix: string[]): boolean {
|
||||
return prefix.length <= full.length && arraysEqual(full.slice(0, prefix.length), prefix);
|
||||
}
|
||||
|
||||
function endsWithWords(full: string[], suffix: string[]): boolean {
|
||||
return suffix.length <= full.length && arraysEqual(full.slice(full.length - suffix.length), suffix);
|
||||
}
|
||||
|
||||
function suffixPrefixOverlap(left: string[], right: string[]): number {
|
||||
const max = Math.min(left.length, right.length);
|
||||
for (let size = max; size >= 1; size--) {
|
||||
if (arraysEqual(left.slice(left.length - size), right.slice(0, size))) return size;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
function isMeaningfulPhrase(words: string[]): boolean {
|
||||
return words.length >= MIN_MEANINGFUL_WORDS && words.join(' ').length >= MIN_MEANINGFUL_CHARS;
|
||||
}
|
||||
|
||||
function isShortCarryover(seg: Segment, words: string[]): boolean {
|
||||
return seg.end - seg.start <= 0.2 || words.length <= 2 || words.join(' ').length <= 16;
|
||||
}
|
||||
|
||||
function trimLeadingWords(text: string, count: number): string {
|
||||
return splitWords(text).slice(count).join(' ').trim();
|
||||
}
|
||||
|
||||
function collapseIncrementalSegments(segments: Segment[]): Segment[] {
|
||||
const out: Segment[] = [];
|
||||
|
||||
for (const seg of segments) {
|
||||
let current: Segment = {
|
||||
...seg,
|
||||
text: seg.text.trim()
|
||||
};
|
||||
|
||||
if (!current.text) continue;
|
||||
|
||||
const last = out[out.length - 1];
|
||||
if (!last) {
|
||||
out.push(current);
|
||||
continue;
|
||||
}
|
||||
|
||||
const gap = current.start - last.end;
|
||||
if (gap > MAX_CHAIN_GAP_SECS) {
|
||||
out.push(current);
|
||||
continue;
|
||||
}
|
||||
|
||||
const lastWords = normaliseWords(last.text);
|
||||
const currentWords = normaliseWords(current.text);
|
||||
if (lastWords.length === 0 || currentWords.length === 0) {
|
||||
out.push(current);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (
|
||||
currentWords.length > lastWords.length &&
|
||||
startsWithWords(currentWords, lastWords) &&
|
||||
(isMeaningfulPhrase(lastWords) || isShortCarryover(last, lastWords))
|
||||
) {
|
||||
last.text = current.text;
|
||||
last.end = current.end;
|
||||
last.words = current.words;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (
|
||||
endsWithWords(lastWords, currentWords) &&
|
||||
(isMeaningfulPhrase(currentWords) || isShortCarryover(current, currentWords))
|
||||
) {
|
||||
last.end = Math.max(last.end, current.end);
|
||||
continue;
|
||||
}
|
||||
|
||||
const overlapWords = suffixPrefixOverlap(lastWords, currentWords);
|
||||
if (overlapWords >= MIN_OVERLAP_WORDS) {
|
||||
const trimmedText = trimLeadingWords(current.text, overlapWords);
|
||||
if (!trimmedText) {
|
||||
last.end = Math.max(last.end, current.end);
|
||||
continue;
|
||||
}
|
||||
|
||||
current = {
|
||||
...current,
|
||||
start: Math.max(current.start, last.end),
|
||||
text: trimmedText,
|
||||
words: []
|
||||
};
|
||||
}
|
||||
|
||||
out.push(current);
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
// ── N-gram deduplication ─────────────────────────────────────────────────────
|
||||
|
||||
const NGRAM_N = 6;
|
||||
const LOOKBACK_CHARS = 500;
|
||||
const SIMILARITY_THRESHOLD = 0.6;
|
||||
|
||||
function ngrams(text: string, n: number): string[] {
|
||||
const words = text.toLowerCase().split(/\s+/);
|
||||
const grams: string[] = [];
|
||||
for (let i = 0; i <= words.length - n; i++) {
|
||||
grams.push(words.slice(i, i + n).join(' '));
|
||||
}
|
||||
return grams;
|
||||
}
|
||||
|
||||
function jaccardSimilarity(a: string, b: string): number {
|
||||
const ga = new Set(ngrams(a, NGRAM_N));
|
||||
const gb = new Set(ngrams(b, NGRAM_N));
|
||||
// If neither text is long enough to produce n-grams they cannot be compared;
|
||||
// treat as dissimilar so short segments are never incorrectly discarded.
|
||||
if (ga.size === 0 && gb.size === 0) return 0;
|
||||
const intersection = [...ga].filter((g) => gb.has(g)).length;
|
||||
const union = new Set([...ga, ...gb]).size;
|
||||
return union === 0 ? 0 : intersection / union;
|
||||
}
|
||||
|
||||
function ngramDedup(segments: Segment[]): Segment[] {
|
||||
const out: Segment[] = [];
|
||||
for (const seg of segments) {
|
||||
const windowText = out
|
||||
.slice(-20)
|
||||
.map((s) => s.text)
|
||||
.join(' ')
|
||||
.slice(-LOOKBACK_CHARS);
|
||||
|
||||
if (windowText.length > 0 && jaccardSimilarity(seg.text, windowText) >= SIMILARITY_THRESHOLD) {
|
||||
continue; // duplicate — skip
|
||||
}
|
||||
out.push(seg);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ── Full deduplication pipeline ──────────────────────────────────────────────
|
||||
|
||||
export function deduplicateSegments(segments: Segment[]): Segment[] {
|
||||
if (!Array.isArray(segments)) return [];
|
||||
// 1. Collapse repeats within each segment's text
|
||||
let result = segments.map((s) => ({
|
||||
...s,
|
||||
text: collapseRepeats(s.text.trim())
|
||||
}));
|
||||
|
||||
// 2. Remove empty segments
|
||||
result = result.filter((s) => s.text.length > 0);
|
||||
|
||||
// 3. Collapse rolling backend hypotheses before generic dedup
|
||||
result = collapseIncrementalSegments(result);
|
||||
|
||||
// 4. First merge pass
|
||||
result = mergeConsecutive(result);
|
||||
|
||||
// 5. N-gram dedup
|
||||
result = ngramDedup(result);
|
||||
|
||||
// 6. Re-run rolling collapse after removals create new adjacencies
|
||||
result = collapseIncrementalSegments(result);
|
||||
|
||||
// 7. Second merge pass (catches new adjacencies after dedup)
|
||||
result = mergeConsecutive(result);
|
||||
|
||||
// 8. Re-index
|
||||
result.forEach((s, i) => (s.index = i));
|
||||
|
||||
return result;
|
||||
}
|
||||
Reference in New Issue
Block a user