Tonemark is a SvelteKit PWA for transcribing YouTube videos, audio and video files, and microphone recordings using a local Whisper backend. Features: - Dark glassmorphic UI with electric-lime accent (5 switchable themes) - Rail nav (desktop) / tab bar (mobile) layout - Drop zone, YouTube URL input, and live audio recording inputs - Audio mode waveform cards (none / standard / aggressive / auto) - Real-time transcription progress with animated waveform - Job queue with SSE streaming updates - Push notifications on job completion - PWA with native SvelteKit service worker - SRT / TXT / MD / JSON transcript downloads Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,162 @@
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import { existsSync } from 'fs';
|
||||
import { mkdir, unlink, writeFile } from 'fs/promises';
|
||||
import { join } from 'path';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
const TMP_DIR = join(process.env.DATA_DIR ?? '/tmp/.whisper-pwa', 'downloads');
|
||||
|
||||
export async function ensureTmpDir() {
|
||||
if (!existsSync(TMP_DIR)) await mkdir(TMP_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
export interface CaptionResult {
|
||||
type: 'captions';
|
||||
segments: Array<{ index: number; start: number; end: number; text: string; words: [] }>;
|
||||
title: string;
|
||||
}
|
||||
|
||||
export interface AudioResult {
|
||||
type: 'audio';
|
||||
audioPath: string;
|
||||
title: string;
|
||||
}
|
||||
|
||||
export type DownloadResult = CaptionResult | AudioResult;
|
||||
|
||||
/** Try to get auto-generated captions from YouTube. Returns null if unavailable. */
|
||||
async function tryGetCaptions(url: string, outDir: string): Promise<CaptionResult | null> {
|
||||
const jsonPath = join(outDir, 'info.json');
|
||||
try {
|
||||
await execFileAsync('yt-dlp', [
|
||||
'--write-auto-subs',
|
||||
'--sub-langs', 'en.*',
|
||||
'--skip-download',
|
||||
'--write-info-json',
|
||||
'--no-playlist',
|
||||
'-o', join(outDir, '%(title)s.%(ext)s'),
|
||||
url
|
||||
]);
|
||||
|
||||
// Find the VTT/SRT file
|
||||
const { readdirSync } = await import('fs');
|
||||
const files = readdirSync(outDir);
|
||||
const vttFile = files.find((f) => f.endsWith('.vtt') || f.endsWith('.srt'));
|
||||
if (!vttFile) return null;
|
||||
|
||||
let title = 'Untitled';
|
||||
if (existsSync(jsonPath)) {
|
||||
try {
|
||||
const info = JSON.parse((await import('fs')).readFileSync(jsonPath, 'utf8'));
|
||||
title = info.title ?? title;
|
||||
} catch { /* ignore */ }
|
||||
}
|
||||
|
||||
const content = (await import('fs')).readFileSync(join(outDir, vttFile), 'utf8');
|
||||
const segments = parseVtt(content);
|
||||
if (segments.length === 0) return null;
|
||||
|
||||
return { type: 'captions', segments, title };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Download best audio from YouTube. Returns path to audio file. */
|
||||
async function downloadAudio(url: string, outDir: string): Promise<{ audioPath: string; title: string }> {
|
||||
await execFileAsync('yt-dlp', [
|
||||
'-f', 'bestaudio',
|
||||
'--write-info-json',
|
||||
'--no-playlist',
|
||||
'-o', join(outDir, 'audio.%(ext)s'),
|
||||
url
|
||||
]);
|
||||
|
||||
const { readdirSync, readFileSync } = await import('fs');
|
||||
const files = readdirSync(outDir);
|
||||
const audioFile = files.find((f) => f.startsWith('audio.') && !f.endsWith('.json'));
|
||||
if (!audioFile) throw new Error('yt-dlp did not produce an audio file');
|
||||
|
||||
let title = 'Untitled';
|
||||
const jsonFile = files.find((f) => f.endsWith('.info.json'));
|
||||
if (jsonFile) {
|
||||
try {
|
||||
title = JSON.parse(readFileSync(join(outDir, jsonFile), 'utf8')).title ?? title;
|
||||
} catch { /* ignore */ }
|
||||
}
|
||||
|
||||
return { audioPath: join(outDir, audioFile), title };
|
||||
}
|
||||
|
||||
/** Download a YouTube URL: try captions first, fall back to audio. */
|
||||
export async function downloadYouTube(url: string, jobId: string): Promise<DownloadResult> {
|
||||
await ensureTmpDir();
|
||||
const outDir = join(TMP_DIR, jobId);
|
||||
await mkdir(outDir, { recursive: true });
|
||||
|
||||
const captions = await tryGetCaptions(url, outDir);
|
||||
if (captions) return captions;
|
||||
|
||||
const { audioPath, title } = await downloadAudio(url, outDir);
|
||||
return { type: 'audio', audioPath, title };
|
||||
}
|
||||
|
||||
/** Save an uploaded file to tmp dir and return its path. */
|
||||
export async function saveUploadedFile(
|
||||
buffer: Buffer,
|
||||
filename: string,
|
||||
jobId: string
|
||||
): Promise<string> {
|
||||
await ensureTmpDir();
|
||||
const outDir = join(TMP_DIR, jobId);
|
||||
await mkdir(outDir, { recursive: true });
|
||||
const dest = join(outDir, filename);
|
||||
await writeFile(dest, buffer);
|
||||
return dest;
|
||||
}
|
||||
|
||||
export async function cleanupJobTmp(jobId: string) {
|
||||
const outDir = join(TMP_DIR, jobId);
|
||||
try {
|
||||
const { rm } = await import('fs/promises');
|
||||
await rm(outDir, { recursive: true, force: true });
|
||||
} catch { /* ignore */ }
|
||||
}
|
||||
|
||||
/** Parse a WebVTT string into segments. */
|
||||
function parseVtt(
|
||||
content: string
|
||||
): Array<{ index: number; start: number; end: number; text: string; words: [] }> {
|
||||
const segments: Array<{ index: number; start: number; end: number; text: string; words: [] }> = [];
|
||||
const blocks = content.split(/\n\n+/);
|
||||
let index = 0;
|
||||
|
||||
for (const block of blocks) {
|
||||
const lines = block.trim().split('\n');
|
||||
const timeLine = lines.find((l) => l.includes('-->'));
|
||||
if (!timeLine) continue;
|
||||
|
||||
const [startStr, endStr] = timeLine.split('-->').map((s) => s.trim().split(' ')[0]);
|
||||
const start = vttTimeToSec(startStr);
|
||||
const end = vttTimeToSec(endStr);
|
||||
const text = lines
|
||||
.filter((l) => !l.includes('-->') && !/^\d+$/.test(l.trim()) && l.trim())
|
||||
.join(' ')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
.trim();
|
||||
|
||||
if (text) {
|
||||
segments.push({ index: index++, start, end, text, words: [] });
|
||||
}
|
||||
}
|
||||
|
||||
return segments;
|
||||
}
|
||||
|
||||
function vttTimeToSec(t: string): number {
|
||||
const parts = t.split(':').map(Number);
|
||||
if (parts.length === 3) return parts[0] * 3600 + parts[1] * 60 + parts[2];
|
||||
if (parts.length === 2) return parts[0] * 60 + parts[1];
|
||||
return parts[0];
|
||||
}
|
||||
Reference in New Issue
Block a user