Files
alc-english/pimsleur-study/transcribe.mjs

199 lines
7.4 KiB
JavaScript

/**
* Transcribe Pimsleur units via Soniox async STT, preserving the long pauses
* (the "your turn" gaps that define the Pimsleur method).
*
* Reuses the Soniox flow from reactor-module-tools/module-convert/transcribe-soniox.js
* but with Spanish+English hints and pause annotation, for studying the tape format.
*
* Usage:
* node transcribe.mjs "<file.mp3>" ["<file2.mp3>" ...]
* node transcribe.mjs --level I --units 1,2,15 # convenience for Spanish <level>
*
* Output (in ./transcripts/, mirroring the source name):
* <name>.txt — readable transcript: [mm:ss] lines + "(pause Ns)" markers
* <name>.json — raw Soniox tokens + assembled words (for deeper analysis)
*
* Env: SONIOX_API_KEY (loaded from reactor-module-tools/.env).
*/
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const PIMSLEUR_ROOT = "/home/j/projects/Pimsleur Spanish";
const OUT_DIR = path.join(__dirname, "transcripts");
// Load the Soniox key from reactor-module-tools/.env (no dotenv dep needed).
function loadKey() {
if (process.env.SONIOX_API_KEY) return process.env.SONIOX_API_KEY;
const envPath = "/home/j/projects/reactor-module-tools/.env";
try {
for (const line of fs.readFileSync(envPath, "utf8").split("\n")) {
const m = /^\s*SONIOX_API_KEY\s*=\s*(.+?)\s*$/.exec(line);
if (m) return m[1].replace(/^["']|["']$/g, "");
}
} catch {}
return null;
}
const API = "https://api.soniox.com";
const MODEL = "stt-async-v4";
const POLL_INTERVAL_MS = 4000;
const POLL_MAX = 300; // ~20 min ceiling per unit (units are ~30 min audio)
const GAP_NEWLINE_S = 1.2; // new utterance line
const GAP_PAUSE_S = 2.5; // annotate as a "your turn" pause
const authHeaders = (key) => ({ Authorization: `Bearer ${key}` });
async function sx(key, urlPath, opts = {}) {
const resp = await fetch(`${API}${urlPath}`, { ...opts, headers: { ...authHeaders(key), ...(opts.headers || {}) } });
const text = await resp.text();
let json;
try { json = text ? JSON.parse(text) : {}; } catch { json = { _raw: text }; }
if (!resp.ok) throw new Error(`${urlPath} -> HTTP ${resp.status}: ${text.slice(0, 300)}`);
return json;
}
async function uploadFile(key, filePath) {
const buf = fs.readFileSync(filePath);
const form = new FormData();
form.append("file", new Blob([buf]), path.basename(filePath));
const json = await sx(key, "/v1/files", { method: "POST", body: form });
return json.id;
}
async function createTranscription(key, fileId) {
const json = await sx(key, "/v1/transcriptions", {
method: "POST",
headers: { "Content-Type": "application/json" },
// Spanish + English: Pimsleur interleaves an English narrator with Spanish.
body: JSON.stringify({ model: MODEL, file_id: fileId, language_hints: ["es", "en"] }),
});
return json.id;
}
async function waitFor(key, id) {
for (let i = 0; i < POLL_MAX; i++) {
await new Promise((r) => setTimeout(r, POLL_INTERVAL_MS));
const j = await sx(key, `/v1/transcriptions/${id}`);
if (j.status === "completed") return;
if (j.status === "error") throw new Error(`soniox: ${j.error_type || ""} ${j.error_message || ""}`.trim());
}
throw new Error("soniox: timed out");
}
const getTokens = async (key, id) => (await sx(key, `/v1/transcriptions/${id}/transcript`)).tokens || [];
async function cleanup(key, fileId, tid) {
for (const p of [`/v1/transcriptions/${tid}`, `/v1/files/${fileId}`]) {
if (!p.endsWith("/undefined")) { try { await sx(key, p, { method: "DELETE" }); } catch {} }
}
}
// Soniox returns subword tokens (leading space = new word) with per-token
// language. Reassemble into words, carrying language + timing (seconds).
function tokensToWords(tokens) {
const words = [];
for (const t of tokens) {
const tx = t.text ?? "";
if (!tx.trim()) continue;
const startsWord = tx.startsWith(" ") || words.length === 0;
if (startsWord) {
words.push({ raw: tx.trim(), start: t.start_ms / 1000, end: t.end_ms / 1000, lang: t.language || null });
} else {
const w = words[words.length - 1];
w.raw += tx;
w.end = t.end_ms / 1000;
}
}
return words;
}
const mmss = (s) => `${String(Math.floor(s / 60)).padStart(2, "0")}:${String(Math.floor(s % 60)).padStart(2, "0")}`;
// Group words into utterances (gap-split), tagging language and inter-utterance
// pauses so the prompt -> (your turn) -> confirmation rhythm is visible.
function render(words) {
const lines = [];
let cur = [];
const flush = () => {
if (!cur.length) return;
const text = cur.map((w) => w.raw).join(" ");
const langs = [...new Set(cur.map((w) => w.lang).filter(Boolean))];
const tag = langs.length === 1 ? langs[0].toUpperCase() : langs.length ? "ES/EN" : "??";
lines.push(`[${mmss(cur[0].start)}] (${tag}) ${text}`);
cur = [];
};
for (let i = 0; i < words.length; i++) {
cur.push(words[i]);
const next = words[i + 1];
const gap = next ? next.start - words[i].end : 0;
if (!next || gap > GAP_NEWLINE_S) {
const endAt = words[i].end;
flush();
if (next && gap > GAP_PAUSE_S) lines.push(` … (pause ${gap.toFixed(1)}s) …`);
}
}
return lines.join("\n") + "\n";
}
async function transcribe(key, mp3Path) {
const rel = path.relative(PIMSLEUR_ROOT, mp3Path);
const safe = rel.replace(/[\\/]/g, "__").replace(/\.mp3$/i, "");
const outTxt = path.join(OUT_DIR, `${safe}.txt`);
const outJson = path.join(OUT_DIR, `${safe}.json`);
if (fs.existsSync(outTxt)) { console.log(` skip (done): ${rel}`); return; }
let fileId, tid;
const t0 = Date.now();
try {
fileId = await uploadFile(key, mp3Path);
tid = await createTranscription(key, fileId);
await waitFor(key, tid);
const tokens = await getTokens(key, tid);
const words = tokensToWords(tokens);
fs.mkdirSync(OUT_DIR, { recursive: true });
fs.writeFileSync(outTxt, render(words));
fs.writeFileSync(outJson, JSON.stringify({ source: rel, model: MODEL, words }, null, 1));
console.log(` ${rel}: ${words.length} words [${((Date.now() - t0) / 1000).toFixed(0)}s] -> ${path.basename(outTxt)}`);
} finally {
if (fileId) await cleanup(key, fileId, tid);
}
}
function resolveArgs() {
const args = process.argv.slice(2);
const li = args.indexOf("--level");
if (li !== -1) {
const level = args[li + 1];
const ui = args.indexOf("--units");
const dir = path.join(PIMSLEUR_ROOT, `Pimsleur - Spanish ${level}`);
let files = fs.readdirSync(dir).filter((f) => /Unit \d+\.mp3$/i.test(f)).sort();
if (ui !== -1 && args[ui + 1]) {
const want = new Set(args[ui + 1].split(",").map((n) => n.trim().padStart(2, "0")));
files = files.filter((f) => want.has(f.match(/Unit (\d+)/i)[1].padStart(2, "0")));
}
return files.map((f) => path.join(dir, f));
}
return args.filter((a) => a.endsWith(".mp3"));
}
async function main() {
const key = loadKey();
if (!key) { console.error("SONIOX_API_KEY not found"); process.exit(1); }
const files = resolveArgs();
if (!files.length) {
console.error('Usage: node transcribe.mjs "<file.mp3>" ... | --level I --units 1,2,15');
process.exit(1);
}
console.log(`Transcribing ${files.length} file(s) via Soniox ${MODEL}`);
for (const f of files) {
try { await transcribe(key, f); }
catch (e) { console.error(` FAILED ${f}: ${e.message}`); }
}
console.log("Done.");
}
main().catch((e) => { console.error(e); process.exit(1); });