-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtts.js
More file actions
129 lines (112 loc) · 5.1 KB
/
Copy pathtts.js
File metadata and controls
129 lines (112 loc) · 5.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
/* Speech synthesis.
*
* Two shapes of engine, and the difference matters:
* · Browser TTS speaks straight out of the OS. The page never sees the
* samples, so it can drive a live preview but can NOT be recorded.
* · Everything else returns an audio Blob, which we can decode, analyse and
* mux into an exported video.
* Cloud calls go through the local Python server because none of these APIs
* send CORS headers a browser will accept. */
(function (global) {
'use strict';
const PROXY = '/api';
/* ── browser (offline, preview only) ─────────────────────────────── */
function listVoices() {
return new Promise((resolve) => {
const voices = speechSynthesis.getVoices();
if (voices.length) return resolve(voices);
// Chrome populates the list asynchronously on first call.
speechSynthesis.addEventListener('voiceschanged', () => resolve(speechSynthesis.getVoices()), { once: true });
setTimeout(() => resolve(speechSynthesis.getVoices()), 1000);
});
}
function speakBrowser(text, { voiceURI, rate = 1, pitch = 1, onBoundary, onEnd, onStart } = {}) {
speechSynthesis.cancel();
const utter = new SpeechSynthesisUtterance(text);
const voice = speechSynthesis.getVoices().find((v) => v.voiceURI === voiceURI);
if (voice) utter.voice = voice;
utter.rate = rate;
utter.pitch = pitch;
utter.onstart = () => onStart && onStart();
utter.onboundary = (e) => onBoundary && onBoundary(e.charIndex, e.elapsedTime);
utter.onend = () => onEnd && onEnd();
utter.onerror = (e) => {
// 'interrupted'/'canceled' are what Stop looks like — not failures.
if (e.error !== 'interrupted' && e.error !== 'canceled') console.warn('TTS error:', e.error);
onEnd && onEnd();
};
speechSynthesis.speak(utter);
return { cancel: () => speechSynthesis.cancel() };
}
/* Browser TTS gives no duration up front, so estimate one to drive the
* viseme clock. ~2.9 syllables/sec at rate 1 is close for most voices. */
function estimateDuration(text, rate = 1) {
const syllables = String(text)
.toLowerCase()
.split(/\s+/)
.filter(Boolean)
.reduce((n, w) => n + Math.max(1, (w.match(/[aeiouy]+/g) || []).length), 0);
return Math.max(0.6, (syllables / 2.9) / rate);
}
/* ── blob-returning engines ──────────────────────────────────────── */
async function synthesize(engine, text, opts = {}) {
const signal = opts.signal; // lets Stop cancel a synthesis already in flight
const routes = {
elevenlabs: () => post(`${PROXY}/tts/elevenlabs`, { text, apiKey: opts.apiKey, voiceId: opts.voiceId }, signal),
openai: () => post(`${PROXY}/tts/openai`, { text, apiKey: opts.apiKey, voice: opts.voice }, signal),
gemini: () => post(`${PROXY}/tts/gemini`, { text, apiKey: opts.apiKey, voice: opts.voice }, signal),
piper: () => post(`${PROXY}/tts/piper`, { text, baseUrl: opts.baseUrl, voice: opts.voice }, signal),
};
const call = routes[engine];
if (!call) throw new Error(`Unknown TTS engine: ${engine}`);
return call();
}
/* Instant voice cloning. The clip can be a video — the proxy strips the audio
* track with ffmpeg before anything is sent onward. Returns a voice ID usable
* as the ElevenLabs voice immediately. */
async function cloneVoice({ file, name, apiKey }) {
let res;
try {
res = await fetch(`${PROXY}/voice/clone`, {
method: 'POST',
headers: {
'Content-Type': file.type || 'application/octet-stream',
'X-Api-Key': apiKey,
'X-Voice-Name': name,
},
body: file,
});
} catch {
throw new Error('Cannot reach the local server. Voice cloning needs `python python/app.py`.');
}
const text = await res.text();
let json = {};
try { json = JSON.parse(text); } catch { /* fall through to the status check */ }
if (!res.ok) {
const msg = json.detail?.message || json.detail || json.message || text.slice(0, 200) || res.statusText;
throw new Error(typeof msg === 'string' ? msg : JSON.stringify(msg));
}
if (!json.voice_id) throw new Error('ElevenLabs accepted the clip but returned no voice_id.');
return json.voice_id;
}
async function post(url, body, signal) {
let res;
try {
res = await fetch(url, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
signal,
});
} catch (err) {
if (err.name === 'AbortError') throw err; // Stop was pressed — not a failure
throw new Error('Cannot reach the local server. Start it with `python python/app.py`.');
}
if (!res.ok) {
const detail = await res.text().catch(() => '');
throw new Error(`${res.status} ${res.statusText}${detail ? ` — ${detail.slice(0, 200)}` : ''}`);
}
return res.blob();
}
global.TTS = { listVoices, speakBrowser, estimateDuration, synthesize, cloneVoice };
})(window);