Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
88 changes: 87 additions & 1 deletion lib/synth-audio.js
Original file line number Diff line number Diff line change
Expand Up @@ -81,7 +81,7 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc

assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
'resemble', 'murf', 'xai', 'fishaudio']
'resemble', 'murf', 'xai', 'fishaudio', 'speechify']
.includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
Expand Down Expand Up @@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
} else if ('gradium' === vendor) {
assert.ok(voice, 'synthAudio requires voice when gradium is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
} else if ('speechify' === vendor) {
assert.ok(voice, 'synthAudio requires voice when speechify is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when speechify is used');
} else if ('kugelaudio' === vendor) {
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
Expand Down Expand Up @@ -242,6 +245,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'speechify':
audioData = await synthSpeechify(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'nineninesix':
audioData = await synthNineninesix(logger, {
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
Expand Down Expand Up @@ -1602,6 +1610,84 @@ const synthKugelaudio = async(logger, {
}
};

/* simba-3.2 is English only; with no model set, other languages get simba-3.0 */
const speechifyModel = (model_id, language) => {
if (model_id) return model_id;
return !language || /^en/i.test(language) ? 'simba-3.2' : 'simba-3.0';
};
const synthSpeechify = async(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache
}) => {
const {api_key, model_id} = credentials;
let credOptions = credentials.options || {};
if (typeof credOptions === 'string') {
try {
credOptions = JSON.parse(credOptions);
} catch {
credOptions = {};
}
}
const {api_uri, loudness_normalization, text_normalization} = {...credOptions, ...options};
const isSet = (v) => v !== null && v !== undefined;
const model = speechifyModel(options?.model_id || model_id, language);

/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=speechify';
params += `,voice=${voice}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += `,model_id=${model}`;
if (language) params += `,language=${language}`;
if (api_uri) params += `,api_uri=${api_uri}`;
if (isSet(loudness_normalization)) params += `,loudness_normalization=${loudness_normalization}`;
if (isSet(text_normalization)) params += `,text_normalization=${text_normalization}`;
params += '}';

return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}

try {
/* other pcm rates are mislabelled 24 kHz on workspaces pinned before 2026-09-30 */
const sampleRate = 24000;
const host = (api_uri || 'api.speechify.ai').replace(/^[a-z]+:\/\//, '').replace(/\/$/, '');
const post = bent(`https://${host}`, 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Content-Type': 'application/json',
'Speechify-Caller': 'jambonz'
});
const toBool = (v) => v === true || v === 'true';
const speechOptions = {
...(isSet(loudness_normalization) && {loudness_normalization: toBool(loudness_normalization)}),
...(isSet(text_normalization) && {text_normalization: toBool(text_normalization)})
};
const audioContent = await post('/v1/audio/stream', {
input: text,
voice_id: voice,
model,
output_format: `pcm_${sampleRate}`,
...(language && {language}),
...(Object.keys(speechOptions).length && {options: speechOptions})
});
return {
audioContent,
extension: 'r24',
sampleRate
};
} catch (err) {
logger.info({err}, 'synth speechify returned error');
stats.increment('tts.count', ['vendor:speechify', 'accepted:no']);
throw err;
}
};

/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
Expand Down
29 changes: 29 additions & 0 deletions test/synth.js
Original file line number Diff line number Diff line change
Expand Up @@ -1120,6 +1120,35 @@ test('kugelaudio speech synth tests', async(t) => {
client.quit();
});

test('speechify speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);

if (!process.env.SPEECHIFY_API_KEY) {
t.pass('skipping speechify speech synth tests since SPEECHIFY_API_KEY is not provided');
return t.end();
}
const text = 'Hi there and welcome to jambonz! ' + Date.now();
try {
const opts = await synthAudio(stats, {
vendor: 'speechify',
credentials: {
api_key: process.env.SPEECHIFY_API_KEY
},
language: 'en-US',
voice: 'geffen_32',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthed speechify audio to ${opts.filePath}`);

} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});

test('fishaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
Expand Down
Loading