Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
94 changes: 92 additions & 2 deletions lib/synth-audio.js
Original file line number Diff line number Diff line change
Expand Up @@ -80,8 +80,8 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
logger = logger || noopLogger;

assert.ok(['google', 'aws', 'polly', 'microsoft', 'wellsaid', 'nvidia', 'elevenlabs',
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'nineninesix', 'inworld', 'resemble',
'murf', 'xai', 'fishaudio']
'whisper', 'deepgram', 'deepgramflux', 'rimelabs', 'cartesia', 'gradium', 'kugelaudio', 'nineninesix', 'inworld',
'resemble', 'murf', 'xai', 'fishaudio']
.includes(vendor) ||
vendor.startsWith('custom'),
`synthAudio supported vendors are google, aws, microsoft, nvidia and wellsaid ..etc, not ${vendor}`);
Expand Down Expand Up @@ -139,6 +139,9 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
} else if ('gradium' === vendor) {
assert.ok(voice, 'synthAudio requires voice when gradium is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when gradium is used');
} else if ('kugelaudio' === vendor) {
assert.ok(voice, 'synthAudio requires voice when kugelaudio is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when kugelaudio is used');
} else if ('nineninesix' === vendor) {
assert.ok(voice, 'synthAudio requires voice when nineninesix is used');
assert.ok(credentials.api_key, 'synthAudio requires api_key when nineninesix is used');
Expand Down Expand Up @@ -234,6 +237,11 @@ async function synthAudio(client, createHash, retrieveHash, logger, stats, { acc
credentials, options, stats, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'kugelaudio':
audioData = await synthKugelaudio(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache});
break;
case 'nineninesix':
audioData = await synthNineninesix(logger, {
credentials, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
Expand Down Expand Up @@ -1512,6 +1520,88 @@ const synthGradium = async(logger, {
}
};

/* kugelaudio — json websocket (/ws/tts/stream) for streaming, and POST /v1/tts/generate
for the cache render. the POST streams back bare little-endian 16-bit samples at the
requested sample_rate, which is exactly the r8 container at 8000.

voices are numeric ids (or public handles). language is an ISO 639-1 code that drives
text normalization; jambonz carries BCP-47, so only the primary subtag is sent. the
api rejects codes outside its list, so an unsupported or unset language is omitted
and the voice's own language applies. options.api_uri pins a region
(e.g. api.eu.kugelaudio.com).
*/
const KUGELAUDIO_LANGUAGES = ['ar', 'bg', 'bn', 'cs', 'da', 'de', 'el', 'en', 'es', 'fa', 'fi', 'fr', 'he', 'hi',
'hr', 'hu', 'id', 'it', 'ja', 'ko', 'ms', 'nl', 'no', 'pl', 'pt', 'ro', 'ru', 'sk', 'sl', 'sr', 'sv', 'ta', 'th',
'tr', 'uk', 'ur', 'vi', 'yue', 'zh'];
const synthKugelaudio = async(logger, {
credentials, options, stats, language, voice, key, text, renderForCaching, disableTtsStreaming,
disableTtsCache
}) => {
const {api_key, model_id} = credentials;
const {api_uri, speed, cfg_scale, temperature, normalize, project_id, dictionary_ids} = options || {};
const isSet = (v) => v !== null && v !== undefined;

/* default to using the streaming interface, unless disabled by env var OR we want just a cache file */
if (!JAMBONES_DISABLE_TTS_STREAMING && !renderForCaching && !disableTtsStreaming) {
let params = '{';
params += `api_key=${api_key}`;
params += `,playback_id=${key}`;
params += ',vendor=kugelaudio';
params += `,voice=${voice}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
params += `,model_id=${model_id || 'kugel-3'}`;
if (language) params += `,language=${language}`;
if (api_uri) params += `,api_uri=${api_uri}`;
if (isSet(speed)) params += `,speed=${speed}`;
if (isSet(cfg_scale)) params += `,cfg_scale=${cfg_scale}`;
if (isSet(temperature)) params += `,temperature=${temperature}`;
if (isSet(normalize)) params += `,normalize=${normalize}`;
if (isSet(project_id)) params += `,project_id=${project_id}`;
/* the say: param parser is bracket-aware, so the json array survives intact */
if (Array.isArray(dictionary_ids)) params += `,dictionary_ids=${JSON.stringify(dictionary_ids)}`;
params += '}';

return {
filePath: `say:${params}${text.replace(/\n/g, ' ').replace(/\r/g, ' ')}`,
servedFromCache: false,
rtt: 0
};
}

try {
const sampleRate = 8000;
const host = (api_uri || 'api.kugelaudio.com').replace(/^[a-z]+:\/\//, '').replace(/\/$/, '');
const post = bent(`https://${host}`, 'POST', 'buffer', {
'Authorization': `Bearer ${api_key}`,
'Content-Type': 'application/json; charset=utf-8'
});
const voiceId = /^\d+$/.test(`${voice}`) ? Number(voice) : voice;
const lang = language && language.split('-')[0].toLowerCase();
const audioContent = await post('/v1/tts/generate', {
text,
voice_id: voiceId,
model_id: model_id || 'kugel-3',
sample_rate: sampleRate,
...(KUGELAUDIO_LANGUAGES.includes(lang) && {language: lang}),
...(isSet(speed) && {speed: Number(speed)}),
...(isSet(cfg_scale) && {cfg_scale: Number(cfg_scale)}),
...(isSet(temperature) && {temperature: Number(temperature)}),
...(isSet(normalize) && {normalize: normalize === true || normalize === 'true'}),
...(isSet(project_id) && {project_id: Number(project_id)}),
...(Array.isArray(dictionary_ids) && {dictionary_ids})
});
return {
audioContent,
extension: 'r8',
sampleRate
};
} catch (err) {
logger.info({err}, 'synth kugelaudio returned error');
stats.increment('tts.count', ['vendor:kugelaudio', 'accepted:no']);
throw err;
}
};

/* nineninesix.ai — a Cartesia-compatible API, but only raw/wav come back
(mp3 is rejected), so the cache render asks for wav rather than mp3. */
/* fish.audio — msgpack websocket for streaming, and a POST endpoint for the cache
Expand Down
30 changes: 30 additions & 0 deletions test/synth.js
Original file line number Diff line number Diff line change
Expand Up @@ -1090,6 +1090,36 @@ test('gradium speech synth tests', async(t) => {
client.quit();
});

test('kugelaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);

if (!process.env.KUGELAUDIO_API_KEY) {
t.pass('skipping kugelaudio speech synth tests since KUGELAUDIO_API_KEY is not provided');
return t.end();
}
const text = 'Guten Tag und willkommen bei jambonz! Ihre Bestellung kostet 12,99 Euro. ' + Date.now();
try {
const opts = await synthAudio(stats, {
vendor: 'kugelaudio',
credentials: {
api_key: process.env.KUGELAUDIO_API_KEY,
model_id: 'kugel-3'
},
language: 'de-DE',
voice: '1930',
text,
renderForCaching: true
});
t.ok(!opts.servedFromCache, `successfully synthed kugelaudio audio to ${opts.filePath}`);

} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});

test('fishaudio speech synth tests', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
Expand Down
Loading