Compare commits

...
3 Commits
Author SHA1 Message Date
Dave Horton d176a644fe 1.0.4 2026-06-06 10:44:11 +02:00
Hoan Luu Huu 644f2918dc support cartesia sonic3.5 (#143) 2026-06-06 10:42:40 +02:00
Dave HortonandClaude Opus 4.5 4f430b9785 update publish workflow to use actions v4
Fixes npm warning about deprecated always-auth config by updating
setup-node from v3 to v4.

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2026-06-03 08:51:56 -04:00
4 changed files with 48 additions and 10 deletions
+2 -2
View File
@@ -12,8 +12,8 @@ jobs:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- uses: actions/setup-node@v3
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: lts/*
registry-url: 'https://registry.npmjs.org'
+43 -5
View File
@@ -1343,6 +1343,18 @@ const synthCartesia = async(logger, {
try {
const client = new CartesiaClient({ apiKey: api_key });
const sampleRate = 48000;
// omit a control unless explicitly provided (0 is a valid value, so test for nullish only)
const has = (v) => v !== null && v !== undefined;
/* Voice controls are model-family specific:
- sonic-2 takes `experimentalControls` (emotion is an array, no volume).
- sonic-3 family (sonic-3, sonic-3.5, pinned sonic-3.x snapshots) takes
`generationConfig` (emotion is a string, volume supported). Match the same
"starts with sonic-3" predicate the freeswitch streaming module uses
(mod_cartesia_tts_streaming: strncmp(model_id, "sonic-3", ...)), so cached
and streamed synthesis behave identically.
Older models (sonic, sonic-english, sonic-multilingual, sonic-2024-*) take neither. */
const isSonic3 = model_id?.startsWith('sonic-3');
const mp3Stream = await client.tts.bytes({
modelId: model_id,
transcript: text,
@@ -1358,16 +1370,16 @@ const synthCartesia = async(logger, {
),
...(model_id === 'sonic-2' && (opts.speed || opts.emotion) && {
experimentalControls: {
...(opts.speed !== null && opts.speed !== undefined && {speed: opts.speed}),
...(has(opts.speed) && {speed: opts.speed}),
...(opts.emotion && {emotion: [opts.emotion]}),
}
}),
},
...(model_id === 'sonic-3' && (opts.speed || opts.emotion || opts.volume) && {
...(isSonic3 && (has(opts.speed) || has(opts.emotion) || has(opts.volume)) && {
generationConfig: {
...(opts.volume !== null && opts.volume !== undefined && {volume: opts.volume}),
...(opts.speed !== null && opts.speed !== undefined && {speed: opts.speed}),
...(opts.emotion !== null && opts.emotion !== undefined && {emotion: opts.emotion}),
...(has(opts.volume) && {volume: opts.volume}),
...(has(opts.speed) && {speed: opts.speed}),
...(has(opts.emotion) && {emotion: opts.emotion}),
}
}),
language: language,
@@ -1391,6 +1403,32 @@ const synthCartesia = async(logger, {
sampleRate
};
} catch (err) {
/* Cartesia's tts.bytes() uses a streaming response, so on an HTTP error the SDK
throws a CartesiaError whose `body` is an unconsumed stream-wrapper object
(async-iterable, yielding Uint8Array chunks) rather than the parsed error
JSON. Read it so callers get a meaningful message instead of a serialized
stream object. */
if (err && err.body && typeof err.body !== 'string' && typeof err.body[Symbol.asyncIterator] === 'function') {
try {
const chunks = [];
for await (const chunk of err.body) {
chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk));
}
const text = Buffer.concat(chunks).toString('utf8');
if (text) {
let parsed;
try {
parsed = JSON.parse(text);
} catch {
parsed = null;
}
err.message = (parsed && (parsed.error || parsed.message)) || text;
err.body = parsed || text;
}
} catch (readErr) {
logger.info({readErr}, 'synth Cartesia: failed to read error response body');
}
}
logger.info({err}, 'synth Cartesia returned error');
stats.increment('tts.count', ['vendor:cartesia', 'accepted:no']);
throw err;
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.3",
"version": "1.0.4",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "1.0.3",
"version": "1.0.4",
"license": "MIT",
"dependencies": {
"23": "^0.0.0",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.3",
"version": "1.0.4",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",