Compare commits

..
9 Commits
Author SHA1 Message Date
Dave Horton e8b2009d29 1.0.17 2026-08-10 16:45:32 -04:00
Dave HortonandClaude Opus 5 27e07bb551 fix(tts): forward gradium json_config to the streaming path (#157)
synthGradium destructured json_config from options but only forwarded
pronunciation_id into the say: param block, so the setting could never
reach mediajam's streaming dialect — only the cache-render POST. The
say: parser is brace-aware, so a nested json object survives intact.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-10 16:45:07 -04:00
Dave Horton 33456b93d9 1.0.16 2026-08-08 23:07:43 -04:00
Dave HortonandClaude Opus 5 70e91b5057 test(inworld): gate the say: param test on INWORLD_API_KEY (#156)
The test I added in #155 ran unconditionally, which broke `npm test` without
credentials — the path husky's pre-commit hook takes, so `npm version patch`
could not commit.

Two causes, both addressed:

- no credential gate, unlike every other vendor test in this file. Now skips
  without INWORLD_API_KEY, and closes its redis client on that path so the
  run can still exit.
- it assumed streaming was enabled. The Google non-streaming test sets
  JAMBONES_DISABLE_TTS_STREAMING and, on its no-credentials skip path,
  deletes the env var WITHOUT clearing the require cache (unlike its finally
  block, which clears both) — so lib/config still held 'true' further down
  the file and synthInworld took the non-streaming branch, attempting a real
  vendor call. The test now re-requires with streaming enabled so it does not
  depend on what ran before it.

Verified both ways: skips and exits 0 with no key; 11/11 with a key even
under the leaked state. Re-introducing the #155 bug still fails 3 assertions,
so the regression value is intact.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-08 23:06:52 -04:00
Dave HortonandClaude Opus 5 c066840f85 fix(inworld): read pitch and speakingRate from audioConfig in the say: params (#155)
The streaming say: path guarded on opts.audioConfig?.pitch and
opts.audioConfig?.speakingRate but interpolated opts.pitch and
opts.speakingRate, which are undefined — so anyone setting them under
audioConfig (what the docs and the portal defaults tell you to do) got
'pitch=undefined,speakingRate=undefined' on the wire and their setting
silently dropped.

Adds a test for the say: params that needs no credentials, since the
streaming branch builds the path without calling the vendor.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-08 22:49:50 -04:00
Dave Horton d89457f67d 1.0.15 2026-08-06 10:23:47 -04:00
Dave HortonandClaude Opus 5 3c15976669 fix(deps): remove junk "23" dependency (#154)
"23": "^0.0.0" is not a real dependency - it is an empty placeholder
package (0.0.0, no deps, unrelated third-party maintainer) that landed
here from a stray npm install. Nothing in the package references it.

Beyond the noise, it is a small supply-chain liability: a dependency on
a squatted single-number name owned by nobody we know, shipped to every
consumer of speech-utils.

Lint passes; full test suite passes 105/105 with live vendor credentials.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-06 10:23:31 -04:00
Dave Horton 70abb04941 1.0.14 2026-08-06 10:13:51 -04:00
Dave HortonandClaude Opus 5 724af7fcab fix(deps): drop unused undici dependency (#153)
undici was declared as a direct dependency but is never required
anywhere in this package - grep across index.js, lib/ and stubs/ finds
no reference to undici, ProxyAgent, setGlobalDispatcher or Dispatcher.
The one HTTP call in lib/synth-audio.js uses the global fetch, and
Azure proxy support goes through the SDK's own setProxy plus the
http_proxy_ip/http_proxy_port params.

Removing it clears all seven open undici advisories from npm audit for
this package and its consumers, and stops speech-utils pulling a
duplicate undici 7.x into trees where the consumer already depends on
undici 8 (feature-server does).

Lint passes; full test suite passes 105/105 with live vendor credentials.

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
2026-08-06 10:13:31 -04:00
4 changed files with 75 additions and 36 deletions
+7 -2
View File
@@ -941,8 +941,9 @@ const synthInworld = async(logger, {
params += `,voice=${voice}`;
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (opts.temperature) params += `,temperature=${opts.temperature}`;
if (opts.audioConfig?.pitch) params += `,pitch=${opts.pitch}`;
if (opts.audioConfig?.speakingRate) params += `,speakingRate=${opts.speakingRate}`;
/* pitch and speakingRate are nested under audioConfig, matching Inworld's API */
if (opts.audioConfig?.pitch) params += `,pitch=${opts.audioConfig.pitch}`;
if (opts.audioConfig?.speakingRate) params += `,speakingRate=${opts.audioConfig.speakingRate}`;
params += '}';
return {
@@ -1462,6 +1463,10 @@ const synthGradium = async(logger, {
params += `,write_cache_file=${disableTtsCache ? 0 : 1}`;
if (model_id) params += `,model_id=${model_id}`;
if (pronunciation_id) params += `,pronunciation_id=${pronunciation_id}`;
/* the say: param parser is brace-aware, so a nested json object survives intact */
if (json_config) {
params += `,json_config=${typeof json_config === 'string' ? json_config : JSON.stringify(json_config)}`;
}
params += '}';
return {
+3 -30
View File
@@ -1,15 +1,14 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.13",
"version": "1.0.17",
"lockfileVersion": 2,
"requires": true,
"packages": {
"": {
"name": "@jambonz/speech-utils",
"version": "1.0.13",
"version": "1.0.17",
"license": "MIT",
"dependencies": {
"23": "^0.0.0",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@cartesia/cartesia-js": "^2.2.7",
@@ -20,8 +19,7 @@
"debug": "^4.3.4",
"google-protobuf": "^3.21.2",
"microsoft-cognitiveservices-speech-sdk": "^1.51.0",
"openai": "^4.98.0",
"undici": "^7.5.0"
"openai": "^4.98.0"
},
"devDependencies": {
"config": "^4.2.0",
@@ -2445,12 +2443,6 @@
"node": ">=22.0.0"
}
},
"node_modules/23": {
"version": "0.0.0",
"resolved": "https://registry.npmjs.org/23/-/23-0.0.0.tgz",
"integrity": "sha512-uAETf9Okr72trtp1pNXYKhFCTTI1EKGcYMA8gw3jLGhlbaDX+grrNToEWrpt8luxRAvrZWBTvyB5wk3PpNjGQQ==",
"license": "ISC"
},
"node_modules/abort-controller": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/abort-controller/-/abort-controller-3.0.0.tgz",
@@ -7066,15 +7058,6 @@
"url": "https://github.com/sponsors/ljharb"
}
},
"node_modules/undici": {
"version": "7.24.5",
"resolved": "https://registry.npmjs.org/undici/-/undici-7.24.5.tgz",
"integrity": "sha512-3IWdCpjgxp15CbJnsi/Y9TCDE7HWVN19j1hmzVhoAkY/+CJx449tVxT5wZc1Gwg8J+P0LWvzlBzxYRnHJ+1i7Q==",
"license": "MIT",
"engines": {
"node": ">=20.18.1"
}
},
"node_modules/undici-types": {
"version": "5.26.5",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-5.26.5.tgz",
@@ -7377,11 +7360,6 @@
}
},
"dependencies": {
"23": {
"version": "0.0.0",
"resolved": "https://registry.npmjs.org/23/-/23-0.0.0.tgz",
"integrity": "sha512-uAETf9Okr72trtp1pNXYKhFCTTI1EKGcYMA8gw3jLGhlbaDX+grrNToEWrpt8luxRAvrZWBTvyB5wk3PpNjGQQ=="
},
"@aashutoshrathi/word-wrap": {
"version": "1.2.6",
"resolved": "https://registry.npmjs.org/@aashutoshrathi/word-wrap/-/word-wrap-1.2.6.tgz",
@@ -12415,11 +12393,6 @@
"which-boxed-primitive": "^1.0.2"
}
},
"undici": {
"version": "7.24.5",
"resolved": "https://registry.npmjs.org/undici/-/undici-7.24.5.tgz",
"integrity": "sha512-3IWdCpjgxp15CbJnsi/Y9TCDE7HWVN19j1hmzVhoAkY/+CJx449tVxT5wZc1Gwg8J+P0LWvzlBzxYRnHJ+1i7Q=="
},
"undici-types": {
"version": "5.26.5",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-5.26.5.tgz",
+2 -4
View File
@@ -1,6 +1,6 @@
{
"name": "@jambonz/speech-utils",
"version": "1.0.13",
"version": "1.0.17",
"description": "TTS-related speech utilities for jambonz",
"main": "index.js",
"author": "Dave Horton",
@@ -25,7 +25,6 @@
},
"homepage": "https://github.com/jambonz/speech-utils#readme",
"dependencies": {
"23": "^0.0.0",
"@aws-sdk/client-polly": "^3.496.0",
"@aws-sdk/client-sts": "^3.496.0",
"@cartesia/cartesia-js": "^2.2.7",
@@ -36,8 +35,7 @@
"debug": "^4.3.4",
"google-protobuf": "^3.21.2",
"microsoft-cognitiveservices-speech-sdk": "^1.51.0",
"openai": "^4.98.0",
"undici": "^7.5.0"
"openai": "^4.98.0"
},
"devDependencies": {
"config": "^4.2.0",
+63
View File
@@ -1147,6 +1147,69 @@ test('inworld speech synth', async(t) => {
client.quit();
});
test('inworld streaming say: params', async(t) => {
/* This test asserts the streaming say: path, so it must run with streaming
enabled. The Google non-streaming test above sets
JAMBONES_DISABLE_TTS_STREAMING and, on its no-credentials skip path,
deletes the env var WITHOUT clearing the require cache — so lib/config can
still be holding 'true' by the time we get here. Re-require to be
independent of what ran before us.
*/
delete process.env.JAMBONES_DISABLE_TTS_STREAMING;
delete require.cache[require.resolve('../lib/config')];
delete require.cache[require.resolve('../lib/synth-audio')];
delete require.cache[require.resolve('..')];
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);
if (!process.env.INWORLD_API_KEY) {
t.pass('skipping inworld streaming say: param tests since INWORLD_API_KEY is not provided');
client.quit();
return t.end();
}
try {
let result = await synthAudio(stats, {
vendor: 'inworld',
credentials: {api_key: process.env.INWORLD_API_KEY, model_id: 'inworld-tts-1.5-mini'},
language: 'en',
voice: 'Ashley',
text: 'This is a test of inworld streaming.',
options: {temperature: 0.9, audioConfig: {pitch: 2.5, speakingRate: 1.2}},
disableTtsCache: true
});
t.ok(result.filePath.startsWith('say:'), 'inworld returns streaming say: path');
t.ok(result.filePath.includes('vendor=inworld'), 'streaming path contains vendor=inworld');
t.ok(result.filePath.includes('voice=Ashley'), 'streaming path contains voice');
t.ok(result.filePath.includes('model_id=inworld-tts-1.5-mini'), 'streaming path contains model_id');
t.ok(result.filePath.includes('temperature=0.9'), 'streaming path contains temperature');
/* pitch and speakingRate are nested under audioConfig; they used to be read
from the top level and emitted as "undefined"
*/
t.ok(result.filePath.includes('speakingRate=1.2'), 'audioConfig.speakingRate reaches the say: params');
t.ok(result.filePath.includes('pitch=2.5'), 'audioConfig.pitch reaches the say: params');
t.ok(!result.filePath.includes('undefined'), 'no undefined values in the say: params');
/* options omitted entirely: no stray keys */
result = await synthAudio(stats, {
vendor: 'inworld',
credentials: {api_key: process.env.INWORLD_API_KEY, model_id: 'inworld-tts-1.5-mini'},
language: 'en',
voice: 'Ashley',
text: 'This is a test of inworld streaming.',
disableTtsCache: true
});
t.ok(!result.filePath.includes('speakingRate='), 'speakingRate omitted when unset');
t.ok(!result.filePath.includes('pitch='), 'pitch omitted when unset');
t.ok(!result.filePath.includes('undefined'), 'no undefined values when options are omitted');
} catch (err) {
console.error(JSON.stringify(err));
t.end(err);
}
client.quit();
});
test('resemble speech synth', async(t) => {
const fn = require('..');
const {synthAudio, client} = fn(opts, logger);