Skip to content

have to patch the npm package to make it work #38

Description

@salir-admin
t a/dist/tts/transcribe.js b/dist/tts/transcribe.js
index fd0ed51c1c0eb61f980559bf8e4fa73945fd148c..8e8db15dd6107fc1381a26e1b1f1aa94e0a44802 100644
--- a/dist/tts/transcribe.js
+++ b/dist/tts/transcribe.js
@@ -11,7 +11,6 @@
  * runs locally via ONNX, no cloud round-trip. The pipeline instance is
  * cached per process so repeated calls reuse the loaded model.
  */
-import { pipeline } from '@huggingface/transformers';
 import { spawnSync } from 'node:child_process';
 export const DEFAULT_WHISPER_MODEL = 'Xenova/whisper-base.en';
 /** Whisper's required audio sample rate. */
@@ -22,6 +21,10 @@ let cached = null;
 async function getTranscriber(model) {
     if (cached?.model === model)
         return cached.transcriber;
+    // Lazy import (patched for bun's isolated linker): a static top-level import made every
+    // CLI invocation — even `--help` — load transformers/onnxruntime, whose nested deps do
+    // not resolve under isolated linking. Whisper is only needed when transcription runs.
+    const { pipeline } = await import('@huggingface/transformers');
     const transcriber = (await pipeline('automatic-speech-recognition', model));
     cached = { model, transcriber };
     return transcriber;
diff --git a/dist/tts/engine.js b/dist/tts/engine.js
index 6c0901d..f2af935 100644
--- a/dist/tts/engine.js
+++ b/dist/tts/engine.js
@@ -153,10 +153,11 @@ export function parseWavHeader(wav) {
  * Convert arbitrary audio (MP3, OGG, PCM, etc.) to Argo's WAV format
  * (mono, Float32, 24kHz) using ffmpeg.
  */
-export function convertToWav(audioBuffer) {
+export function convertToWav(audioBuffer, speed = 1) {
     const { execFileSync } = childProcess;
     const result = execFileSync('ffmpeg', [
         '-i', 'pipe:0',
+        ...(speed === 1 ? [] : ['-filter:a', `atempo=${speed}`]),
         '-f', 'wav',
         '-acodec', 'pcm_f32le',
         '-ac', '1',
diff --git a/dist/tts/engine.d.ts b/dist/tts/engine.d.ts
index 21e6fbf..9b31bd7 100644
--- a/dist/tts/engine.d.ts
+++ b/dist/tts/engine.d.ts
@@ -52,7 +52,7 @@ export declare function parseWavHeader(wav: Buffer): WavInfo;
  * Convert arbitrary audio (MP3, OGG, PCM, etc.) to Argo's WAV format
  * (mono, Float32, 24kHz) using ffmpeg.
  */
-export declare function convertToWav(audioBuffer: Buffer): Buffer;
+export declare function convertToWav(audioBuffer: Buffer, speed?: number): Buffer;
 /**
  * Creates a mock TTS engine that produces silent WAV buffers of the given
  * duration and records all calls for test assertions.
diff --git a/dist/tts/engines/elevenlabs.js b/dist/tts/engines/elevenlabs.js
index 4d194d1..3cc2860 100644
--- a/dist/tts/engines/elevenlabs.js
+++ b/dist/tts/engines/elevenlabs.js
@@ -49,7 +49,7 @@ export class ElevenLabsEngine {
         const mp3Buffer = Buffer.concat(chunks);
         // Convert MP3 to Argo WAV format
         const { convertToWav } = await import('../engine.js');
-        return convertToWav(mp3Buffer);
+        return convertToWav(mp3Buffer, options.speed ?? 1.0);
     }
 }
 //# sourceMappingURL=elevenlabs.js.map

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions