gpu self hosted setup guide (no-mistakes)

2025-12-21 12:49:06 +00:00 · 2025-12-09 11:25:09 -05:00
parent 5779478d3c
commit 2b3f28993f
14 changed files with 799 additions and 26 deletions
--- a/gpu/self_hosted/app/services/transcriber.py
+++ b/gpu/self_hosted/app/services/transcriber.py
@@ -129,6 +129,11 @@ class WhisperService:
            audio = np.frombuffer(proc.stdout, dtype=np.float32)
            return audio

+        # IMPORTANT: This VAD segment logic is duplicated in multiple files for deployment isolation.
+        # If you modify this function, you MUST update all copies:
+        #   - gpu/modal_deployments/reflector_transcriber.py
+        #   - gpu/modal_deployments/reflector_transcriber_parakeet.py
+        #   - gpu/self_hosted/app/services/transcriber.py (this file)
        def vad_segments(
            audio_array,
            sample_rate: int = SAMPLE_RATE,
@@ -153,6 +158,10 @@ class WhisperService:
                    end = speech["end"]
                    yield (start / float(SAMPLE_RATE), end / float(SAMPLE_RATE))
                    start = None
+            # Handle case where audio ends while speech is still active
+            if start is not None:
+                audio_duration = len(audio_array) / float(sample_rate)
+                yield (start / float(SAMPLE_RATE), audio_duration)
            iterator.reset_states()

        audio_array = load_audio_via_ffmpeg(file_path, SAMPLE_RATE)
--- a/gpu/self_hosted/app/utils.py
+++ b/gpu/self_hosted/app/utils.py
@@ -34,6 +34,12 @@ def ensure_dirs():
    UPLOADS_PATH.mkdir(parents=True, exist_ok=True)


+# IMPORTANT: This function is duplicated in multiple files for deployment isolation.
+# If you modify the audio format detection logic, you MUST update all copies:
+#   - gpu/self_hosted/app/utils.py (this file)
+#   - gpu/modal_deployments/reflector_transcriber.py (2 copies)
+#   - gpu/modal_deployments/reflector_transcriber_parakeet.py
+#   - gpu/modal_deployments/reflector_diarizer.py
 def detect_audio_format(url: str, headers: Mapping[str, str]) -> str:
    url_path = urlparse(url).path
    for ext in SUPPORTED_FILE_EXTENSIONS:
@@ -47,6 +53,8 @@ def detect_audio_format(url: str, headers: Mapping[str, str]) -> str:
        return "wav"
    if "audio/mp4" in content_type:
        return "mp4"
+    if "audio/webm" in content_type or "video/webm" in content_type:
+        return "webm"

    raise HTTPException(
        status_code=400,