Spaces:
Running
Running
feat: redesign chat interface with unified input container and update Dockerfile and ASR routes
Browse files- Dockerfile +6 -0
- asr/routes.py +42 -0
Dockerfile
CHANGED
|
@@ -2,6 +2,12 @@ FROM python:3.12-slim
|
|
| 2 |
|
| 3 |
WORKDIR /app
|
| 4 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5 |
COPY requirements.txt .
|
| 6 |
RUN pip install --no-cache-dir --extra-index-url https://download.pytorch.org/whl/cpu -r requirements.txt
|
| 7 |
|
|
|
|
| 2 |
|
| 3 |
WORKDIR /app
|
| 4 |
|
| 5 |
+
# Install system dependencies for audio processing (soundfile & ffmpeg)
|
| 6 |
+
RUN apt-get update && apt-get install -y --no-install-recommends \
|
| 7 |
+
ffmpeg \
|
| 8 |
+
libsndfile1 \
|
| 9 |
+
&& rm -rf /var/lib/apt/lists/*
|
| 10 |
+
|
| 11 |
COPY requirements.txt .
|
| 12 |
RUN pip install --no-cache-dir --extra-index-url https://download.pytorch.org/whl/cpu -r requirements.txt
|
| 13 |
|
asr/routes.py
CHANGED
|
@@ -20,6 +20,45 @@ _ALLOWED_MIME_PREFIXES = ("audio/", "video/webm") # webm is video/* but contain
|
|
| 20 |
_ASR_TIMEOUT_S = 60 # max seconds to wait for a transcription result
|
| 21 |
|
| 22 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
@router.post("/transcribe")
|
| 24 |
async def transcribe_audio(
|
| 25 |
audio: UploadFile = File(..., description="Audio recording from the browser (.webm, .wav, .ogg)"),
|
|
@@ -59,6 +98,9 @@ async def transcribe_audio(
|
|
| 59 |
logger.error(f"Failed to save audio upload: {e}", exc_info=True)
|
| 60 |
raise HTTPException(500, "Failed to save audio file.")
|
| 61 |
|
|
|
|
|
|
|
|
|
|
| 62 |
# Create job and queue it
|
| 63 |
job_id = str(uuid.uuid4())
|
| 64 |
job = AudioJob(job_id=job_id, audio_path=tmp_path, language=language)
|
|
|
|
| 20 |
_ASR_TIMEOUT_S = 60 # max seconds to wait for a transcription result
|
| 21 |
|
| 22 |
|
| 23 |
+
def convert_to_wav_16k(input_path: str) -> str:
|
| 24 |
+
"""
|
| 25 |
+
Convert an input audio file (e.g. .webm, .ogg, .mp3) to a standard 16kHz mono WAV file
|
| 26 |
+
using ffmpeg. If ffmpeg is not available or fails, returns the original path.
|
| 27 |
+
"""
|
| 28 |
+
import subprocess
|
| 29 |
+
import tempfile
|
| 30 |
+
import os
|
| 31 |
+
|
| 32 |
+
fd, output_path = tempfile.mkstemp(suffix=".wav")
|
| 33 |
+
os.close(fd)
|
| 34 |
+
|
| 35 |
+
cmd = [
|
| 36 |
+
"ffmpeg",
|
| 37 |
+
"-y",
|
| 38 |
+
"-i", input_path,
|
| 39 |
+
"-ar", "16000",
|
| 40 |
+
"-ac", "1",
|
| 41 |
+
"-f", "wav",
|
| 42 |
+
output_path
|
| 43 |
+
]
|
| 44 |
+
|
| 45 |
+
try:
|
| 46 |
+
subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, check=True)
|
| 47 |
+
# Delete original temporary file if conversion succeeded
|
| 48 |
+
try:
|
| 49 |
+
os.unlink(input_path)
|
| 50 |
+
except Exception:
|
| 51 |
+
pass
|
| 52 |
+
return output_path
|
| 53 |
+
except Exception as e:
|
| 54 |
+
logger.warning(f"ffmpeg conversion failed: {e}. Falling back to original file.")
|
| 55 |
+
try:
|
| 56 |
+
os.unlink(output_path)
|
| 57 |
+
except Exception:
|
| 58 |
+
pass
|
| 59 |
+
return input_path
|
| 60 |
+
|
| 61 |
+
|
| 62 |
@router.post("/transcribe")
|
| 63 |
async def transcribe_audio(
|
| 64 |
audio: UploadFile = File(..., description="Audio recording from the browser (.webm, .wav, .ogg)"),
|
|
|
|
| 98 |
logger.error(f"Failed to save audio upload: {e}", exc_info=True)
|
| 99 |
raise HTTPException(500, "Failed to save audio file.")
|
| 100 |
|
| 101 |
+
# Convert to standard 16kHz mono WAV format using ffmpeg
|
| 102 |
+
tmp_path = convert_to_wav_16k(tmp_path)
|
| 103 |
+
|
| 104 |
# Create job and queue it
|
| 105 |
job_id = str(uuid.uuid4())
|
| 106 |
job = AudioJob(job_id=job_id, audio_path=tmp_path, language=language)
|