Hamdy005 commited on
Commit
1716986
·
1 Parent(s): c62d436

feat: redesign chat interface with unified input container and update Dockerfile and ASR routes

Browse files
Files changed (2) hide show
  1. Dockerfile +6 -0
  2. asr/routes.py +42 -0
Dockerfile CHANGED
@@ -2,6 +2,12 @@ FROM python:3.12-slim
2
 
3
  WORKDIR /app
4
 
 
 
 
 
 
 
5
  COPY requirements.txt .
6
  RUN pip install --no-cache-dir --extra-index-url https://download.pytorch.org/whl/cpu -r requirements.txt
7
 
 
2
 
3
  WORKDIR /app
4
 
5
+ # Install system dependencies for audio processing (soundfile & ffmpeg)
6
+ RUN apt-get update && apt-get install -y --no-install-recommends \
7
+ ffmpeg \
8
+ libsndfile1 \
9
+ && rm -rf /var/lib/apt/lists/*
10
+
11
  COPY requirements.txt .
12
  RUN pip install --no-cache-dir --extra-index-url https://download.pytorch.org/whl/cpu -r requirements.txt
13
 
asr/routes.py CHANGED
@@ -20,6 +20,45 @@ _ALLOWED_MIME_PREFIXES = ("audio/", "video/webm") # webm is video/* but contain
20
  _ASR_TIMEOUT_S = 60 # max seconds to wait for a transcription result
21
 
22
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
23
  @router.post("/transcribe")
24
  async def transcribe_audio(
25
  audio: UploadFile = File(..., description="Audio recording from the browser (.webm, .wav, .ogg)"),
@@ -59,6 +98,9 @@ async def transcribe_audio(
59
  logger.error(f"Failed to save audio upload: {e}", exc_info=True)
60
  raise HTTPException(500, "Failed to save audio file.")
61
 
 
 
 
62
  # Create job and queue it
63
  job_id = str(uuid.uuid4())
64
  job = AudioJob(job_id=job_id, audio_path=tmp_path, language=language)
 
20
  _ASR_TIMEOUT_S = 60 # max seconds to wait for a transcription result
21
 
22
 
23
+ def convert_to_wav_16k(input_path: str) -> str:
24
+ """
25
+ Convert an input audio file (e.g. .webm, .ogg, .mp3) to a standard 16kHz mono WAV file
26
+ using ffmpeg. If ffmpeg is not available or fails, returns the original path.
27
+ """
28
+ import subprocess
29
+ import tempfile
30
+ import os
31
+
32
+ fd, output_path = tempfile.mkstemp(suffix=".wav")
33
+ os.close(fd)
34
+
35
+ cmd = [
36
+ "ffmpeg",
37
+ "-y",
38
+ "-i", input_path,
39
+ "-ar", "16000",
40
+ "-ac", "1",
41
+ "-f", "wav",
42
+ output_path
43
+ ]
44
+
45
+ try:
46
+ subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, check=True)
47
+ # Delete original temporary file if conversion succeeded
48
+ try:
49
+ os.unlink(input_path)
50
+ except Exception:
51
+ pass
52
+ return output_path
53
+ except Exception as e:
54
+ logger.warning(f"ffmpeg conversion failed: {e}. Falling back to original file.")
55
+ try:
56
+ os.unlink(output_path)
57
+ except Exception:
58
+ pass
59
+ return input_path
60
+
61
+
62
  @router.post("/transcribe")
63
  async def transcribe_audio(
64
  audio: UploadFile = File(..., description="Audio recording from the browser (.webm, .wav, .ogg)"),
 
98
  logger.error(f"Failed to save audio upload: {e}", exc_info=True)
99
  raise HTTPException(500, "Failed to save audio file.")
100
 
101
+ # Convert to standard 16kHz mono WAV format using ffmpeg
102
+ tmp_path = convert_to_wav_16k(tmp_path)
103
+
104
  # Create job and queue it
105
  job_id = str(uuid.uuid4())
106
  job = AudioJob(job_id=job_id, audio_path=tmp_path, language=language)