r0kaxmin commited on
Commit
04045ff
·
1 Parent(s): ad0d988

Add application file

Browse files
Files changed (3) hide show
  1. Dockerfile +31 -0
  2. app.py +29 -0
  3. requirements.txt +3 -0
Dockerfile ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.10
2
+
3
+ RUN apt-get update && apt-get install -y \
4
+ git \
5
+ git-lfs \
6
+ ffmpeg \
7
+ libsm6 \
8
+ libxext6 \
9
+ cmake \
10
+ rsync \
11
+ libgl1 \
12
+ libglx-mesa0 \
13
+ && rm -rf /var/lib/apt/lists/* \
14
+ && git lfs install
15
+
16
+ WORKDIR /app
17
+
18
+ COPY requirements.txt .
19
+ RUN pip install --no-cache-dir -r requirements.txt
20
+
21
+ # Hugging Face cache => /tmp
22
+ ENV HF_HOME=/tmp/huggingface_cache
23
+ ENV HF_HUB_CACHE=/tmp/huggingface_cache
24
+ RUN mkdir -p /tmp/huggingface_cache && chmod -R 777 /tmp/huggingface_cache
25
+
26
+ # Pre-download model into /tmp (optional for faster startup in Spaces docker build)
27
+ RUN python -c "from faster_whisper import WhisperModel; WhisperModel('Systran/faster-whisper-small', device='cpu', compute_type='int8')"
28
+
29
+ COPY . .
30
+
31
+ CMD ["python", "app.py"]
app.py ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ os.environ["HF_HOME"] = "/tmp/huggingface_cache"
3
+ os.environ["HF_HUB_CACHE"] = "/tmp/huggingface_cache"
4
+
5
+ from faster_whisper import WhisperModel
6
+ import gradio as gr
7
+
8
+ model = WhisperModel("Systran/faster-whisper-small", device="cpu", compute_type="int8")
9
+
10
+ def transcribe_audio(audio_filepath, language):
11
+ if audio_filepath is None:
12
+ return "Error: No audio file provided."
13
+ lang = None if language == "auto" else language
14
+ segments, _ = model.transcribe(audio_filepath, beam_size=5, language=lang, vad_filter=True)
15
+ return " ".join(seg.text for seg in segments)
16
+
17
+ iface = gr.Interface(
18
+ fn=transcribe_audio,
19
+ inputs=[
20
+ gr.Audio(type="filepath", label="Upload Audio File"),
21
+ gr.Radio(['en', 'bn', 'auto'], label="Select Language", value='auto')
22
+ ],
23
+ outputs="text",
24
+ title="⚡ Zen Speech-to-Text",
25
+ description="Upload audio → get transcription"
26
+ )
27
+
28
+ if __name__ == "__main__":
29
+ iface.launch(server_name="0.0.0.0", server_port=int(os.getenv("PORT", "7860")))
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ gradio>=4.0.0
2
+ faster-whisper>=1.0.0
3
+ ffmpeg-python