Spaces:
Sleeping
Sleeping
| import gradio as gr | |
| from asr import transcribe, ASR_EXAMPLES | |
| def create_interface() -> gr.Blocks: | |
| """ | |
| Create and configure the Gradio interface for ASR demo. | |
| Returns: | |
| Configured Gradio Blocks interface | |
| """ | |
| with gr.Blocks(title="Shan ASR Demo") as demo: | |
| gr.Markdown( | |
| """ | |
| # 🎙️ Shan Language Speech Recognition | |
| Choose between the original MMS model or our fine-tuned version for better accuracy. | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| # Model selection | |
| model_dropdown = gr.Dropdown( | |
| choices=["original", "finetune"], | |
| label="ASR Model", | |
| value="finetune", | |
| info="'finetune' model provides better accuracy for Shan language" | |
| ) | |
| # Audio source selection | |
| audio_source = gr.Radio( | |
| choices=["Record from Mic", "Upload audio"], | |
| label="Audio Input Method", | |
| value="Record from Mic", | |
| ) | |
| # Microphone input | |
| mic_input = gr.Audio( | |
| sources=["microphone"], | |
| type="filepath", | |
| label="Record Audio", | |
| visible=True | |
| ) | |
| # File upload input | |
| file_input = gr.Audio( | |
| sources=["upload"], | |
| type="filepath", | |
| label="Upload Audio File", | |
| visible=False | |
| ) | |
| # Submit button | |
| submit_btn = gr.Button( | |
| "Transcribe", | |
| variant="primary", | |
| size="lg" | |
| ) | |
| with gr.Column(scale=1): | |
| # Output text | |
| output_text = gr.Textbox( | |
| label="Transcription", | |
| placeholder="Transcribed text will appear here...", | |
| lines=10, | |
| max_lines=20, | |
| ) | |
| # Examples section | |
| gr.Markdown("### 📝 Try These Examples") | |
| gr.Examples( | |
| examples=ASR_EXAMPLES, | |
| inputs=[model_dropdown, audio_source, mic_input, file_input], | |
| ) | |
| # Information section | |
| with gr.Accordion("ℹ️ About This Demo", open=False): | |
| gr.Markdown( | |
| """ | |
| ### Models | |
| - **Original**: Facebook's MMS-1B model with Shan adapter | |
| - **Finetune**: Custom fine-tuned model optimized for Shan language | |
| ### Supported Audio Formats | |
| - WAV, MP3, FLAC, OGG, and other common formats | |
| - Recommended: 16kHz sample rate, mono channel | |
| ### Tips | |
| - Use a quiet environment for better accuracy | |
| - Speak clearly and at a moderate pace | |
| - The fine-tuned model generally performs better for Shan language | |
| """ | |
| ) | |
| # Event handlers | |
| def toggle_audio_inputs(source: str): | |
| """Toggle visibility of audio input components based on source.""" | |
| return ( | |
| gr.update(visible=source == "Record from Mic"), | |
| gr.update(visible=source == "Upload audio") | |
| ) | |
| audio_source.change( | |
| fn=toggle_audio_inputs, | |
| inputs=[audio_source], | |
| outputs=[mic_input, file_input], | |
| queue=False | |
| ) | |
| submit_btn.click( | |
| fn=transcribe, | |
| inputs=[model_dropdown, audio_source, mic_input, file_input], | |
| outputs=output_text, | |
| api_name="transcribe" | |
| ) | |
| return demo | |
| if __name__ == "__main__": | |
| demo = create_interface() | |
| demo.launch() | |