File size: 5,841 Bytes
2c52bed
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e43f6a6
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
語音錄音應用程式
部署在 Hugging Face Spaces 上,從麥克風錄音並保存為 16kHz WAV 文件

Usage:
    python app.py
"""

import gradio as gr
import numpy as np
import wave
import os
from datetime import datetime
from pathlib import Path

# 設定
SAMPLE_RATE = 16000  # 16kHz
OUTPUT_DIR = "recordings"  # 錄音保存目錄

# 確保輸出目錄存在
os.makedirs(OUTPUT_DIR, exist_ok=True)


def save_audio(audio_data: np.ndarray, sr: int) -> str:
    """
    將音頻數據保存為 WAV 文件
    
    Args:
        audio_data: 音頻數據 (numpy array)
        sr: 採樣率
        
    Returns:
        保存的文件路徑
    """
    # 生成文件名
    timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
    filename = f"recording_{timestamp}.wav"
    filepath = os.path.join(OUTPUT_DIR, filename)
    
    # 確保是單聲道
    if len(audio_data.shape) > 1:
        audio_data = audio_data.mean(axis=1)
    
    # 歸一化到 16-bit 範圍
    audio_int16 = (audio_data * 32767).astype(np.int16)
    
    # 保存為 WAV
    with wave.open(filepath, 'w') as wav_file:
        wav_file.setnchannels(1)  # 單聲道
        wav_file.setsampwidth(2)  # 2 bytes (16-bit)
        wav_file.setframerate(sr)
        wav_file.writeframes(audio_int16.tobytes())
    
    return filepath


def process_recording(audio_tuple):
    """
    處理錄音並保存
    
    Args:
        audio_tuple: (sampling_rate, audio_data) 元組
        
    Returns:
        保存的文件路徑和狀態訊息
    """
    if audio_tuple is None:
        return None, "❌ 沒有錄音數據"
    
    sr, audio_data = audio_tuple
    
    # 如果採樣率不是 16kHz,需要重採樣
    if sr != SAMPLE_RATE:
        # 使用 librosa 重採樣
        try:
            import librosa
            audio_data = librosa.resample(audio_data, orig_sr=sr, target_sr=SAMPLE_RATE)
            sr = SAMPLE_RATE
        except ImportError:
            # 如果沒有 librosa,使用簡單的降採樣
            ratio = SAMPLE_RATE / sr
            new_length = int(len(audio_data) * ratio)
            audio_data = np.interp(
                np.linspace(0, len(audio_data), new_length),
                np.arange(len(audio_data)),
                audio_data
            )
            sr = SAMPLE_RATE
    
    # 保存文件
    filepath = save_audio(audio_data, sr)
    
    # 計算錄音長度
    duration = len(audio_data) / sr
    
    return filepath, f"✅ 已保存:{filepath}\n📊 長度:{duration:.2f}秒 | 採樣率:{sr}Hz"


def create_demo():
    """創建 Gradio 介面"""
    
    with gr.Blocks(title="語音錄音器") as demo:
        gr.Markdown("""
        # 🎤 語音錄音器
        
        按下「開始錄音」按鈕開始錄音,再次按下「停止錄音」按鈕停止並保存。
        錄音將保存為 16kHz 採樣率的 WAV 文件。
        """)
        
        with gr.Row():
            with gr.Column(scale=1):
                gr.Markdown("### 📝 錄音說明")
                gr.Markdown("""
                1. 點擊 **🔴 開始錄音** 按鈕
                2. 對著麥克風說話
                3. 點擊 **⏹️ 停止錄音** 按鈕
                4. 錄音將自動保存為 WAV 文件
                """)
                
                gr.Markdown("### ⚙️ 設定")
                gr.Markdown(f"""
                - 採樣率:**{SAMPLE_RATE} Hz**
                - 格式:**WAV**
                - 聲道:**單聲道**
                - 位元深度:**16-bit**
                """)
            
            with gr.Column(scale=2):
                # 錄音組件
                audio_input = gr.Audio(
                    label="🎤 錄音",
                    sources=["microphone"],
                    type="numpy",
                    waveform_options=gr.WaveformOptions(
                        sample_rate=SAMPLE_RATE,
                    ),
                )
                
                # 狀態顯示
                status_text = gr.Textbox(
                    label="📋 狀態",
                    placeholder="等待錄音...",
                    interactive=False,
                )
                
                # 保存的文件顯示
                audio_output = gr.Audio(
                    label="📁 已保存的錄音",
                    type="filepath",
                    interactive=False,
                )
                
                # 按鈕
                with gr.Row():
                    save_btn = gr.Button(
                        "💾 保存錄音",
                        variant="primary",
                        size="lg",
                    )
                    clear_btn = gr.Button(
                        "🗑️ 清除",
                        variant="secondary",
                        size="lg",
                    )
        
        # 錄音說明
        gr.Markdown("""
        ---
        ### 💡 使用提示
        
        - **錄音時**:請確保瀏覽器已授權麥克風權限
        - **停止錄音**:錄音完成後點擊「保存錄音」按鈕
        - **文件位置**:錄音保存在 `recordings/` 目錄下
        - **下載**:點擊音頻播放器右側的下載按鈕可下載文件
        """)
        
        # 事件處理
        save_btn.click(
            fn=process_recording,
            inputs=[audio_input],
            outputs=[audio_output, status_text],
        )
        
        clear_btn.click(
            fn=lambda: (None, "🗑️ 已清除"),
            inputs=[],
            outputs=[audio_input, status_text],
        )
    
    return demo


if __name__ == "__main__":
    # 創建並啟動應用
    demo = create_demo()
    demo.launch()