Repository navigation
Expand file tree
/
Copy pathcmd_audio_processor.py
More file actions
219 lines (189 loc) · 10.8 KB
/
Copy pathcmd_audio_processor.py
File metadata and controls
219 lines (189 loc) · 10.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
命令行版本的音频处理助手
"""
import os
import sys
import time
import argparse
import traceback
from audio_processor import (
VolumeEnhancer, SpeechToText, NoiseReducer,
VoiceExtractor, AudioRestorer,
NUMPY_AVAILABLE, LIBROSA_AVAILABLE, WHISPER_AVAILABLE,
TORCH_AVAILABLE, NOISEREDUCE_AVAILABLE
)
def main():
parser = argparse.ArgumentParser(description='命令行音频处理工具')
parser.add_argument('input_file', help='输入音频文件路径')
parser.add_argument('--output', '-o', help='输出文件路径 (默认为原文件名+后缀)')
parser.add_argument('--operation', '-p', choices=[
'enhance', 'transcribe', 'reduce_noise', 'extract_voice', 'restore'
], required=True, help='要执行的操作')
# 操作特定参数
parser.add_argument('--gain', type=float, default=1.5, help='音量增强增益值 (默认: 1.5)')
parser.add_argument('--language', choices=['中文', '英语', '自动检测'], default='自动检测',
help='语音转文字的语言 (默认: 自动检测)')
parser.add_argument('--strength', choices=['低', '中', '高'], default='中',
help='降噪或修复强度 (默认: 中)')
parser.add_argument('--level', type=int, choices=[1, 2, 3, 4, 5], default=3,
help='人声提取的分离程度 (默认: 3)')
parser.add_argument('--intensity', choices=['轻度', '中度', '深度'], default='中度',
help='音频修复强度 (默认: 中度)')
parser.add_argument('--debug', action='store_true', help='启用调试模式,显示完整错误信息')
args = parser.parse_args()
# 检查输入文件是否存在
if not os.path.exists(args.input_file):
print(f"错误: 输入文件不存在: {args.input_file}")
return 1
# 显示可用模块和功能状态
print("模块状态:")
print(f"- NumPy: {'可用' if NUMPY_AVAILABLE else '不可用'}")
print(f"- Librosa: {'可用' if LIBROSA_AVAILABLE else '不可用'}")
print(f"- Whisper: {'可用' if WHISPER_AVAILABLE else '不可用'}")
print(f"- PyTorch: {'可用' if TORCH_AVAILABLE else '不可用'}")
print(f"- NoiseReduce: {'可用' if NOISEREDUCE_AVAILABLE else '不可用'}")
print()
# 设置输出文件路径
filename, ext = os.path.splitext(args.input_file)
# 执行请求的操作
try:
if args.operation == 'enhance':
if not args.output:
args.output = f"{filename}_enhanced{ext}"
print(f"正在增强音量 (增益值: {args.gain})...")
processor = VolumeEnhancer()
processor.process(args.input_file, args.output, {'gain': args.gain})
elif args.operation == 'transcribe':
if not args.output:
args.output = f"{filename}_transcript.txt"
print(f"正在分析音频并生成报告 (语言: {args.language})...")
# 检查输出路径是否可写
try:
output_dir = os.path.dirname(args.output)
if output_dir and not os.path.exists(output_dir):
os.makedirs(output_dir)
print(f"创建输出目录: {output_dir}")
# 确保输出路径可写
with open(args.output, 'w', encoding='utf-8') as test_file:
test_file.write("测试写入权限\n")
print(f"输出文件路径可写: {args.output}")
except Exception as e:
print(f"警告: 无法写入输出文件: {str(e)}")
if args.debug:
traceback.print_exc()
# 尝试使用临时路径
old_output = args.output
args.output = os.path.join(os.getcwd(), f"transcript_{os.path.basename(filename)}.txt")
print(f"尝试使用替代输出路径: {args.output}")
# 尝试直接使用simple_audio_info模块
try:
import simple_audio_info
print(f"使用simple_audio_info模块分析音频: {args.input_file}")
result = simple_audio_info.analyze_audio(args.input_file, args.output)
if result:
print(f"音频分析完成,结果已保存至: {args.output}")
return 0
else:
print("音频分析失败,尝试使用备用方法...")
except ImportError:
print("未找到simple_audio_info模块,尝试使用内置音频处理器...")
# 如果simple_audio_info不可用,使用SpeechToText处理器
print("使用内置音频处理器...")
processor = SpeechToText()
try:
print(f"处理文件: {args.input_file} -> {args.output}")
result = processor.process(args.input_file, args.output, {'language': args.language})
print(f"处理完成,结果已保存至: {result}")
except Exception as e:
print(f"处理出错: {str(e)}")
if args.debug:
traceback.print_exc()
# 确保即使出错也生成一个报告文件
try:
# 使用直接方法生成报告
import wave
import codecs
try:
with wave.open(args.input_file, 'rb') as wav:
channels = wav.getnchannels()
sample_width = wav.getsampwidth()
frame_rate = wav.getframerate()
n_frames = wav.getnframes()
duration = n_frames / frame_rate
# 生成报告文件
with codecs.open(args.output, 'w', encoding='utf-8-sig') as f:
f.write("音频分析报告\n")
f.write("==============\n\n")
f.write(f"文件: {os.path.basename(args.input_file)}\n")
f.write(f"通道数: {channels}\n")
f.write(f"采样宽度: {sample_width * 8} 位\n")
f.write(f"采样率: {frame_rate} Hz\n")
f.write(f"总帧数: {n_frames}\n")
f.write(f"时长: {duration:.2f} 秒\n\n")
f.write("音频质量评估:\n")
if frame_rate >= 44100 and sample_width >= 2:
f.write("- 音频质量评级: 高质量 (CD级别或更高)\n")
elif frame_rate >= 22050 and sample_width >= 2:
f.write("- 音频质量评级: 中等质量 (适合语音)\n")
else:
f.write("- 音频质量评级: 低质量\n")
f.write("\n备注: 此分析仅提供音频的基本信息,不含实际的语音转文字功能。\n")
f.write("如需使用完整的语音转文字功能,请安装以下依赖:\n")
f.write("1. pip install -i https://pypi.tuna.tsinghua.edu.cn/simple openai-whisper\n")
f.write("2. 安装ffmpeg并设置系统环境变量\n")
print(f"生成了音频分析报告: {args.output}")
except Exception as wave_err:
print(f"无法使用wave模块分析音频: {str(wave_err)}")
# 使用最基本的文件信息
with codecs.open(args.output, 'w', encoding='utf-8-sig') as f:
f.write("基本文件信息\n")
f.write("============\n\n")
f.write(f"文件: {os.path.basename(args.input_file)}\n")
f.write(f"大小: {os.path.getsize(args.input_file) / 1024:.2f} KB\n")
f.write(f"修改时间: {time.ctime(os.path.getmtime(args.input_file))}\n\n")
f.write("无法分析此音频文件的详细信息。\n")
f.write("请确保文件是有效的WAV格式音频。\n\n")
f.write("如需使用语音转文字功能,请安装以下依赖:\n")
f.write("1. pip install -i https://pypi.tuna.tsinghua.edu.cn/simple openai-whisper\n")
f.write("2. 安装ffmpeg并设置系统环境变量\n")
except Exception as report_err:
print(f"生成基本报告失败: {str(report_err)}")
if args.debug:
traceback.print_exc()
elif args.operation == 'reduce_noise':
if not NOISEREDUCE_AVAILABLE:
print("错误: 缺少noisereduce模块,无法执行音频降噪功能")
return 1
if not args.output:
args.output = f"{filename}_denoised{ext}"
print(f"正在降噪 (强度: {args.strength})...")
processor = NoiseReducer()
processor.process(args.input_file, args.output, {'strength': args.strength})
elif args.operation == 'extract_voice':
if not args.output:
args.output = f"{filename}_vocals{ext}"
print(f"正在提取人声 (分离程度: {args.level})...")
processor = VoiceExtractor()
processor.process(args.input_file, args.output, {'level': args.level})
elif args.operation == 'restore':
if not args.output:
args.output = f"{filename}_restored{ext}"
print(f"正在修复音频 (强度: {args.intensity})...")
processor = AudioRestorer()
processor.process(args.input_file, args.output, {'intensity': args.intensity})
# 检查输出文件是否存在
if os.path.exists(args.output):
print(f"处理完成,结果已保存至: {args.output}")
return 0
else:
print(f"警告: 输出文件 {args.output} 未生成,处理可能失败")
return 1
except Exception as e:
print(f"处理出错: {str(e)}")
if args.debug:
traceback.print_exc()
return 1
if __name__ == "__main__":
sys.exit(main())