Repository navigation
Expand file tree
/
Copy pathtranscribe.py
More file actions
400 lines (315 loc) Β· 11.4 KB
/
Copy pathtranscribe.py
File metadata and controls
400 lines (315 loc) Β· 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
#!/usr/bin/env python3
"""
Voice Memo to Notes Converter
Converts audio recordings to optimized format, transcribes using Gemini,
and generates structured markdown notes.
"""
import os
import sys
import subprocess
import argparse
import tempfile
from pathlib import Path
from datetime import datetime
import google.generativeai as genai
# ============================================================================
# CONFIGURATION
# ============================================================================
GEMINI_MODEL = "gemini-2.0-flash" # Fast and excellent for transcription
AUDIO_BITRATE = "48k" # Good quality at small size for speech
OUTPUT_FORMAT = "opus" # Excellent compression for voice
TRANSCRIPTION_PROMPT = """You are a transcription assistant. Transcribe the following audio exactly as spoken.
Instructions:
- Transcribe verbatim, capturing every word
- Preserve the natural flow including filler words (um, uh, like, you know)
- For Hindi/Hinglish parts, transliterate to Roman script (not Devanagari)
- Use proper punctuation and paragraph breaks for readability
- If multiple speakers are clearly distinguishable, label them (Speaker 1, Speaker 2, etc.)
- Preserve any emphasized words or phrases
Output the transcription directly without any preamble or commentary."""
BREAKDOWN_PROMPT = """You are an expert note-taker. Analyze the following transcript and create a comprehensive, well-structured breakdown.
Create a detailed markdown document with:
## π Summary
A 2-3 sentence executive summary of the entire recording.
## π― Key Topics
For each major topic/section discussed:
### Topic Name
- **Context**: Brief context of this section
- **Key Points**: Bullet points of main ideas
- **Details**: Important details, examples, or explanations mentioned
- **Quotes**: Any notable quotes or statements (if relevant)
## β
Action Items
- List any tasks, to-dos, or follow-ups mentioned
- Include who is responsible (if mentioned)
- Include deadlines (if mentioned)
## π‘ Key Insights
- Important insights, decisions, or conclusions
- Any "aha moments" or notable realizations
## π Additional Notes
- Any other relevant information
- References, names, or resources mentioned
---
Guidelines:
- Be thorough but concise
- Use clear, scannable formatting
- Preserve important details and nuances
- For Hindi/Hinglish content, keep it in Roman transliteration
- Group related ideas logically
- Use emoji sparingly for visual organization
Here is the transcript to analyze:
---
{transcript}
---
Create the structured breakdown:"""
# ============================================================================
# AUDIO PROCESSING
# ============================================================================
def check_ffmpeg():
"""Check if FFmpeg is installed."""
try:
subprocess.run(
["ffmpeg", "-version"],
capture_output=True,
check=True
)
return True
except (subprocess.CalledProcessError, FileNotFoundError):
return False
def get_audio_info(input_path: Path) -> dict:
"""Get audio file information using ffprobe."""
try:
result = subprocess.run(
[
"ffprobe", "-v", "quiet",
"-print_format", "json",
"-show_format", "-show_streams",
str(input_path)
],
capture_output=True,
text=True,
check=True
)
import json
return json.loads(result.stdout)
except Exception as e:
print(f"β οΈ Could not get audio info: {e}")
return {}
def compress_audio(input_path: Path, output_path: Path) -> bool:
"""Compress audio to opus format for optimal size/quality."""
print(f"π Compressing audio...")
original_size = input_path.stat().st_size / (1024 * 1024) # MB
try:
subprocess.run(
[
"ffmpeg", "-y", "-i", str(input_path),
"-vn", # No video
"-c:a", "libopus",
"-b:a", AUDIO_BITRATE,
"-ar", "16000", # 16kHz is enough for speech
"-ac", "1", # Mono
str(output_path)
],
capture_output=True,
check=True
)
compressed_size = output_path.stat().st_size / (1024 * 1024) # MB
ratio = original_size / compressed_size if compressed_size > 0 else 0
print(f" Original: {original_size:.2f} MB")
print(f" Compressed: {compressed_size:.2f} MB")
print(f" Compression ratio: {ratio:.1f}x")
return True
except subprocess.CalledProcessError as e:
print(f"β Compression failed: {e.stderr.decode()}")
return False
# ============================================================================
# GEMINI TRANSCRIPTION & ANALYSIS
# ============================================================================
def setup_gemini(api_key: str):
"""Configure Gemini API."""
genai.configure(api_key=api_key)
return genai.GenerativeModel(GEMINI_MODEL)
def transcribe_audio(model, audio_path: Path) -> str:
"""Transcribe audio using Gemini."""
print(f"ποΈ Transcribing with Gemini ({GEMINI_MODEL})...")
# Upload the audio file
audio_file = genai.upload_file(str(audio_path))
# Generate transcription
response = model.generate_content(
[TRANSCRIPTION_PROMPT, audio_file],
generation_config=genai.GenerationConfig(
temperature=0.1, # Low temperature for accuracy
max_output_tokens=8192,
)
)
# Clean up uploaded file
try:
audio_file.delete()
except:
pass
return response.text
def generate_breakdown(model, transcript: str) -> str:
"""Generate structured breakdown using Gemini."""
print(f"π Generating structured breakdown...")
prompt = BREAKDOWN_PROMPT.format(transcript=transcript)
response = model.generate_content(
prompt,
generation_config=genai.GenerationConfig(
temperature=0.3, # Slightly creative for better organization
max_output_tokens=8192,
)
)
return response.text
# ============================================================================
# FILE OUTPUT
# ============================================================================
def save_markdown(content: str, output_path: Path, title: str):
"""Save content to markdown file with metadata."""
timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
header = f"""---
title: {title}
created: {timestamp}
generator: voice-to-notes
---
"""
output_path.write_text(header + content, encoding="utf-8")
print(f" β
Saved: {output_path}")
# ============================================================================
# MAIN WORKFLOW
# ============================================================================
def process_audio(
input_path: Path,
output_dir: Path,
api_key: str,
skip_compression: bool = False
):
"""Main processing workflow."""
# Validate input
if not input_path.exists():
print(f"β File not found: {input_path}")
return False
# Setup output directory
output_dir.mkdir(parents=True, exist_ok=True)
# Base name for output files
base_name = input_path.stem
print(f"\n{'='*60}")
print(f"π΅ Processing: {input_path.name}")
print(f"{'='*60}\n")
# Step 1: Compress audio (optional)
if skip_compression:
audio_to_transcribe = input_path
print("βοΈ Skipping compression (using original file)")
else:
if not check_ffmpeg():
print("β οΈ FFmpeg not found. Install with: brew install ffmpeg")
print(" Proceeding with original file...")
audio_to_transcribe = input_path
else:
with tempfile.NamedTemporaryFile(suffix=".opus", delete=False) as tmp:
compressed_path = Path(tmp.name)
if compress_audio(input_path, compressed_path):
audio_to_transcribe = compressed_path
else:
audio_to_transcribe = input_path
print(" Using original file instead...")
# Step 2: Setup Gemini
try:
model = setup_gemini(api_key)
except Exception as e:
print(f"β Failed to setup Gemini: {e}")
return False
# Step 3: Transcribe
try:
transcript = transcribe_audio(model, audio_to_transcribe)
print(f" β
Transcription complete ({len(transcript)} characters)")
except Exception as e:
print(f"β Transcription failed: {e}")
return False
# Step 4: Generate breakdown
try:
breakdown = generate_breakdown(model, transcript)
print(f" β
Breakdown complete")
except Exception as e:
print(f"β Breakdown generation failed: {e}")
breakdown = None
# Step 5: Save outputs
print(f"\nπ Saving output files...")
# Raw transcript
transcript_path = output_dir / f"{base_name}_transcript.md"
save_markdown(
f"# Transcript: {base_name}\n\n{transcript}",
transcript_path,
f"Transcript - {base_name}"
)
# Breakdown
if breakdown:
breakdown_path = output_dir / f"{base_name}_breakdown.md"
save_markdown(
breakdown,
breakdown_path,
f"Breakdown - {base_name}"
)
# Cleanup temp files
if not skip_compression and audio_to_transcribe != input_path:
try:
audio_to_transcribe.unlink()
except:
pass
print(f"\n{'='*60}")
print(f"β
Processing complete!")
print(f"{'='*60}\n")
return True
def main():
parser = argparse.ArgumentParser(
description="Convert voice memos to structured markdown notes",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
Examples:
%(prog)s recording.m4a
%(prog)s recording.mp3 -o ./notes
%(prog)s meeting.m4a --no-compress
Environment:
Set GEMINI_API_KEY environment variable or use --api-key flag
"""
)
parser.add_argument(
"input",
type=Path,
help="Input audio file (.m4a, .mp3, .wav, etc.)"
)
parser.add_argument(
"-o", "--output",
type=Path,
default=None,
help="Output directory (default: same as input file)"
)
parser.add_argument(
"--api-key",
type=str,
default=None,
help="Gemini API key (or set GEMINI_API_KEY env var)"
)
parser.add_argument(
"--no-compress",
action="store_true",
help="Skip audio compression step"
)
args = parser.parse_args()
# Get API key
api_key = args.api_key or os.environ.get("GEMINI_API_KEY")
if not api_key:
print("β Error: Gemini API key required")
print(" Set GEMINI_API_KEY environment variable or use --api-key flag")
print(" Get your key from: https://aistudio.google.com/apikey")
sys.exit(1)
# Set output directory
output_dir = args.output or args.input.parent
# Process
success = process_audio(
input_path=args.input,
output_dir=output_dir,
api_key=api_key,
skip_compression=args.no_compress
)
sys.exit(0 if success else 1)
if __name__ == "__main__":
main()