forked from heygen-com/hyperframes
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgenerate_emu_audio.py
More file actions
69 lines (59 loc) · 2.97 KB
/
Copy pathgenerate_emu_audio.py
File metadata and controls
69 lines (59 loc) · 2.97 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
import json
import subprocess
import os
cuts = [
{"id": "cut_01", "text": "Nước Úc từng thua một cuộc chiến — trước loài đà điểu emu."},
{"id": "cut_02", "text": "Cuối năm 1932, khoảng hai mươi nghìn con emu tràn vào Tây Úc, phá nát mùa lúa mì của những cựu binh vừa định cư sau Thế chiến thứ nhất."},
{"id": "cut_03", "text": "Nông dân cầu cứu, và chính phủ Úc làm một việc không ai ngờ tới:"},
{"id": "cut_04", "text": "điều quân đội, trang bị súng máy Lewis, để tiêu diệt đàn chim,"},
{"id": "cut_05", "text": "dưới chỉ huy của Thiếu tá Meredith."},
{"id": "cut_06", "text": "Nhưng emu chạy nhanh hơn năm mươi ki-lô-mét một giờ, tách đàn thành hàng chục nhóm nhỏ, né đạn thành thục như đã được huấn luyện."},
{"id": "cut_07", "text": "Quân đội bắn gần hai nghìn năm trăm viên đạn, chỉ hạ được vài trăm con."},
{"id": "cut_08", "text": "Chưa đầy một tháng sau, chiến dịch bị hủy bỏ."},
{"id": "cut_09", "text": "Báo chí Úc gọi thẳng đó là một thất bại quân sự. Chiến tranh Emu chính thức đi vào lịch sử — con người thua, loài chim thắng."}
]
output_dir = os.path.abspath("public/audio/cuts")
os.makedirs(output_dir, exist_ok=True)
durations = []
for cut in cuts:
cid = cut["id"]
text = cut["text"]
raw_mp3 = os.path.join(output_dir, f"{cid}_raw.mp3")
final_mp3 = os.path.join(output_dir, f"{cid}.mp3")
print(f"Generating TTS for {cid}...")
cmd_tts = [
"/Applications/_QuangLB/Workspace/Agent/capcut-tts-api/.venv/bin/python",
"/Applications/_QuangLB/Workspace/Agent/capcut-tts-api/generate_audio.py",
"--text", text,
"--output", raw_mp3,
"--voice", "thanh_nien_tu_tin"
]
subprocess.run(cmd_tts, check=True, cwd="/Applications/_QuangLB/Workspace/Agent/capcut-tts-api")
print(f"Speeding up {cid} by 1.3x...")
cmd_ffmpeg = [
"ffmpeg", "-y", "-i", raw_mp3,
"-filter:a", "atempo=1.3",
"-vn", final_mp3
]
subprocess.run(cmd_ffmpeg, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
if os.path.exists(raw_mp3):
os.remove(raw_mp3)
cmd_ffprobe = [
"ffprobe", "-v", "error",
"-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1",
final_mp3
]
res = subprocess.run(cmd_ffprobe, check=True, capture_output=True, text=True)
dur = float(res.stdout.strip())
print(f"{cid} duration: {dur:.3f}s")
durations.append({
"id": cid,
"filename": f"{cid}.mp3",
"duration": round(dur, 3),
"text": text
})
durations_path = os.path.join(output_dir, "durations.json")
with open(durations_path, "w", encoding="utf-8") as f:
json.dump(durations, f, ensure_ascii=False, indent=2)
print("Audio generation finished successfully!")