-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathutils.py
More file actions
88 lines (71 loc) · 2.28 KB
/
Copy pathutils.py
File metadata and controls
88 lines (71 loc) · 2.28 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
import ffmpeg
# ===============================
# CONVERSÃO PARA WAV
# ===============================
def converter_para_wav(input_path, output_path):
"""
Converte qualquer arquivo de áudio para WAV mono 16 kHz.
Retorna o caminho convertido se bem-sucedido, ou None se falhar.
"""
try:
(
ffmpeg
.input(input_path)
.output(
output_path,
ac=1, # mono
ar=16000 # 16 kHz
)
.overwrite_output()
.run(capture_stdout=True, capture_stderr=True)
)
return output_path
except ffmpeg.Error as e:
# Log detalhado útil para debug e transparência
print("FFmpeg error:", e.stderr.decode("utf-8"))
return None
# ===============================
# CARREGAR ÁUDIO EM BYTES
# ===============================
def carregar_audio(path):
"""
Lê áudio e retorna bytes (compatível com st.audio).
Retorna None caso o arquivo não exista ou não possa ser lido.
"""
try:
with open(path, "rb") as f:
return f.read()
except FileNotFoundError:
return None
except Exception:
return None
# ===============================
# RECONSTRUIR CONVERSA ROTULADA
# ===============================
def reconstruir_conversa(rotulos, texto_original):
"""
Recebe rótulos no formato retornado pelo text_labeler e o texto original.
Retorna um texto final no formato:
Pessoa: fala 1
Pessoa: fala 2
com preservação da ordem com base nos offsets start/end.
"""
itens = []
# Achatar os rótulos em uma lista única
for speaker, spans in rotulos.items():
for span in spans:
itens.append({
"speaker": speaker,
"start": span["start"],
"end": span["end"],
"content": texto_original[span["start"]:span["end"]]
})
# Ordenar pelo início do trecho
itens_ordenados = sorted(itens, key=lambda x: x["start"])
# Construir texto final
linhas = []
for item in itens_ordenados:
fala = item["content"].strip()
linhas.append(f"{item['speaker']}: {fala}")
# Dupla quebra para separar falas visualmente
return "\n\n".join(linhas)