JARVIS: asistente de voz local para Linux
Nucleo propio: oye con whisper.cpp, piensa con un modelo de Ollama, habla con Piper, y hace RAG sobre los apuntes del usuario. 100% local, sin cuentas ni claves. Escrito bajo una restriccion dura, 4 GB de VRAM: el cerebro y whisper comparten tarjeta y solo caben porque estan dimensionados para ello. El RAG usa embeddings estaticos con busqueda hibrida; la voz clonada se sirve de una cache de frases. Incluye instalador (install.sh), requisitos, y documentacion del stack, del manejo de root y de las acciones. Los apuntes indexados y el diario NO se incluyen: son privados y el .gitignore los bloquea.
This commit is contained in:
commit
8e4bc8ad94
125 changed files with 25033 additions and 0 deletions
107
voz/calibrar_parecido.py
Normal file
107
voz/calibrar_parecido.py
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Da escala a la puntuacion de parecido: techo, suelo y donde cae el clon.
|
||||
|
||||
Un 0,68 no significa nada por si solo. Hace falta saber cuanto puntuan DOS
|
||||
grabaciones distintas del mismo original (el techo realista: ni el propio
|
||||
hablante da 1,0 consigo mismo) y cuanto puntua una voz que no tiene nada que
|
||||
ver (el suelo). Entre esos dos numeros se lee si el clon esta cerca o lejos.
|
||||
|
||||
De paso prueba usar los mp3 ORIGINALES descargados como referencia, sin pasar
|
||||
por los wav normalizados: si normalizar quito informacion, se vera aqui.
|
||||
|
||||
python calibrar_parecido.py [cuda|cpu]
|
||||
"""
|
||||
import glob
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
import torch
|
||||
|
||||
DISPOSITIVO = sys.argv[1] if len(sys.argv) > 1 else "cuda"
|
||||
BASE = os.path.dirname(os.path.abspath(__file__))
|
||||
REF = os.path.join(BASE, "ref")
|
||||
COMPARATIVA = os.path.join(BASE, "comparativa")
|
||||
DESCARGAS = os.path.expanduser("~/Downloads")
|
||||
|
||||
os.environ.setdefault("COQUI_TOS_AGREED", "1")
|
||||
LARGO = dict(gpt_cond_len=30, gpt_cond_chunk_len=6, max_ref_length=30)
|
||||
|
||||
|
||||
def parecido(a, b):
|
||||
return float(torch.nn.functional.cosine_similarity(
|
||||
a.squeeze().float(), b.squeeze().float(), dim=0))
|
||||
|
||||
|
||||
def a_wav(origen, destino):
|
||||
"""mp3 -> wav mono 24k, sin tocar niveles."""
|
||||
subprocess.run(["ffmpeg", "-y", "-loglevel", "error", "-i", origen,
|
||||
"-ar", "24000", "-ac", "1", destino], check=True)
|
||||
return destino
|
||||
|
||||
|
||||
def main():
|
||||
from TTS.api import TTS
|
||||
print(f"cargando XTTS en {DISPOSITIVO}...", flush=True)
|
||||
api = TTS("tts_models/multilingual/multi-dataset/xtts_v2").to(DISPOSITIVO)
|
||||
modelo = api.synthesizer.tts_model
|
||||
|
||||
def emb(rutas):
|
||||
_, e = modelo.get_conditioning_latents(audio_path=list(rutas), **LARGO)
|
||||
return e
|
||||
|
||||
patron = emb([os.path.join(REF, "jarvis_lento_b.wav")])
|
||||
|
||||
print("\n TECHO — otras grabaciones del MISMO original:")
|
||||
for nombre in ("jarvis_lento_a.wav", "jarvis_ref.wav", "jarvis_ref2.wav",
|
||||
"jarvis_ref_corto.wav"):
|
||||
ruta = os.path.join(REF, nombre)
|
||||
if os.path.exists(ruta):
|
||||
print(f" {parecido(patron, emb([ruta])):.4f} {nombre}")
|
||||
|
||||
print("\n SUELO — una voz que no es esa (Piper, castellano):")
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
piper = os.path.join(tmp, "piper.wav")
|
||||
guion = os.path.expanduser(
|
||||
"~/COFRE/CODERS/JARVIS/config/jarvis-piper.sh")
|
||||
try:
|
||||
subprocess.run([guion, piper, "Buenas noches, señor. Soy su asistente personal."],
|
||||
check=True, capture_output=True, timeout=120)
|
||||
print(f" {parecido(patron, emb([piper])):.4f} piper es_ES")
|
||||
except Exception as e:
|
||||
print(f" (no pude generar con Piper: {e})")
|
||||
|
||||
print("\n EL CLON — lo generado en comparar_ajustes.py:")
|
||||
for wav in sorted(glob.glob(os.path.join(COMPARATIVA, "*.wav"))):
|
||||
print(f" {parecido(patron, emb([wav])):.4f} {os.path.basename(wav)}")
|
||||
|
||||
print("\n ORIGINALES sin normalizar (mp3 de fish.audio) como referencia:")
|
||||
mp3s = sorted(glob.glob(os.path.join(DESCARGAS, "Jarvis-*.mp3")))
|
||||
if not mp3s:
|
||||
print(" (no hay mp3 en ~/Downloads)")
|
||||
return
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
convertidos = [a_wav(m, os.path.join(tmp, f"o{i}.wav"))
|
||||
for i, m in enumerate(mp3s)]
|
||||
for c, m in zip(convertidos, mp3s):
|
||||
print(f" {parecido(patron, emb([c])):.4f} {os.path.basename(m)[:52]}")
|
||||
|
||||
# y clonar usando los originales como referencia, a ver si mejora
|
||||
gpt, hablante = modelo.get_conditioning_latents(
|
||||
audio_path=convertidos, **LARGO)
|
||||
salida = modelo.inference(
|
||||
text="Buenas noches, señor. Soy su asistente personal.",
|
||||
language="es", gpt_cond_latent=gpt, speaker_embedding=hablante,
|
||||
temperature=0.65, repetition_penalty=2.0, speed=1.0,
|
||||
enable_text_splitting=True)
|
||||
import soundfile as sf
|
||||
destino = os.path.join(COMPARATIVA, "F_originales_sin_normalizar.wav")
|
||||
sf.write(destino, salida["wav"], 24000)
|
||||
print(f"\n {parecido(patron, emb([destino])):.4f} "
|
||||
f"F_originales_sin_normalizar.wav (clon con los mp3 crudos)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue