feat(voice): migration CosyVoice → OmniVoice (.cvps → .ovsp)
Suit docs/OMNIVOICE_VOICE_MIGRATION.md (app patiente migrée vc17+, 22/06). - enroll.py : réécriture encodeur OmniVoice (create_voice_clone_prompt → format binaire OVSP : magic|T_ref|text_len|tokens[8*T_ref]|ref_text). Cap réf 16s. prepare_wav 24kHz. - transcribe.py : ASR interne OmniVoice (plus d'openai-whisper, absent de ov_venv). - transfer.py : omnivoice_dir_for (<MODELS_DIR>/omnivoice/voices), deployed_ovsp, push_ovsp, suffixe .ovsp. - bridge.py : OV_PYTHON (/opt/Kazeia/ov_venv). worker dispatch inchangé. - orchestrator.py : .ovsp, ovsp_bytes ; verrou d'exclusivité + archive WAV INCHANGÉS. - store : colonne cvps_bytes → ovsp_bytes (+ migration ALTER RENAME idempotente). - tests adaptés. 39/39. - validé live sous ov_venv : enroll → .ovsp format byte-identique à damien.ovsp (réf engine) ; auto-ASR FR correcte ; bridge ping/transcribe/enroll chaud OK. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
46343c9ad5
commit
adc0904009
|
|
@ -226,10 +226,10 @@ def create_app(adb: Adb | None = None, store: Store | None = None,
|
|||
now=_now_ms())
|
||||
return report
|
||||
|
||||
# ---- voix (enrôlement CosyVoice : WAV device → .cvps → flotte) --------
|
||||
# ---- voix (enrôlement OmniVoice : WAV device → .ovsp → flotte) --------
|
||||
@app.get("/api/voices/health")
|
||||
def voices_health():
|
||||
return {"encoder_available": bridge.available(), "cv_python": bridge.cv_python}
|
||||
return {"encoder_available": bridge.available(), "ov_python": bridge.ov_python}
|
||||
|
||||
# ⚠️ Routes littérales sous /api/voices/ AVANT /{serial} sinon "autosync" est
|
||||
# capté comme un serial (FastAPI matche dans l'ordre de déclaration).
|
||||
|
|
@ -251,7 +251,7 @@ def create_app(adb: Adb | None = None, store: Store | None = None,
|
|||
|
||||
@app.post("/api/voices/{serial}/{voice_id}/prepare")
|
||||
def voices_prepare(serial: str, voice_id: str):
|
||||
# pull + archive chiffré + transcription Whisper (à relire avant enrôlement)
|
||||
# pull + archive chiffré + transcription (ASR OmniVoice, à relire avant enrôlement)
|
||||
return _guard(lambda: vo.prepare(adb, store, bridge, serial, voice_id,
|
||||
now=_now_ms(), actor=_OPERATOR))
|
||||
|
||||
|
|
|
|||
|
|
@ -73,10 +73,10 @@ CREATE TABLE IF NOT EXISTS voices (
|
|||
source_wav_path TEXT, -- chemin device d'origine
|
||||
wav_file TEXT, -- WAV chiffré archivé (relatif au dossier voices/)
|
||||
wav_sha256 TEXT, -- intégrité du WAV d'origine
|
||||
cvps_bytes INTEGER,
|
||||
ovsp_bytes INTEGER,
|
||||
archived_at INTEGER, -- WAV récupéré + archivé sur le PC
|
||||
enrolled_at INTEGER, -- .cvps produit
|
||||
deployed_at INTEGER, -- .cvps poussé sur la tablette
|
||||
enrolled_at INTEGER, -- .ovsp produit
|
||||
deployed_at INTEGER, -- .ovsp poussé sur la tablette
|
||||
wav_deleted_at INTEGER, -- WAV supprimé de la tablette
|
||||
locked_profile_id TEXT, -- assignation exclusive à un profil patient (NULL = généraliste)
|
||||
PRIMARY KEY (serial, voice_id)
|
||||
|
|
@ -113,6 +113,9 @@ class Store:
|
|||
cols = {r["name"] for r in self._db.execute("PRAGMA table_info(voices)")}
|
||||
if "locked_profile_id" not in cols:
|
||||
self._db.execute("ALTER TABLE voices ADD COLUMN locked_profile_id TEXT")
|
||||
# Migration OmniVoice (23/06/2026) : cvps_bytes → ovsp_bytes.
|
||||
if "cvps_bytes" in cols and "ovsp_bytes" not in cols:
|
||||
self._db.execute("ALTER TABLE voices RENAME COLUMN cvps_bytes TO ovsp_bytes")
|
||||
|
||||
def close(self) -> None:
|
||||
with self._lock:
|
||||
|
|
@ -327,9 +330,9 @@ class Store:
|
|||
self._voice_upsert(serial, voice_id, transcription_enc=v.seal(text),
|
||||
language=language)
|
||||
|
||||
def mark_voice_enrolled(self, serial: str, voice_id: str, cvps_bytes: int, *, now: int) -> None:
|
||||
def mark_voice_enrolled(self, serial: str, voice_id: str, ovsp_bytes: int, *, now: int) -> None:
|
||||
with self._lock:
|
||||
self._voice_upsert(serial, voice_id, cvps_bytes=cvps_bytes, enrolled_at=now)
|
||||
self._voice_upsert(serial, voice_id, ovsp_bytes=ovsp_bytes, enrolled_at=now)
|
||||
|
||||
def mark_voice_deployed(self, serial: str, voice_id: str, *, now: int) -> None:
|
||||
with self._lock:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
"""Sous-système voix (enrôlement CosyVoice WAV → .cvps).
|
||||
"""Sous-système voix (enrôlement OmniVoice WAV → .ovsp).
|
||||
|
||||
⚠️ `enroll.py`/`worker.py` tournent sous l'env encodeur (cv_venv : torch). Ne PAS
|
||||
les importer depuis l'API (Python 3.14 sans torch) — l'API parle au worker par
|
||||
⚠️ `enroll.py`/`worker.py` tournent sous l'env encodeur (ov_venv : torch + omnivoice).
|
||||
Ne PAS les importer depuis l'API (Python 3.14 sans torch) — l'API parle au worker par
|
||||
subprocess. Ce `__init__` reste volontairement vide d'imports lourds.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
"""Pont API ↔ worker d'enrôlement (côté Python 3.14, SANS torch).
|
||||
|
||||
L'encodeur vit dans cv_venv (torch), l'API en Python 3.14 → on ne peut pas l'importer
|
||||
en-process. `VoiceBridge` lance et pilote le worker `cv_venv` (`voice/worker.py`) en
|
||||
subprocess, lui parle en JSON-lines, et sérialise les jobs (un verrou : torch n'est de
|
||||
toute façon pas thread-safe en inférence concurrente). Démarrage paresseux (le worker
|
||||
n'est lancé qu'au 1ᵉʳ job → l'API ne paie le coût que si on enrôle).
|
||||
L'encodeur OmniVoice vit dans ov_venv (torch + omnivoice + transformers), l'API en
|
||||
Python 3.14 → on ne peut pas l'importer en-process. `VoiceBridge` lance et pilote le
|
||||
worker `ov_venv` (`voice/worker.py`) en subprocess, lui parle en JSON-lines, et
|
||||
sérialise les jobs (torch n'est pas thread-safe en inférence concurrente). Démarrage
|
||||
paresseux (le worker n'est lancé qu'au 1ᵉʳ job → l'API ne paie le coût que si on enrôle).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -15,8 +15,8 @@ import subprocess
|
|||
import threading
|
||||
from pathlib import Path
|
||||
|
||||
# Interpréteur de l'env encodeur (torch + cosyvoice + whisper + gguf).
|
||||
CV_PYTHON = os.environ.get("CV_PYTHON", "/opt/Kazeia/cv_venv/bin/python")
|
||||
# Interpréteur de l'env encodeur (torch + omnivoice + transformers + librosa/soundfile).
|
||||
OV_PYTHON = os.environ.get("OV_PYTHON", "/opt/Kazeia/ov_venv/bin/python")
|
||||
_REPO = Path(__file__).resolve().parents[2] # /opt/Kazeia-central
|
||||
|
||||
|
||||
|
|
@ -25,27 +25,27 @@ class VoiceWorkerError(RuntimeError):
|
|||
|
||||
|
||||
class VoiceBridge:
|
||||
"""Gère un worker cv_venv chaud. Thread-safe (jobs sérialisés)."""
|
||||
"""Gère un worker ov_venv chaud. Thread-safe (jobs sérialisés)."""
|
||||
|
||||
def __init__(self, cv_python: str = CV_PYTHON) -> None:
|
||||
self.cv_python = cv_python
|
||||
def __init__(self, ov_python: str = OV_PYTHON) -> None:
|
||||
self.ov_python = ov_python
|
||||
self._proc: subprocess.Popen | None = None
|
||||
self._lock = threading.Lock()
|
||||
self._id = 0
|
||||
|
||||
def available(self) -> bool:
|
||||
return os.path.exists(self.cv_python)
|
||||
return os.path.exists(self.ov_python)
|
||||
|
||||
def _ensure(self) -> None:
|
||||
if self._proc and self._proc.poll() is None:
|
||||
return
|
||||
if not os.path.exists(self.cv_python):
|
||||
if not os.path.exists(self.ov_python):
|
||||
raise VoiceWorkerError(
|
||||
f"interpréteur encodeur introuvable: {self.cv_python} "
|
||||
f"(provisionner cv_venv, ou définir CV_PYTHON)")
|
||||
f"interpréteur encodeur introuvable: {self.ov_python} "
|
||||
f"(provisionner ov_venv, ou définir OV_PYTHON)")
|
||||
env = dict(os.environ, PYTHONPATH=str(_REPO), PYTHONUNBUFFERED="1")
|
||||
self._proc = subprocess.Popen(
|
||||
[self.cv_python, "-m", "kazeia_central.voice.worker"],
|
||||
[self.ov_python, "-m", "kazeia_central.voice.worker"],
|
||||
stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=None, # stderr → console
|
||||
text=True, env=env, bufsize=1,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,164 +1,100 @@
|
|||
"""Enrôlement vocal CosyVoice : WAV de référence → artefact `.cvps`.
|
||||
"""Enrôlement vocal OmniVoice : WAV de référence → clone-prompt `.ovsp`.
|
||||
|
||||
⚠️ DOIT s'exécuter sous l'environnement encodeur (`cv_venv` : torch + cosyvoice +
|
||||
gguf + onnxruntime), PAS sous l'interpréteur de l'API Kazeia-central (Python 3.14,
|
||||
sans torch). C'est pourquoi ce module est invoqué par un **worker subprocess**
|
||||
(`voice/worker.py`) tournant sous `cv_venv`, pas importé en-process par l'API.
|
||||
|
||||
Réimplémente la logique des 25 lignes de `/opt/Kazeia/cv_make_prompt.py` (cf.
|
||||
VOICE_ENROLLMENT_SPEC §2), avec deux différences voulues :
|
||||
- chemins repo/teacher **configurables** (env `CV_REPO`, `CV_TEACHER`), pas codés en dur ;
|
||||
- teacher chargé **une seule fois** (singleton) pour enchaîner les voix à chaud.
|
||||
|
||||
Contrat de sortie (VOICE_ENROLLMENT_SPEC §3) — conteneur GGUF, arch
|
||||
`cosyvoice-prompt-speech`, 4 tenseurs aux noms/dtypes EXACTS (le loader C++
|
||||
`nativeLoadVoice` ne tolère aucune dérive) :
|
||||
feat mel prompt numpy [T, 80] float32 (GGUF: ne0=80, ne1=T)
|
||||
embedding x-vector numpy [1, dim] float32 (GGUF: ne0=dim, ne1=1)
|
||||
tokens speech tokens numpy [n] int32
|
||||
text transcription numpy [octets] int8 (UTF-8 brut)
|
||||
⚠️ Sous l'env encodeur `ov_venv` (torch + omnivoice + transformers + librosa/soundfile),
|
||||
invoqué par le worker subprocess (pas en-process API py3.14).
|
||||
Réimplémente /opt/Kazeia-engine/dist/omnivoice/scripts/make_ovsp.py.
|
||||
Format .ovsp : int32 'OVSP' | int32 T_ref | int32 text_len | int32 tokens[8*T_ref] | char ref_text[].
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
import logging, os, struct, sys
|
||||
import numpy as np
|
||||
|
||||
# La stack CosyVoice/librosa/numba crache du DEBUG sur stdout/stderr. Le worker (à
|
||||
# venir) utilisera stdout pour un protocole JSON → on muselle ces loggers ici pour
|
||||
# garder stdout propre. Indispensable avant de brancher l'IPC.
|
||||
for _noisy in ("numba", "matplotlib", "jieba", "modelscope", "cosyvoice"):
|
||||
for _noisy in ("numba", "matplotlib", "transformers", "omnivoice", "huggingface_hub"):
|
||||
logging.getLogger(_noisy).setLevel(logging.WARNING)
|
||||
os.environ.setdefault("NUMBA_DISABLE_JIT", "0")
|
||||
|
||||
CV_REPO = os.environ.get("CV_REPO", "/opt/Kazeia/cosyvoice-repo")
|
||||
CV_TEACHER = os.environ.get("CV_TEACHER", "/opt/Kazeia/_models_dl/cosyvoice3-0.5b")
|
||||
|
||||
GGUF_ARCH = "cosyvoice-prompt-speech"
|
||||
|
||||
# Préparation du clip. Le tokenizer s3 plante DUR au-delà de 30 s ("do not support
|
||||
# extract speech token for audio longer than 30s") et la similarité chute sous ~8 s
|
||||
# (§4). Les captures device ne sont pas calibrées (ex. 111 s stéréo 44 kHz observé) →
|
||||
# on normalise systématiquement : mono, 16 kHz, silences rognés, fenêtre cible ~15 s.
|
||||
# 22 s (≤ _HARD_MAX_S) : la phrase de consentement (~13 s) + la fin phonétiquement
|
||||
# riche (nasales, semi-voyelles, ch/j/r, + question/exclamation) entrent ENTIÈREMENT
|
||||
# dans le prompt enrôlé ; sinon la tête légale remplit seule les 15 s et la fin
|
||||
# (l'info utile au clonage) est tronquée. Prompt plus long ⇒ meilleure similarité.
|
||||
CV_TARGET_S = float(os.environ.get("CV_TARGET_S", "22.0"))
|
||||
_PREP_SR = 16000
|
||||
_HARD_MAX_S = 28.0 # marge sous la limite dure 30 s du tokenizer
|
||||
OV_MODEL = os.environ.get("OV_MODEL", "k2-fsa/OmniVoice")
|
||||
OVSP_MAGIC = 0x4F565350
|
||||
# Cap ~16 s : T_ref ~390 -> bucket 640 (rapide). Au-delà = bucket 768/1024 (lent) +
|
||||
# risque de saturer le plus grand bucket. prepare_wav partagé avec la transcription.
|
||||
OV_TARGET_S = float(os.environ.get("OV_TARGET_S", "16.0"))
|
||||
_PREP_SR = 24000
|
||||
_HARD_MAX_S = 20.0
|
||||
_MIN_WARN_S = 6.0
|
||||
|
||||
_teacher_singleton = None
|
||||
|
||||
_model_singleton = None
|
||||
|
||||
class EnrollmentError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
def prepare_wav(in_path: str, out_path: str, target_s: float = CV_TARGET_S) -> dict:
|
||||
"""Normalise un WAV de capture en segment exploitable : mono, 16 kHz, silences de
|
||||
bord rognés, tronqué à `target_s` (borné dur sous 30 s). Renvoie un résumé."""
|
||||
import librosa
|
||||
import soundfile as sf
|
||||
|
||||
y, _ = librosa.load(in_path, sr=_PREP_SR, mono=True) # downmix + resample
|
||||
def prepare_wav(in_path: str, out_path: str, target_s: float = OV_TARGET_S) -> dict:
|
||||
import librosa, soundfile as sf
|
||||
y, _ = librosa.load(in_path, sr=_PREP_SR, mono=True)
|
||||
orig_s = len(y) / _PREP_SR
|
||||
yt, _ = librosa.effects.trim(y, top_db=30) # rogne silences de bord
|
||||
yt, _ = librosa.effects.trim(y, top_db=30)
|
||||
if yt.size == 0:
|
||||
yt = y
|
||||
keep_s = min(target_s, _HARD_MAX_S)
|
||||
yt = yt[: int(keep_s * _PREP_SR)]
|
||||
yt = yt[: int(min(target_s, _HARD_MAX_S) * _PREP_SR)]
|
||||
final_s = len(yt) / _PREP_SR
|
||||
sf.write(out_path, yt, _PREP_SR, subtype="PCM_16")
|
||||
return {"orig_s": round(orig_s, 1), "final_s": round(final_s, 1),
|
||||
"too_short": final_s < _MIN_WARN_S}
|
||||
return {"orig_s": round(orig_s, 1), "final_s": round(final_s, 1), "too_short": final_s < _MIN_WARN_S}
|
||||
|
||||
|
||||
def load_teacher():
|
||||
"""Charge le teacher CosyVoice3-0.5B (~9,1 Go) UNE fois. Coûteux (dizaines de
|
||||
secondes) → garder le process chaud (worker). Lève EnrollmentError si l'env
|
||||
encodeur n'est pas provisionné (torch/cosyvoice/teacher absents)."""
|
||||
global _teacher_singleton
|
||||
if _teacher_singleton is not None:
|
||||
return _teacher_singleton
|
||||
if not os.path.isdir(CV_TEACHER):
|
||||
raise EnrollmentError(
|
||||
f"teacher CosyVoice introuvable: {CV_TEACHER} "
|
||||
f"(provisionner le modèle 0.5B ~9,1 Go, ou définir CV_TEACHER)")
|
||||
if not os.path.isdir(CV_REPO):
|
||||
raise EnrollmentError(f"repo CosyVoice introuvable: {CV_REPO} (définir CV_REPO)")
|
||||
for p in (CV_REPO, os.path.join(CV_REPO, "third_party/Matcha-TTS")):
|
||||
if p not in sys.path:
|
||||
sys.path.insert(0, p)
|
||||
def load_model():
|
||||
global _model_singleton
|
||||
if _model_singleton is not None:
|
||||
return _model_singleton
|
||||
try:
|
||||
from cosyvoice.cli.cosyvoice import CosyVoice3
|
||||
except Exception as e: # torch/cosyvoice absents = mauvais interpréteur
|
||||
raise EnrollmentError(
|
||||
f"import CosyVoice échoué ({e}). Exécuter sous l'env encodeur "
|
||||
f"(cv_venv : torch + cosyvoice + gguf).") from e
|
||||
_teacher_singleton = CosyVoice3(CV_TEACHER, load_trt=False, fp16=False)
|
||||
return _teacher_singleton
|
||||
import torch
|
||||
from omnivoice.models.omnivoice import OmniVoice
|
||||
except Exception as e:
|
||||
raise EnrollmentError(f"import OmniVoice échoué ({e}). Exécuter sous ov_venv.") from e
|
||||
_model_singleton = OmniVoice.from_pretrained(OV_MODEL, torch_dtype=torch.float32).to("cpu").eval()
|
||||
return _model_singleton
|
||||
|
||||
def _clone_prompt(prep_wav: str, ref_text):
|
||||
import torch
|
||||
m = load_model()
|
||||
with torch.no_grad():
|
||||
return m.create_voice_clone_prompt(ref_audio=prep_wav, ref_text=ref_text or None)
|
||||
|
||||
def enroll(wav_path: str, text: str, out_path: str) -> dict:
|
||||
"""WAV + transcription → `.cvps` à `out_path`. Renvoie un résumé (shapes/taille)
|
||||
pour validation. `text` doit correspondre au contenu parlé du WAV (§4)."""
|
||||
import gguf
|
||||
|
||||
def transcribe_ref(wav_path: str) -> dict:
|
||||
"""ASR du segment préparé (16 s) via OmniVoice, pour relecture opérateur."""
|
||||
import tempfile
|
||||
if not os.path.isfile(wav_path):
|
||||
raise EnrollmentError(f"WAV introuvable: {wav_path}")
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
raise EnrollmentError("transcription vide : CosyVoice exige le texte de référence")
|
||||
|
||||
import tempfile
|
||||
|
||||
cv = load_teacher()
|
||||
fe = cv.frontend
|
||||
|
||||
# Normalise la capture (mono/16k/≤~15s) dans un fichier temporaire avant extraction.
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
prep = os.path.join(td, "prep.wav")
|
||||
prep_info = prepare_wav(wav_path, prep)
|
||||
feat, _ = fe._extract_speech_feat(prep) # [1, T, 80]
|
||||
emb = fe._extract_spk_embedding(prep) # [1, dim]
|
||||
tok, _ = fe._extract_speech_token(prep) # [1, n]
|
||||
|
||||
feat = feat.squeeze(0).cpu().numpy().astype(np.float32) # [T, 80]
|
||||
emb = emb.cpu().numpy().astype(np.float32) # [1, dim]
|
||||
tok = tok.squeeze(0).cpu().numpy().astype(np.int32) # [n]
|
||||
txt = np.frombuffer(text.encode("utf-8"), dtype=np.int8) # [octets]
|
||||
vcp = _clone_prompt(prep, None)
|
||||
return {"text": (getattr(vcp, "ref_text", "") or "").strip(),
|
||||
"language": None, "model": "omnivoice-asr", "prep": prep_info}
|
||||
|
||||
def enroll(wav_path: str, text: str, out_path: str) -> dict:
|
||||
import tempfile
|
||||
if not os.path.isfile(wav_path):
|
||||
raise EnrollmentError(f"WAV introuvable: {wav_path}")
|
||||
text = (text or "").strip()
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
prep = os.path.join(td, "prep.wav")
|
||||
prep_info = prepare_wav(wav_path, prep)
|
||||
vcp = _clone_prompt(prep, text if text else None)
|
||||
toks = vcp.ref_audio_tokens.cpu().numpy().astype(np.int32) # [8, T_ref]
|
||||
C, T = toks.shape
|
||||
if C != 8:
|
||||
raise EnrollmentError(f"attendu 8 codebooks, eu {C}")
|
||||
ref_text = (getattr(vcp, "ref_text", "") or text or "").strip()
|
||||
txt = ref_text.encode("utf-8")
|
||||
os.makedirs(os.path.dirname(os.path.abspath(out_path)), exist_ok=True)
|
||||
w = gguf.GGUFWriter(out_path, GGUF_ARCH)
|
||||
w.add_tensor("feat", feat)
|
||||
w.add_tensor("embedding", emb)
|
||||
w.add_tensor("tokens", tok)
|
||||
w.add_tensor("text", txt)
|
||||
w.write_header_to_file()
|
||||
w.write_kv_data_to_file()
|
||||
w.write_tensors_to_file()
|
||||
w.close()
|
||||
|
||||
return {
|
||||
"out_path": out_path,
|
||||
"bytes": os.path.getsize(out_path),
|
||||
"feat_shape": list(feat.shape),
|
||||
"embedding_dim": int(emb.shape[1]),
|
||||
"n_tokens": int(tok.shape[0]),
|
||||
"text_bytes": int(txt.shape[0]),
|
||||
"prep": prep_info,
|
||||
}
|
||||
|
||||
with open(out_path, "wb") as f:
|
||||
f.write(struct.pack("<i", OVSP_MAGIC))
|
||||
f.write(struct.pack("<i", T))
|
||||
f.write(struct.pack("<i", len(txt)))
|
||||
f.write(toks.tobytes())
|
||||
f.write(txt)
|
||||
return {"out_path": out_path, "bytes": os.path.getsize(out_path), "t_ref": int(T),
|
||||
"ref_seconds": round(T / 25.0, 1), "text_bytes": len(txt),
|
||||
"ref_text": ref_text, "prep": prep_info}
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Usage : cv_venv/bin/python -m kazeia_central.voice.enroll <wav> <text|fichier.txt> <out.cvps>
|
||||
import json
|
||||
|
||||
ref, ref_txt, out = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||
text = open(ref_txt).read() if ref_txt.endswith(".txt") and os.path.isfile(ref_txt) else ref_txt
|
||||
text = "" if ref_txt == "-" else (open(ref_txt).read() if ref_txt.endswith(".txt") and os.path.isfile(ref_txt) else ref_txt)
|
||||
print(json.dumps(enroll(ref, text, out), ensure_ascii=False))
|
||||
|
|
|
|||
|
|
@ -1,11 +1,11 @@
|
|||
"""Orchestration de l'enrôlement vocal (le workflow complet).
|
||||
|
||||
Compose : provider (`/voices` → wav_path), transfert adb, bridge (worker cv_venv),
|
||||
Compose : provider (`/voices` → wav_path), transfert adb, bridge (worker ov_venv),
|
||||
store chiffré. Tourne sous l'API py3.14 ; le calcul lourd est délégué au worker.
|
||||
|
||||
Workflow (VOICE_ENROLLMENT_SPEC §1, décisions 2026-06-19) :
|
||||
connexion → lister voix sans .cvps → pull WAV → ARCHIVER (store chiffré) →
|
||||
transcrire (Whisper, relecture opérateur) → enrôler (.cvps) → pousser sur la
|
||||
connexion → lister voix sans .ovsp → pull WAV → ARCHIVER (store chiffré) →
|
||||
transcrire (ASR OmniVoice, relecture opérateur) → enrôler (.ovsp) → pousser sur la
|
||||
tablette → SUPPRIMER le WAV de la tablette (gardé sur Kazeia-central).
|
||||
|
||||
Découpé pour permettre la relecture de la transcription : `prepare` (pull+archive+
|
||||
|
|
@ -65,8 +65,8 @@ def list_voices(adb: Adb, store: Store, serial: str) -> list[dict[str, Any]]:
|
|||
"""Inventaire voix de la tablette croisé avec l'archive store (statut, verrou,
|
||||
profils utilisateurs) par voix."""
|
||||
rows = _provider_voices(adb, serial)
|
||||
cosy = transfer.cosyvoice_dir_for(rows[0].wav_path) if rows else None
|
||||
deployed = transfer.deployed_cvps(adb, serial, cosy) if cosy else set()
|
||||
ov = transfer.omnivoice_dir_for(rows[0].wav_path) if rows else None
|
||||
deployed = transfer.deployed_ovsp(adb, serial, ov) if ov else set()
|
||||
recs = {r["voice_id"]: r for r in store.voices(serial)} if store.is_unlocked else {}
|
||||
used_by, _ = _profiles_using(adb, serial)
|
||||
out = []
|
||||
|
|
@ -139,7 +139,7 @@ def _ensure_wav_archived(adb: Adb, store: Store, serial: str, voice_id: str, *,
|
|||
|
||||
def prepare(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id: str, *,
|
||||
now: int, actor: str | None = None) -> dict[str, Any]:
|
||||
"""Pull + archive (chiffré) + transcription Whisper. Renvoie le texte à relire."""
|
||||
"""Pull + archive (chiffré) + transcription (ASR OmniVoice). Renvoie le texte à relire."""
|
||||
data = _ensure_wav_archived(adb, store, serial, voice_id, now=now)
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
local = os.path.join(td, f"{voice_id}.wav")
|
||||
|
|
@ -154,27 +154,27 @@ def prepare(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id:
|
|||
def enroll_deploy(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id: str,
|
||||
transcription: str, *, delete_source: bool, now: int,
|
||||
actor: str | None = None) -> dict[str, Any]:
|
||||
"""Enrôle (.cvps) → pousse sur la tablette → (option) supprime le WAV device.
|
||||
"""Enrôle (.ovsp) → pousse sur la tablette → (option) supprime le WAV device.
|
||||
Le WAV reste archivé côté PC → suppression device réversible par ré-enrôlement."""
|
||||
data = _ensure_wav_archived(adb, store, serial, voice_id, now=now)
|
||||
wav_path = _find_wav_path(adb, serial, voice_id)
|
||||
cosy = transfer.cosyvoice_dir_for(wav_path) if wav_path else None
|
||||
if not cosy:
|
||||
raise ValueError(f"dossier cosyvoice indéterminé pour {voice_id}")
|
||||
ov = transfer.omnivoice_dir_for(wav_path) if wav_path else None
|
||||
if not ov:
|
||||
raise ValueError(f"dossier omnivoice indéterminé pour {voice_id}")
|
||||
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
local = os.path.join(td, f"{voice_id}.wav")
|
||||
open(local, "wb").write(data)
|
||||
cvps = os.path.join(td, f"{voice_id}.cvps")
|
||||
info = bridge.enroll(local, transcription, cvps)
|
||||
ovsp = os.path.join(td, f"{voice_id}.ovsp")
|
||||
info = bridge.enroll(local, transcription, ovsp)
|
||||
store.set_voice_transcription(serial, voice_id, transcription, None, now=now)
|
||||
store.mark_voice_enrolled(serial, voice_id, info["bytes"], now=now)
|
||||
remote = transfer.push_cvps(adb, serial, cvps, cosy, voice_id)
|
||||
remote = transfer.push_ovsp(adb, serial, ovsp, ov, voice_id)
|
||||
store.mark_voice_deployed(serial, voice_id, now=now)
|
||||
store.audit("voice_enroll_deploy", actor=actor, target=f"{serial}/{voice_id}",
|
||||
detail={"cvps_bytes": info["bytes"]}, now=now)
|
||||
detail={"ovsp_bytes": info["bytes"]}, now=now)
|
||||
|
||||
result = {"voice_id": voice_id, "cvps_bytes": info["bytes"],
|
||||
result = {"voice_id": voice_id, "ovsp_bytes": info["bytes"],
|
||||
"deployed_to": remote, "wav_deleted": False}
|
||||
if delete_source and wav_path:
|
||||
transfer.delete_device_wav(adb, serial, wav_path)
|
||||
|
|
|
|||
|
|
@ -1,45 +1,8 @@
|
|||
"""Transcription Whisper (PC) du segment de référence — tourne sous cv_venv.
|
||||
|
||||
Important : on transcrit le **segment préparé** (mono/16k/~15 s), PAS la capture
|
||||
brute. Le texte du `.cvps` doit correspondre à ce qui est parlé DANS le segment
|
||||
enrôlé (§3/§4) ; transcrire les 111 s d'origine donnerait un texte qui ne colle pas
|
||||
aux ~15 s réellement extraits. `prepare_wav` étant déterministe, le segment transcrit
|
||||
ici == le segment enrôlé.
|
||||
"""
|
||||
|
||||
"""Transcription du segment de référence (16 s) via l'ASR interne d'OmniVoice — ov_venv.
|
||||
Le texte du .ovsp doit correspondre à l'audio réellement enrôlé ; prepare_wav étant
|
||||
partagé/déterministe, le segment transcrit == le segment enrôlé."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
from .enroll import prepare_wav
|
||||
|
||||
# Défaut "small" : "base" transcrit mal le FR (validé), "small" est le bon compromis
|
||||
# qualité/CPU ; "medium" encore mieux mais lourd. La transcription est de toute façon
|
||||
# relue par l'opérateur. Configurable via CV_WHISPER_MODEL.
|
||||
CV_WHISPER_MODEL = os.environ.get("CV_WHISPER_MODEL", "small")
|
||||
|
||||
_model_cache: dict[str, object] = {}
|
||||
|
||||
|
||||
def _model(size: str):
|
||||
if size not in _model_cache:
|
||||
import whisper # openai-whisper (présent dans cv_venv)
|
||||
_model_cache[size] = whisper.load_model(size)
|
||||
return _model_cache[size]
|
||||
|
||||
from .enroll import transcribe_ref
|
||||
|
||||
def transcribe(wav_path: str, model: str | None = None) -> dict:
|
||||
"""Transcrit le segment de référence. Renvoie {text, language, model, prep}."""
|
||||
size = model or CV_WHISPER_MODEL
|
||||
m = _model(size)
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
prep = os.path.join(td, "prep.wav")
|
||||
prep_info = prepare_wav(wav_path, prep)
|
||||
result = m.transcribe(prep, fp16=False) # CPU → fp16 off
|
||||
return {
|
||||
"text": (result.get("text") or "").strip(),
|
||||
"language": result.get("language"),
|
||||
"model": size,
|
||||
"prep": prep_info,
|
||||
}
|
||||
return transcribe_ref(wav_path) # {text, language, model, prep}
|
||||
|
|
|
|||
|
|
@ -2,8 +2,11 @@
|
|||
|
||||
Py3.14-safe (pas de torch) : seulement des opérations adb. Le chemin des fichiers est
|
||||
**device-dépendant** → on part toujours de la colonne `wav_path` que le provider
|
||||
`/voices` renvoie (autorité), jamais d'un chemin codé en dur (cf. découverte du test :
|
||||
WAV sous `/data/local/tmp/kazeia/…` en dev, `Android/data/com.kazeia/…` en prod).
|
||||
`/voices` renvoie (autorité), jamais d'un chemin codé en dur (WAV sous
|
||||
`/data/local/tmp/kazeia/…` en dev, `Android/data/com.kazeia/…` en prod).
|
||||
|
||||
Voix = clone-prompts **OmniVoice `.ovsp`** (migration 23/06/2026, ex-CosyVoice `.cvps`),
|
||||
déployés dans `<MODELS_DIR>/omnivoice/voices/`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -13,39 +16,39 @@ import posixpath
|
|||
from ..adb import Adb, AdbError
|
||||
|
||||
# Le provider expose VOICE_WAV_DIR = "${MODELS_DIR}/../voix/voix" → ce marqueur
|
||||
# permet d'extraire MODELS_DIR depuis un wav_path, donc le dossier cosyvoice cible.
|
||||
# permet d'extraire MODELS_DIR depuis un wav_path, donc le dossier voix OmniVoice cible.
|
||||
_WAV_MARKER = "/../voix/voix/"
|
||||
|
||||
|
||||
def cosyvoice_dir_for(wav_path: str) -> str:
|
||||
"""Dossier device des `.cvps` (`<MODELS_DIR>/cosyvoice`) dérivé d'un `wav_path`."""
|
||||
def omnivoice_dir_for(wav_path: str) -> str:
|
||||
"""Dossier device des `.ovsp` (`<MODELS_DIR>/omnivoice/voices`) dérivé d'un `wav_path`."""
|
||||
if _WAV_MARKER in wav_path:
|
||||
models = wav_path.split(_WAV_MARKER)[0]
|
||||
return posixpath.join(models, "cosyvoice")
|
||||
# Fallback : …/kazeia/voix/voix/<id>.wav → …/kazeia/models/cosyvoice
|
||||
return posixpath.join(models, "omnivoice", "voices")
|
||||
# Fallback : …/kazeia/voix/voix/<id>.wav → …/kazeia/models/omnivoice/voices
|
||||
voix_dir = posixpath.dirname(wav_path)
|
||||
kazeia = posixpath.dirname(posixpath.dirname(voix_dir))
|
||||
return posixpath.join(kazeia, "models", "cosyvoice")
|
||||
return posixpath.join(kazeia, "models", "omnivoice", "voices")
|
||||
|
||||
|
||||
def deployed_cvps(adb: Adb, serial: str, cosyvoice_dir: str) -> set[str]:
|
||||
"""Ensemble des voice_id ayant un `.cvps` déjà déployé sur la tablette."""
|
||||
out = adb.shell(f"ls {cosyvoice_dir} 2>/dev/null", serial=serial)
|
||||
def deployed_ovsp(adb: Adb, serial: str, ovsp_dir: str) -> set[str]:
|
||||
"""Ensemble des voice_id ayant un `.ovsp` déjà déployé sur la tablette."""
|
||||
out = adb.shell(f"ls {ovsp_dir} 2>/dev/null", serial=serial)
|
||||
return {line[:-5].strip() for line in out.splitlines()
|
||||
if line.strip().endswith(".cvps")}
|
||||
if line.strip().endswith(".ovsp")}
|
||||
|
||||
|
||||
def pull_wav(adb: Adb, serial: str, wav_path: str, local_path: str) -> None:
|
||||
adb.pull(wav_path, local_path, serial=serial)
|
||||
|
||||
|
||||
def push_cvps(adb: Adb, serial: str, local_cvps: str, cosyvoice_dir: str,
|
||||
def push_ovsp(adb: Adb, serial: str, local_ovsp: str, ovsp_dir: str,
|
||||
voice_id: str) -> str:
|
||||
"""Pousse le `.cvps` dans le dossier cosyvoice de la tablette. Renvoie le chemin
|
||||
device. Crée le dossier au besoin."""
|
||||
adb.shell(f"mkdir -p {cosyvoice_dir}", serial=serial)
|
||||
remote = posixpath.join(cosyvoice_dir, f"{voice_id}.cvps")
|
||||
adb.push(local_cvps, remote, serial=serial)
|
||||
"""Pousse le `.ovsp` dans le dossier voix OmniVoice de la tablette. Renvoie le
|
||||
chemin device. Crée le dossier au besoin."""
|
||||
adb.shell(f"mkdir -p {ovsp_dir}", serial=serial)
|
||||
remote = posixpath.join(ovsp_dir, f"{voice_id}.ovsp")
|
||||
adb.push(local_ovsp, remote, serial=serial)
|
||||
return remote
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,11 +1,11 @@
|
|||
"""Worker d'enrôlement vocal — process persistant sous cv_venv.
|
||||
"""Worker d'enrôlement vocal — process persistant sous ov_venv.
|
||||
|
||||
Lancé par `voice/bridge.py` (côté API py3.14) via :
|
||||
cv_venv/bin/python -m kazeia_central.voice.worker
|
||||
Garde le teacher CosyVoice (et le modèle Whisper) **chauds**, et sert une file de
|
||||
ov_venv/bin/python -m kazeia_central.voice.worker
|
||||
Garde le modèle OmniVoice (enrôlement + ASR de réf) **chaud**, et sert une file de
|
||||
jobs via un protocole JSON-lines sur stdin/stdout.
|
||||
|
||||
Canal protocole propre : la stack torch/librosa/whisper écrit du bruit sur stdout.
|
||||
Canal protocole propre : la stack torch/librosa/transformers écrit du bruit sur stdout.
|
||||
On capture le **vrai** stdout (fd 1) pour le protocole AVANT de rediriger fd 1 vers
|
||||
stderr — ainsi tout `print()`/sortie C de lib part sur stderr (console), et seul le
|
||||
JSON du protocole circule sur le pipe lu par le bridge. Sans ça, l'IPC serait corrompu.
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Tests du sous-système voix : dérivation chemins, archive store, orchestration.
|
||||
|
||||
Le moteur lourd (CosyVoice/Whisper) n'est PAS testé ici (validé live) — on injecte
|
||||
un faux bridge et un faux adb pour prouver la COMPOSITION (workflow) sans le modèle.
|
||||
Le moteur lourd (OmniVoice) n'est PAS testé ici (validé live) — on injecte un faux
|
||||
bridge et un faux adb pour prouver la COMPOSITION (workflow) sans le modèle.
|
||||
"""
|
||||
|
||||
import nacl.pwhash
|
||||
|
|
@ -24,15 +24,15 @@ def _store(tmp_path):
|
|||
|
||||
|
||||
# ---- dérivation des chemins device ---------------------------------------
|
||||
def test_cosyvoice_dir_marker():
|
||||
def test_omnivoice_dir_marker():
|
||||
wp = "/data/local/tmp/kazeia/models/../voix/voix/amir.wav"
|
||||
assert transfer.cosyvoice_dir_for(wp) == "/data/local/tmp/kazeia/models/cosyvoice"
|
||||
assert transfer.omnivoice_dir_for(wp) == "/data/local/tmp/kazeia/models/omnivoice/voices"
|
||||
|
||||
|
||||
def test_cosyvoice_dir_fallback_prod():
|
||||
def test_omnivoice_dir_fallback_prod():
|
||||
wp = "/sdcard/Android/data/com.kazeia/files/kazeia/voix/voix/amir.wav"
|
||||
assert transfer.cosyvoice_dir_for(wp) == \
|
||||
"/sdcard/Android/data/com.kazeia/files/kazeia/models/cosyvoice"
|
||||
assert transfer.omnivoice_dir_for(wp) == \
|
||||
"/sdcard/Android/data/com.kazeia/files/kazeia/models/omnivoice/voices"
|
||||
|
||||
|
||||
# ---- archive voix dans le store ------------------------------------------
|
||||
|
|
@ -62,7 +62,7 @@ def test_voice_status_marks(tmp_path):
|
|||
s.mark_voice_enrolled("D1", "amir", 242000, now=NOW)
|
||||
s.mark_voice_deployed("D1", "amir", now=NOW)
|
||||
r = s.voice_record("D1", "amir")
|
||||
assert r["cvps_bytes"] == 242000 and r["enrolled_at"] and r["deployed_at"]
|
||||
assert r["ovsp_bytes"] == 242000 and r["enrolled_at"] and r["deployed_at"]
|
||||
|
||||
|
||||
# ---- orchestration (faux adb + faux bridge) ------------------------------
|
||||
|
|
@ -87,7 +87,7 @@ class _VoiceAdb(Adb):
|
|||
|
||||
def shell(self, command, *, serial=None, timeout=None):
|
||||
if command.startswith("ls "):
|
||||
return "\n".join(f"{d}.cvps" for d in self._deployed)
|
||||
return "\n".join(f"{d}.ovsp" for d in self._deployed)
|
||||
if "rm -f" in command:
|
||||
self.deleted.append(command.split("rm -f ")[1].split(" &&")[0])
|
||||
return "OK"
|
||||
|
|
@ -105,8 +105,8 @@ class _FakeBridge:
|
|||
return {"text": "bonjour ceci est ma voix", "language": "fr", "model": "small"}
|
||||
|
||||
def enroll(self, wav, text, out):
|
||||
open(out, "wb").write(b"GGUF-CVPS-" + text.encode()[:5])
|
||||
return {"bytes": 242000, "feat_shape": [750, 80], "n_tokens": 375, "text_bytes": len(text)}
|
||||
open(out, "wb").write(b"OVSP" + text.encode()[:5])
|
||||
return {"bytes": 242000, "t_ref": 364, "ref_seconds": 14.6, "text_bytes": len(text)}
|
||||
|
||||
|
||||
_WP = "/data/local/tmp/kazeia/models/../voix/voix/amir.wav"
|
||||
|
|
@ -138,9 +138,9 @@ def test_enroll_deploy_pushes_and_optional_delete(tmp_path):
|
|||
# sans suppression
|
||||
r = vo.enroll_deploy(adb, s, _FakeBridge(), "D1", "amir", "ma voix",
|
||||
delete_source=False, now=NOW)
|
||||
assert r["cvps_bytes"] == 242000 and r["wav_deleted"] is False
|
||||
assert r["ovsp_bytes"] == 242000 and r["wav_deleted"] is False
|
||||
assert adb.pushed and adb.pushed[0][1] == \
|
||||
"/data/local/tmp/kazeia/models/cosyvoice/amir.cvps"
|
||||
"/data/local/tmp/kazeia/models/omnivoice/voices/amir.ovsp"
|
||||
assert s.voice_record("D1", "amir")["deployed_at"] == NOW
|
||||
assert adb.deleted == []
|
||||
# avec suppression
|
||||
|
|
|
|||
Loading…
Reference in New Issue