From adc09040097e775e8e37cd3c3c35124b0179a8e2 Mon Sep 17 00:00:00 2001 From: alf Date: Tue, 23 Jun 2026 10:15:11 +0200 Subject: [PATCH] =?UTF-8?q?feat(voice):=20migration=20CosyVoice=20?= =?UTF-8?q?=E2=86=92=20OmniVoice=20(.cvps=20=E2=86=92=20.ovsp)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Suit docs/OMNIVOICE_VOICE_MIGRATION.md (app patiente migrée vc17+, 22/06). - enroll.py : réécriture encodeur OmniVoice (create_voice_clone_prompt → format binaire OVSP : magic|T_ref|text_len|tokens[8*T_ref]|ref_text). Cap réf 16s. prepare_wav 24kHz. - transcribe.py : ASR interne OmniVoice (plus d'openai-whisper, absent de ov_venv). - transfer.py : omnivoice_dir_for (/omnivoice/voices), deployed_ovsp, push_ovsp, suffixe .ovsp. - bridge.py : OV_PYTHON (/opt/Kazeia/ov_venv). worker dispatch inchangé. - orchestrator.py : .ovsp, ovsp_bytes ; verrou d'exclusivité + archive WAV INCHANGÉS. - store : colonne cvps_bytes → ovsp_bytes (+ migration ALTER RENAME idempotente). - tests adaptés. 39/39. - validé live sous ov_venv : enroll → .ovsp format byte-identique à damien.ovsp (réf engine) ; auto-ASR FR correcte ; bridge ping/transcribe/enroll chaud OK. Co-Authored-By: Claude Opus 4.8 (1M context) --- kazeia_central/api/app.py | 6 +- kazeia_central/store/db.py | 13 +- kazeia_central/voice/__init__.py | 6 +- kazeia_central/voice/bridge.py | 30 ++-- kazeia_central/voice/enroll.py | 198 +++++++++------------------ kazeia_central/voice/orchestrator.py | 30 ++-- kazeia_central/voice/transcribe.py | 47 +------ kazeia_central/voice/transfer.py | 39 +++--- kazeia_central/voice/worker.py | 8 +- tests/test_voice.py | 26 ++-- 10 files changed, 154 insertions(+), 249 deletions(-) diff --git a/kazeia_central/api/app.py b/kazeia_central/api/app.py index c4406ef..4cb9de1 100644 --- a/kazeia_central/api/app.py +++ b/kazeia_central/api/app.py @@ -226,10 +226,10 @@ def create_app(adb: Adb | None = None, store: Store | None = None, now=_now_ms()) return report - # ---- voix (enrôlement CosyVoice : WAV device → .cvps → flotte) -------- + # ---- voix (enrôlement OmniVoice : WAV device → .ovsp → flotte) -------- @app.get("/api/voices/health") def voices_health(): - return {"encoder_available": bridge.available(), "cv_python": bridge.cv_python} + return {"encoder_available": bridge.available(), "ov_python": bridge.ov_python} # ⚠️ Routes littérales sous /api/voices/ AVANT /{serial} sinon "autosync" est # capté comme un serial (FastAPI matche dans l'ordre de déclaration). @@ -251,7 +251,7 @@ def create_app(adb: Adb | None = None, store: Store | None = None, @app.post("/api/voices/{serial}/{voice_id}/prepare") def voices_prepare(serial: str, voice_id: str): - # pull + archive chiffré + transcription Whisper (à relire avant enrôlement) + # pull + archive chiffré + transcription (ASR OmniVoice, à relire avant enrôlement) return _guard(lambda: vo.prepare(adb, store, bridge, serial, voice_id, now=_now_ms(), actor=_OPERATOR)) diff --git a/kazeia_central/store/db.py b/kazeia_central/store/db.py index 47bea8a..945ff05 100644 --- a/kazeia_central/store/db.py +++ b/kazeia_central/store/db.py @@ -73,10 +73,10 @@ CREATE TABLE IF NOT EXISTS voices ( source_wav_path TEXT, -- chemin device d'origine wav_file TEXT, -- WAV chiffré archivé (relatif au dossier voices/) wav_sha256 TEXT, -- intégrité du WAV d'origine - cvps_bytes INTEGER, + ovsp_bytes INTEGER, archived_at INTEGER, -- WAV récupéré + archivé sur le PC - enrolled_at INTEGER, -- .cvps produit - deployed_at INTEGER, -- .cvps poussé sur la tablette + enrolled_at INTEGER, -- .ovsp produit + deployed_at INTEGER, -- .ovsp poussé sur la tablette wav_deleted_at INTEGER, -- WAV supprimé de la tablette locked_profile_id TEXT, -- assignation exclusive à un profil patient (NULL = généraliste) PRIMARY KEY (serial, voice_id) @@ -113,6 +113,9 @@ class Store: cols = {r["name"] for r in self._db.execute("PRAGMA table_info(voices)")} if "locked_profile_id" not in cols: self._db.execute("ALTER TABLE voices ADD COLUMN locked_profile_id TEXT") + # Migration OmniVoice (23/06/2026) : cvps_bytes → ovsp_bytes. + if "cvps_bytes" in cols and "ovsp_bytes" not in cols: + self._db.execute("ALTER TABLE voices RENAME COLUMN cvps_bytes TO ovsp_bytes") def close(self) -> None: with self._lock: @@ -327,9 +330,9 @@ class Store: self._voice_upsert(serial, voice_id, transcription_enc=v.seal(text), language=language) - def mark_voice_enrolled(self, serial: str, voice_id: str, cvps_bytes: int, *, now: int) -> None: + def mark_voice_enrolled(self, serial: str, voice_id: str, ovsp_bytes: int, *, now: int) -> None: with self._lock: - self._voice_upsert(serial, voice_id, cvps_bytes=cvps_bytes, enrolled_at=now) + self._voice_upsert(serial, voice_id, ovsp_bytes=ovsp_bytes, enrolled_at=now) def mark_voice_deployed(self, serial: str, voice_id: str, *, now: int) -> None: with self._lock: diff --git a/kazeia_central/voice/__init__.py b/kazeia_central/voice/__init__.py index 687d086..f78e0fd 100644 --- a/kazeia_central/voice/__init__.py +++ b/kazeia_central/voice/__init__.py @@ -1,6 +1,6 @@ -"""Sous-système voix (enrôlement CosyVoice WAV → .cvps). +"""Sous-système voix (enrôlement OmniVoice WAV → .ovsp). -⚠️ `enroll.py`/`worker.py` tournent sous l'env encodeur (cv_venv : torch). Ne PAS -les importer depuis l'API (Python 3.14 sans torch) — l'API parle au worker par +⚠️ `enroll.py`/`worker.py` tournent sous l'env encodeur (ov_venv : torch + omnivoice). +Ne PAS les importer depuis l'API (Python 3.14 sans torch) — l'API parle au worker par subprocess. Ce `__init__` reste volontairement vide d'imports lourds. """ diff --git a/kazeia_central/voice/bridge.py b/kazeia_central/voice/bridge.py index 347e765..403d46c 100644 --- a/kazeia_central/voice/bridge.py +++ b/kazeia_central/voice/bridge.py @@ -1,10 +1,10 @@ """Pont API ↔ worker d'enrôlement (côté Python 3.14, SANS torch). -L'encodeur vit dans cv_venv (torch), l'API en Python 3.14 → on ne peut pas l'importer -en-process. `VoiceBridge` lance et pilote le worker `cv_venv` (`voice/worker.py`) en -subprocess, lui parle en JSON-lines, et sérialise les jobs (un verrou : torch n'est de -toute façon pas thread-safe en inférence concurrente). Démarrage paresseux (le worker -n'est lancé qu'au 1ᵉʳ job → l'API ne paie le coût que si on enrôle). +L'encodeur OmniVoice vit dans ov_venv (torch + omnivoice + transformers), l'API en +Python 3.14 → on ne peut pas l'importer en-process. `VoiceBridge` lance et pilote le +worker `ov_venv` (`voice/worker.py`) en subprocess, lui parle en JSON-lines, et +sérialise les jobs (torch n'est pas thread-safe en inférence concurrente). Démarrage +paresseux (le worker n'est lancé qu'au 1ᵉʳ job → l'API ne paie le coût que si on enrôle). """ from __future__ import annotations @@ -15,8 +15,8 @@ import subprocess import threading from pathlib import Path -# Interpréteur de l'env encodeur (torch + cosyvoice + whisper + gguf). -CV_PYTHON = os.environ.get("CV_PYTHON", "/opt/Kazeia/cv_venv/bin/python") +# Interpréteur de l'env encodeur (torch + omnivoice + transformers + librosa/soundfile). +OV_PYTHON = os.environ.get("OV_PYTHON", "/opt/Kazeia/ov_venv/bin/python") _REPO = Path(__file__).resolve().parents[2] # /opt/Kazeia-central @@ -25,27 +25,27 @@ class VoiceWorkerError(RuntimeError): class VoiceBridge: - """Gère un worker cv_venv chaud. Thread-safe (jobs sérialisés).""" + """Gère un worker ov_venv chaud. Thread-safe (jobs sérialisés).""" - def __init__(self, cv_python: str = CV_PYTHON) -> None: - self.cv_python = cv_python + def __init__(self, ov_python: str = OV_PYTHON) -> None: + self.ov_python = ov_python self._proc: subprocess.Popen | None = None self._lock = threading.Lock() self._id = 0 def available(self) -> bool: - return os.path.exists(self.cv_python) + return os.path.exists(self.ov_python) def _ensure(self) -> None: if self._proc and self._proc.poll() is None: return - if not os.path.exists(self.cv_python): + if not os.path.exists(self.ov_python): raise VoiceWorkerError( - f"interpréteur encodeur introuvable: {self.cv_python} " - f"(provisionner cv_venv, ou définir CV_PYTHON)") + f"interpréteur encodeur introuvable: {self.ov_python} " + f"(provisionner ov_venv, ou définir OV_PYTHON)") env = dict(os.environ, PYTHONPATH=str(_REPO), PYTHONUNBUFFERED="1") self._proc = subprocess.Popen( - [self.cv_python, "-m", "kazeia_central.voice.worker"], + [self.ov_python, "-m", "kazeia_central.voice.worker"], stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=None, # stderr → console text=True, env=env, bufsize=1, ) diff --git a/kazeia_central/voice/enroll.py b/kazeia_central/voice/enroll.py index 818e809..4a68014 100644 --- a/kazeia_central/voice/enroll.py +++ b/kazeia_central/voice/enroll.py @@ -1,164 +1,100 @@ -"""Enrôlement vocal CosyVoice : WAV de référence → artefact `.cvps`. +"""Enrôlement vocal OmniVoice : WAV de référence → clone-prompt `.ovsp`. -⚠️ DOIT s'exécuter sous l'environnement encodeur (`cv_venv` : torch + cosyvoice + -gguf + onnxruntime), PAS sous l'interpréteur de l'API Kazeia-central (Python 3.14, -sans torch). C'est pourquoi ce module est invoqué par un **worker subprocess** -(`voice/worker.py`) tournant sous `cv_venv`, pas importé en-process par l'API. - -Réimplémente la logique des 25 lignes de `/opt/Kazeia/cv_make_prompt.py` (cf. -VOICE_ENROLLMENT_SPEC §2), avec deux différences voulues : -- chemins repo/teacher **configurables** (env `CV_REPO`, `CV_TEACHER`), pas codés en dur ; -- teacher chargé **une seule fois** (singleton) pour enchaîner les voix à chaud. - -Contrat de sortie (VOICE_ENROLLMENT_SPEC §3) — conteneur GGUF, arch -`cosyvoice-prompt-speech`, 4 tenseurs aux noms/dtypes EXACTS (le loader C++ -`nativeLoadVoice` ne tolère aucune dérive) : - feat mel prompt numpy [T, 80] float32 (GGUF: ne0=80, ne1=T) - embedding x-vector numpy [1, dim] float32 (GGUF: ne0=dim, ne1=1) - tokens speech tokens numpy [n] int32 - text transcription numpy [octets] int8 (UTF-8 brut) +⚠️ Sous l'env encodeur `ov_venv` (torch + omnivoice + transformers + librosa/soundfile), +invoqué par le worker subprocess (pas en-process API py3.14). +Réimplémente /opt/Kazeia-engine/dist/omnivoice/scripts/make_ovsp.py. +Format .ovsp : int32 'OVSP' | int32 T_ref | int32 text_len | int32 tokens[8*T_ref] | char ref_text[]. """ - from __future__ import annotations - -import logging -import os -import sys - +import logging, os, struct, sys import numpy as np -# La stack CosyVoice/librosa/numba crache du DEBUG sur stdout/stderr. Le worker (à -# venir) utilisera stdout pour un protocole JSON → on muselle ces loggers ici pour -# garder stdout propre. Indispensable avant de brancher l'IPC. -for _noisy in ("numba", "matplotlib", "jieba", "modelscope", "cosyvoice"): +for _noisy in ("numba", "matplotlib", "transformers", "omnivoice", "huggingface_hub"): logging.getLogger(_noisy).setLevel(logging.WARNING) -os.environ.setdefault("NUMBA_DISABLE_JIT", "0") -CV_REPO = os.environ.get("CV_REPO", "/opt/Kazeia/cosyvoice-repo") -CV_TEACHER = os.environ.get("CV_TEACHER", "/opt/Kazeia/_models_dl/cosyvoice3-0.5b") - -GGUF_ARCH = "cosyvoice-prompt-speech" - -# Préparation du clip. Le tokenizer s3 plante DUR au-delà de 30 s ("do not support -# extract speech token for audio longer than 30s") et la similarité chute sous ~8 s -# (§4). Les captures device ne sont pas calibrées (ex. 111 s stéréo 44 kHz observé) → -# on normalise systématiquement : mono, 16 kHz, silences rognés, fenêtre cible ~15 s. -# 22 s (≤ _HARD_MAX_S) : la phrase de consentement (~13 s) + la fin phonétiquement -# riche (nasales, semi-voyelles, ch/j/r, + question/exclamation) entrent ENTIÈREMENT -# dans le prompt enrôlé ; sinon la tête légale remplit seule les 15 s et la fin -# (l'info utile au clonage) est tronquée. Prompt plus long ⇒ meilleure similarité. -CV_TARGET_S = float(os.environ.get("CV_TARGET_S", "22.0")) -_PREP_SR = 16000 -_HARD_MAX_S = 28.0 # marge sous la limite dure 30 s du tokenizer +OV_MODEL = os.environ.get("OV_MODEL", "k2-fsa/OmniVoice") +OVSP_MAGIC = 0x4F565350 +# Cap ~16 s : T_ref ~390 -> bucket 640 (rapide). Au-delà = bucket 768/1024 (lent) + +# risque de saturer le plus grand bucket. prepare_wav partagé avec la transcription. +OV_TARGET_S = float(os.environ.get("OV_TARGET_S", "16.0")) +_PREP_SR = 24000 +_HARD_MAX_S = 20.0 _MIN_WARN_S = 6.0 - -_teacher_singleton = None - +_model_singleton = None class EnrollmentError(RuntimeError): pass - -def prepare_wav(in_path: str, out_path: str, target_s: float = CV_TARGET_S) -> dict: - """Normalise un WAV de capture en segment exploitable : mono, 16 kHz, silences de - bord rognés, tronqué à `target_s` (borné dur sous 30 s). Renvoie un résumé.""" - import librosa - import soundfile as sf - - y, _ = librosa.load(in_path, sr=_PREP_SR, mono=True) # downmix + resample +def prepare_wav(in_path: str, out_path: str, target_s: float = OV_TARGET_S) -> dict: + import librosa, soundfile as sf + y, _ = librosa.load(in_path, sr=_PREP_SR, mono=True) orig_s = len(y) / _PREP_SR - yt, _ = librosa.effects.trim(y, top_db=30) # rogne silences de bord + yt, _ = librosa.effects.trim(y, top_db=30) if yt.size == 0: yt = y - keep_s = min(target_s, _HARD_MAX_S) - yt = yt[: int(keep_s * _PREP_SR)] + yt = yt[: int(min(target_s, _HARD_MAX_S) * _PREP_SR)] final_s = len(yt) / _PREP_SR sf.write(out_path, yt, _PREP_SR, subtype="PCM_16") - return {"orig_s": round(orig_s, 1), "final_s": round(final_s, 1), - "too_short": final_s < _MIN_WARN_S} + return {"orig_s": round(orig_s, 1), "final_s": round(final_s, 1), "too_short": final_s < _MIN_WARN_S} - -def load_teacher(): - """Charge le teacher CosyVoice3-0.5B (~9,1 Go) UNE fois. Coûteux (dizaines de - secondes) → garder le process chaud (worker). Lève EnrollmentError si l'env - encodeur n'est pas provisionné (torch/cosyvoice/teacher absents).""" - global _teacher_singleton - if _teacher_singleton is not None: - return _teacher_singleton - if not os.path.isdir(CV_TEACHER): - raise EnrollmentError( - f"teacher CosyVoice introuvable: {CV_TEACHER} " - f"(provisionner le modèle 0.5B ~9,1 Go, ou définir CV_TEACHER)") - if not os.path.isdir(CV_REPO): - raise EnrollmentError(f"repo CosyVoice introuvable: {CV_REPO} (définir CV_REPO)") - for p in (CV_REPO, os.path.join(CV_REPO, "third_party/Matcha-TTS")): - if p not in sys.path: - sys.path.insert(0, p) +def load_model(): + global _model_singleton + if _model_singleton is not None: + return _model_singleton try: - from cosyvoice.cli.cosyvoice import CosyVoice3 - except Exception as e: # torch/cosyvoice absents = mauvais interpréteur - raise EnrollmentError( - f"import CosyVoice échoué ({e}). Exécuter sous l'env encodeur " - f"(cv_venv : torch + cosyvoice + gguf).") from e - _teacher_singleton = CosyVoice3(CV_TEACHER, load_trt=False, fp16=False) - return _teacher_singleton + import torch + from omnivoice.models.omnivoice import OmniVoice + except Exception as e: + raise EnrollmentError(f"import OmniVoice échoué ({e}). Exécuter sous ov_venv.") from e + _model_singleton = OmniVoice.from_pretrained(OV_MODEL, torch_dtype=torch.float32).to("cpu").eval() + return _model_singleton +def _clone_prompt(prep_wav: str, ref_text): + import torch + m = load_model() + with torch.no_grad(): + return m.create_voice_clone_prompt(ref_audio=prep_wav, ref_text=ref_text or None) -def enroll(wav_path: str, text: str, out_path: str) -> dict: - """WAV + transcription → `.cvps` à `out_path`. Renvoie un résumé (shapes/taille) - pour validation. `text` doit correspondre au contenu parlé du WAV (§4).""" - import gguf - +def transcribe_ref(wav_path: str) -> dict: + """ASR du segment préparé (16 s) via OmniVoice, pour relecture opérateur.""" + import tempfile if not os.path.isfile(wav_path): raise EnrollmentError(f"WAV introuvable: {wav_path}") - text = (text or "").strip() - if not text: - raise EnrollmentError("transcription vide : CosyVoice exige le texte de référence") - - import tempfile - - cv = load_teacher() - fe = cv.frontend - - # Normalise la capture (mono/16k/≤~15s) dans un fichier temporaire avant extraction. with tempfile.TemporaryDirectory() as td: prep = os.path.join(td, "prep.wav") prep_info = prepare_wav(wav_path, prep) - feat, _ = fe._extract_speech_feat(prep) # [1, T, 80] - emb = fe._extract_spk_embedding(prep) # [1, dim] - tok, _ = fe._extract_speech_token(prep) # [1, n] - - feat = feat.squeeze(0).cpu().numpy().astype(np.float32) # [T, 80] - emb = emb.cpu().numpy().astype(np.float32) # [1, dim] - tok = tok.squeeze(0).cpu().numpy().astype(np.int32) # [n] - txt = np.frombuffer(text.encode("utf-8"), dtype=np.int8) # [octets] + vcp = _clone_prompt(prep, None) + return {"text": (getattr(vcp, "ref_text", "") or "").strip(), + "language": None, "model": "omnivoice-asr", "prep": prep_info} +def enroll(wav_path: str, text: str, out_path: str) -> dict: + import tempfile + if not os.path.isfile(wav_path): + raise EnrollmentError(f"WAV introuvable: {wav_path}") + text = (text or "").strip() + with tempfile.TemporaryDirectory() as td: + prep = os.path.join(td, "prep.wav") + prep_info = prepare_wav(wav_path, prep) + vcp = _clone_prompt(prep, text if text else None) + toks = vcp.ref_audio_tokens.cpu().numpy().astype(np.int32) # [8, T_ref] + C, T = toks.shape + if C != 8: + raise EnrollmentError(f"attendu 8 codebooks, eu {C}") + ref_text = (getattr(vcp, "ref_text", "") or text or "").strip() + txt = ref_text.encode("utf-8") os.makedirs(os.path.dirname(os.path.abspath(out_path)), exist_ok=True) - w = gguf.GGUFWriter(out_path, GGUF_ARCH) - w.add_tensor("feat", feat) - w.add_tensor("embedding", emb) - w.add_tensor("tokens", tok) - w.add_tensor("text", txt) - w.write_header_to_file() - w.write_kv_data_to_file() - w.write_tensors_to_file() - w.close() - - return { - "out_path": out_path, - "bytes": os.path.getsize(out_path), - "feat_shape": list(feat.shape), - "embedding_dim": int(emb.shape[1]), - "n_tokens": int(tok.shape[0]), - "text_bytes": int(txt.shape[0]), - "prep": prep_info, - } - + with open(out_path, "wb") as f: + f.write(struct.pack(" import json - ref, ref_txt, out = sys.argv[1], sys.argv[2], sys.argv[3] - text = open(ref_txt).read() if ref_txt.endswith(".txt") and os.path.isfile(ref_txt) else ref_txt + text = "" if ref_txt == "-" else (open(ref_txt).read() if ref_txt.endswith(".txt") and os.path.isfile(ref_txt) else ref_txt) print(json.dumps(enroll(ref, text, out), ensure_ascii=False)) diff --git a/kazeia_central/voice/orchestrator.py b/kazeia_central/voice/orchestrator.py index ef8862a..670d171 100644 --- a/kazeia_central/voice/orchestrator.py +++ b/kazeia_central/voice/orchestrator.py @@ -1,11 +1,11 @@ """Orchestration de l'enrôlement vocal (le workflow complet). -Compose : provider (`/voices` → wav_path), transfert adb, bridge (worker cv_venv), +Compose : provider (`/voices` → wav_path), transfert adb, bridge (worker ov_venv), store chiffré. Tourne sous l'API py3.14 ; le calcul lourd est délégué au worker. Workflow (VOICE_ENROLLMENT_SPEC §1, décisions 2026-06-19) : - connexion → lister voix sans .cvps → pull WAV → ARCHIVER (store chiffré) → - transcrire (Whisper, relecture opérateur) → enrôler (.cvps) → pousser sur la + connexion → lister voix sans .ovsp → pull WAV → ARCHIVER (store chiffré) → + transcrire (ASR OmniVoice, relecture opérateur) → enrôler (.ovsp) → pousser sur la tablette → SUPPRIMER le WAV de la tablette (gardé sur Kazeia-central). Découpé pour permettre la relecture de la transcription : `prepare` (pull+archive+ @@ -65,8 +65,8 @@ def list_voices(adb: Adb, store: Store, serial: str) -> list[dict[str, Any]]: """Inventaire voix de la tablette croisé avec l'archive store (statut, verrou, profils utilisateurs) par voix.""" rows = _provider_voices(adb, serial) - cosy = transfer.cosyvoice_dir_for(rows[0].wav_path) if rows else None - deployed = transfer.deployed_cvps(adb, serial, cosy) if cosy else set() + ov = transfer.omnivoice_dir_for(rows[0].wav_path) if rows else None + deployed = transfer.deployed_ovsp(adb, serial, ov) if ov else set() recs = {r["voice_id"]: r for r in store.voices(serial)} if store.is_unlocked else {} used_by, _ = _profiles_using(adb, serial) out = [] @@ -139,7 +139,7 @@ def _ensure_wav_archived(adb: Adb, store: Store, serial: str, voice_id: str, *, def prepare(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id: str, *, now: int, actor: str | None = None) -> dict[str, Any]: - """Pull + archive (chiffré) + transcription Whisper. Renvoie le texte à relire.""" + """Pull + archive (chiffré) + transcription (ASR OmniVoice). Renvoie le texte à relire.""" data = _ensure_wav_archived(adb, store, serial, voice_id, now=now) with tempfile.TemporaryDirectory() as td: local = os.path.join(td, f"{voice_id}.wav") @@ -154,27 +154,27 @@ def prepare(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id: def enroll_deploy(adb: Adb, store: Store, bridge: VoiceBridge, serial: str, voice_id: str, transcription: str, *, delete_source: bool, now: int, actor: str | None = None) -> dict[str, Any]: - """Enrôle (.cvps) → pousse sur la tablette → (option) supprime le WAV device. + """Enrôle (.ovsp) → pousse sur la tablette → (option) supprime le WAV device. Le WAV reste archivé côté PC → suppression device réversible par ré-enrôlement.""" data = _ensure_wav_archived(adb, store, serial, voice_id, now=now) wav_path = _find_wav_path(adb, serial, voice_id) - cosy = transfer.cosyvoice_dir_for(wav_path) if wav_path else None - if not cosy: - raise ValueError(f"dossier cosyvoice indéterminé pour {voice_id}") + ov = transfer.omnivoice_dir_for(wav_path) if wav_path else None + if not ov: + raise ValueError(f"dossier omnivoice indéterminé pour {voice_id}") with tempfile.TemporaryDirectory() as td: local = os.path.join(td, f"{voice_id}.wav") open(local, "wb").write(data) - cvps = os.path.join(td, f"{voice_id}.cvps") - info = bridge.enroll(local, transcription, cvps) + ovsp = os.path.join(td, f"{voice_id}.ovsp") + info = bridge.enroll(local, transcription, ovsp) store.set_voice_transcription(serial, voice_id, transcription, None, now=now) store.mark_voice_enrolled(serial, voice_id, info["bytes"], now=now) - remote = transfer.push_cvps(adb, serial, cvps, cosy, voice_id) + remote = transfer.push_ovsp(adb, serial, ovsp, ov, voice_id) store.mark_voice_deployed(serial, voice_id, now=now) store.audit("voice_enroll_deploy", actor=actor, target=f"{serial}/{voice_id}", - detail={"cvps_bytes": info["bytes"]}, now=now) + detail={"ovsp_bytes": info["bytes"]}, now=now) - result = {"voice_id": voice_id, "cvps_bytes": info["bytes"], + result = {"voice_id": voice_id, "ovsp_bytes": info["bytes"], "deployed_to": remote, "wav_deleted": False} if delete_source and wav_path: transfer.delete_device_wav(adb, serial, wav_path) diff --git a/kazeia_central/voice/transcribe.py b/kazeia_central/voice/transcribe.py index 4f4e06a..a43f2bb 100644 --- a/kazeia_central/voice/transcribe.py +++ b/kazeia_central/voice/transcribe.py @@ -1,45 +1,8 @@ -"""Transcription Whisper (PC) du segment de référence — tourne sous cv_venv. - -Important : on transcrit le **segment préparé** (mono/16k/~15 s), PAS la capture -brute. Le texte du `.cvps` doit correspondre à ce qui est parlé DANS le segment -enrôlé (§3/§4) ; transcrire les 111 s d'origine donnerait un texte qui ne colle pas -aux ~15 s réellement extraits. `prepare_wav` étant déterministe, le segment transcrit -ici == le segment enrôlé. -""" - +"""Transcription du segment de référence (16 s) via l'ASR interne d'OmniVoice — ov_venv. +Le texte du .ovsp doit correspondre à l'audio réellement enrôlé ; prepare_wav étant +partagé/déterministe, le segment transcrit == le segment enrôlé.""" from __future__ import annotations - -import os -import tempfile - -from .enroll import prepare_wav - -# Défaut "small" : "base" transcrit mal le FR (validé), "small" est le bon compromis -# qualité/CPU ; "medium" encore mieux mais lourd. La transcription est de toute façon -# relue par l'opérateur. Configurable via CV_WHISPER_MODEL. -CV_WHISPER_MODEL = os.environ.get("CV_WHISPER_MODEL", "small") - -_model_cache: dict[str, object] = {} - - -def _model(size: str): - if size not in _model_cache: - import whisper # openai-whisper (présent dans cv_venv) - _model_cache[size] = whisper.load_model(size) - return _model_cache[size] - +from .enroll import transcribe_ref def transcribe(wav_path: str, model: str | None = None) -> dict: - """Transcrit le segment de référence. Renvoie {text, language, model, prep}.""" - size = model or CV_WHISPER_MODEL - m = _model(size) - with tempfile.TemporaryDirectory() as td: - prep = os.path.join(td, "prep.wav") - prep_info = prepare_wav(wav_path, prep) - result = m.transcribe(prep, fp16=False) # CPU → fp16 off - return { - "text": (result.get("text") or "").strip(), - "language": result.get("language"), - "model": size, - "prep": prep_info, - } + return transcribe_ref(wav_path) # {text, language, model, prep} diff --git a/kazeia_central/voice/transfer.py b/kazeia_central/voice/transfer.py index 864a01b..d558854 100644 --- a/kazeia_central/voice/transfer.py +++ b/kazeia_central/voice/transfer.py @@ -2,8 +2,11 @@ Py3.14-safe (pas de torch) : seulement des opérations adb. Le chemin des fichiers est **device-dépendant** → on part toujours de la colonne `wav_path` que le provider -`/voices` renvoie (autorité), jamais d'un chemin codé en dur (cf. découverte du test : -WAV sous `/data/local/tmp/kazeia/…` en dev, `Android/data/com.kazeia/…` en prod). +`/voices` renvoie (autorité), jamais d'un chemin codé en dur (WAV sous +`/data/local/tmp/kazeia/…` en dev, `Android/data/com.kazeia/…` en prod). + +Voix = clone-prompts **OmniVoice `.ovsp`** (migration 23/06/2026, ex-CosyVoice `.cvps`), +déployés dans `/omnivoice/voices/`. """ from __future__ import annotations @@ -13,39 +16,39 @@ import posixpath from ..adb import Adb, AdbError # Le provider expose VOICE_WAV_DIR = "${MODELS_DIR}/../voix/voix" → ce marqueur -# permet d'extraire MODELS_DIR depuis un wav_path, donc le dossier cosyvoice cible. +# permet d'extraire MODELS_DIR depuis un wav_path, donc le dossier voix OmniVoice cible. _WAV_MARKER = "/../voix/voix/" -def cosyvoice_dir_for(wav_path: str) -> str: - """Dossier device des `.cvps` (`/cosyvoice`) dérivé d'un `wav_path`.""" +def omnivoice_dir_for(wav_path: str) -> str: + """Dossier device des `.ovsp` (`/omnivoice/voices`) dérivé d'un `wav_path`.""" if _WAV_MARKER in wav_path: models = wav_path.split(_WAV_MARKER)[0] - return posixpath.join(models, "cosyvoice") - # Fallback : …/kazeia/voix/voix/.wav → …/kazeia/models/cosyvoice + return posixpath.join(models, "omnivoice", "voices") + # Fallback : …/kazeia/voix/voix/.wav → …/kazeia/models/omnivoice/voices voix_dir = posixpath.dirname(wav_path) kazeia = posixpath.dirname(posixpath.dirname(voix_dir)) - return posixpath.join(kazeia, "models", "cosyvoice") + return posixpath.join(kazeia, "models", "omnivoice", "voices") -def deployed_cvps(adb: Adb, serial: str, cosyvoice_dir: str) -> set[str]: - """Ensemble des voice_id ayant un `.cvps` déjà déployé sur la tablette.""" - out = adb.shell(f"ls {cosyvoice_dir} 2>/dev/null", serial=serial) +def deployed_ovsp(adb: Adb, serial: str, ovsp_dir: str) -> set[str]: + """Ensemble des voice_id ayant un `.ovsp` déjà déployé sur la tablette.""" + out = adb.shell(f"ls {ovsp_dir} 2>/dev/null", serial=serial) return {line[:-5].strip() for line in out.splitlines() - if line.strip().endswith(".cvps")} + if line.strip().endswith(".ovsp")} def pull_wav(adb: Adb, serial: str, wav_path: str, local_path: str) -> None: adb.pull(wav_path, local_path, serial=serial) -def push_cvps(adb: Adb, serial: str, local_cvps: str, cosyvoice_dir: str, +def push_ovsp(adb: Adb, serial: str, local_ovsp: str, ovsp_dir: str, voice_id: str) -> str: - """Pousse le `.cvps` dans le dossier cosyvoice de la tablette. Renvoie le chemin - device. Crée le dossier au besoin.""" - adb.shell(f"mkdir -p {cosyvoice_dir}", serial=serial) - remote = posixpath.join(cosyvoice_dir, f"{voice_id}.cvps") - adb.push(local_cvps, remote, serial=serial) + """Pousse le `.ovsp` dans le dossier voix OmniVoice de la tablette. Renvoie le + chemin device. Crée le dossier au besoin.""" + adb.shell(f"mkdir -p {ovsp_dir}", serial=serial) + remote = posixpath.join(ovsp_dir, f"{voice_id}.ovsp") + adb.push(local_ovsp, remote, serial=serial) return remote diff --git a/kazeia_central/voice/worker.py b/kazeia_central/voice/worker.py index 1f57637..12fefb4 100644 --- a/kazeia_central/voice/worker.py +++ b/kazeia_central/voice/worker.py @@ -1,11 +1,11 @@ -"""Worker d'enrôlement vocal — process persistant sous cv_venv. +"""Worker d'enrôlement vocal — process persistant sous ov_venv. Lancé par `voice/bridge.py` (côté API py3.14) via : - cv_venv/bin/python -m kazeia_central.voice.worker -Garde le teacher CosyVoice (et le modèle Whisper) **chauds**, et sert une file de + ov_venv/bin/python -m kazeia_central.voice.worker +Garde le modèle OmniVoice (enrôlement + ASR de réf) **chaud**, et sert une file de jobs via un protocole JSON-lines sur stdin/stdout. -Canal protocole propre : la stack torch/librosa/whisper écrit du bruit sur stdout. +Canal protocole propre : la stack torch/librosa/transformers écrit du bruit sur stdout. On capture le **vrai** stdout (fd 1) pour le protocole AVANT de rediriger fd 1 vers stderr — ainsi tout `print()`/sortie C de lib part sur stderr (console), et seul le JSON du protocole circule sur le pipe lu par le bridge. Sans ça, l'IPC serait corrompu. diff --git a/tests/test_voice.py b/tests/test_voice.py index bd0e695..1c37054 100644 --- a/tests/test_voice.py +++ b/tests/test_voice.py @@ -1,7 +1,7 @@ """Tests du sous-système voix : dérivation chemins, archive store, orchestration. -Le moteur lourd (CosyVoice/Whisper) n'est PAS testé ici (validé live) — on injecte -un faux bridge et un faux adb pour prouver la COMPOSITION (workflow) sans le modèle. +Le moteur lourd (OmniVoice) n'est PAS testé ici (validé live) — on injecte un faux +bridge et un faux adb pour prouver la COMPOSITION (workflow) sans le modèle. """ import nacl.pwhash @@ -24,15 +24,15 @@ def _store(tmp_path): # ---- dérivation des chemins device --------------------------------------- -def test_cosyvoice_dir_marker(): +def test_omnivoice_dir_marker(): wp = "/data/local/tmp/kazeia/models/../voix/voix/amir.wav" - assert transfer.cosyvoice_dir_for(wp) == "/data/local/tmp/kazeia/models/cosyvoice" + assert transfer.omnivoice_dir_for(wp) == "/data/local/tmp/kazeia/models/omnivoice/voices" -def test_cosyvoice_dir_fallback_prod(): +def test_omnivoice_dir_fallback_prod(): wp = "/sdcard/Android/data/com.kazeia/files/kazeia/voix/voix/amir.wav" - assert transfer.cosyvoice_dir_for(wp) == \ - "/sdcard/Android/data/com.kazeia/files/kazeia/models/cosyvoice" + assert transfer.omnivoice_dir_for(wp) == \ + "/sdcard/Android/data/com.kazeia/files/kazeia/models/omnivoice/voices" # ---- archive voix dans le store ------------------------------------------ @@ -62,7 +62,7 @@ def test_voice_status_marks(tmp_path): s.mark_voice_enrolled("D1", "amir", 242000, now=NOW) s.mark_voice_deployed("D1", "amir", now=NOW) r = s.voice_record("D1", "amir") - assert r["cvps_bytes"] == 242000 and r["enrolled_at"] and r["deployed_at"] + assert r["ovsp_bytes"] == 242000 and r["enrolled_at"] and r["deployed_at"] # ---- orchestration (faux adb + faux bridge) ------------------------------ @@ -87,7 +87,7 @@ class _VoiceAdb(Adb): def shell(self, command, *, serial=None, timeout=None): if command.startswith("ls "): - return "\n".join(f"{d}.cvps" for d in self._deployed) + return "\n".join(f"{d}.ovsp" for d in self._deployed) if "rm -f" in command: self.deleted.append(command.split("rm -f ")[1].split(" &&")[0]) return "OK" @@ -105,8 +105,8 @@ class _FakeBridge: return {"text": "bonjour ceci est ma voix", "language": "fr", "model": "small"} def enroll(self, wav, text, out): - open(out, "wb").write(b"GGUF-CVPS-" + text.encode()[:5]) - return {"bytes": 242000, "feat_shape": [750, 80], "n_tokens": 375, "text_bytes": len(text)} + open(out, "wb").write(b"OVSP" + text.encode()[:5]) + return {"bytes": 242000, "t_ref": 364, "ref_seconds": 14.6, "text_bytes": len(text)} _WP = "/data/local/tmp/kazeia/models/../voix/voix/amir.wav" @@ -138,9 +138,9 @@ def test_enroll_deploy_pushes_and_optional_delete(tmp_path): # sans suppression r = vo.enroll_deploy(adb, s, _FakeBridge(), "D1", "amir", "ma voix", delete_source=False, now=NOW) - assert r["cvps_bytes"] == 242000 and r["wav_deleted"] is False + assert r["ovsp_bytes"] == 242000 and r["wav_deleted"] is False assert adb.pushed and adb.pushed[0][1] == \ - "/data/local/tmp/kazeia/models/cosyvoice/amir.cvps" + "/data/local/tmp/kazeia/models/omnivoice/voices/amir.ovsp" assert s.voice_record("D1", "amir")["deployed_at"] == NOW assert adb.deleted == [] # avec suppression