From 06b333f0f0e230f7e67ac37042647721d3d19b0c Mon Sep 17 00:00:00 2001 From: Kazeia Team Date: Sun, 7 Jun 2026 23:14:06 +0200 Subject: [PATCH] =?UTF-8?q?feat(speaker):=20rendre=20les=20Speakers=20GGUF?= =?UTF-8?q?=20s=C3=A9lectionnables=20in-app=20(Qwen3.5-4B,=20Qwen2.5-7B)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ModelRegistry devient backend-aware (PTE | GGUF) : enum Backend, ggufFile + ggufPath() sous MODELS_DIR, fileExists() par backend, et un type de template ChatTemplate (QWEN35_THINKOFF | PLAIN_CHATML). Ajout de deux entrées Speaker GGUF servies par le moteur lib : Qwen3.5-4B (q35-lmq4, le défaut lib jusqu'ici codé en dur, désormais visible) et Qwen2.5-7B (Q4_K_M, arch qwen2 dense). KazeiaService : le path lib route le GGUF du Speaker sélectionné (sp.ggufPath()) au lieu de q35-lmq4 en dur, et passe plainChatml selon le template. Pour les modèles sans thinking (Qwen2.5), EngineLlmAdapter construit un ChatML standard sans bloc via generateRaw (le natif l'injecterait sinon). Affordances de test/maintenance (pilotage adb sans UI) : - llm_text / llm_max : génération LLM pure loggée (tok/s, chars), hors pipeline TTS. - cfg_set : persiste engine+model+prompt via am --es (contourne content --bind qui casse sur ':'); cfg_sys=DEFAULT résout le prompt thérapeutique en interne. Provider /models : expose le chemin GGUF pour les entrées GGUF. Validé device : les deux GGUF chargent, FR cohérent, templates propres. Bench : Qwen3.5-4B ~13.5 tok/s, Qwen2.5-7B ~6.8 tok/s (CPU-only) ; qualité FR thérapeutique nettement en faveur du Qwen3.5-4B. Note : KazeiaService.kt et les 3 fichiers nouveaux portent aussi du travail antérieur non commité (migration moteur lib + app admin) ; ce commit en capture l'état courant en plus de la fonctionnalité Speaker GGUF. Co-Authored-By: Claude Opus 4.8 --- .../java/com/kazeia/config/ModelRegistry.kt | 135 ++ .../java/com/kazeia/llm/EngineLlmAdapter.kt | 69 ++ .../java/com/kazeia/service/KazeiaService.kt | 1102 +++++++++++++---- .../telemetry/KazeiaTelemetryProvider.kt | 576 +++++++++ 4 files changed, 1643 insertions(+), 239 deletions(-) create mode 100644 kazeia-android/app/src/main/java/com/kazeia/config/ModelRegistry.kt create mode 100644 kazeia-android/app/src/main/java/com/kazeia/llm/EngineLlmAdapter.kt create mode 100644 kazeia-android/app/src/main/java/com/kazeia/telemetry/KazeiaTelemetryProvider.kt diff --git a/kazeia-android/app/src/main/java/com/kazeia/config/ModelRegistry.kt b/kazeia-android/app/src/main/java/com/kazeia/config/ModelRegistry.kt new file mode 100644 index 0000000..92c11d9 --- /dev/null +++ b/kazeia-android/app/src/main/java/com/kazeia/config/ModelRegistry.kt @@ -0,0 +1,135 @@ +package com.kazeia.config + +import java.io.File + +/** + * Catalogue figé des modèles LLM qu'on sait charger via ExecuTorch QNN HTP. + * Filtrer par `availableOnDisk()` pour ne proposer dans l'UI admin que ceux + * réellement présents sur la tablette. + * + * L'admin app lit cette liste via le ContentProvider URI `/models` ; + * KazeiaService la consulte pour résoudre un `model_id` → (pte, tokenizer). + */ +object ModelRegistry { + + // Racine .pte LLM — résolue par KazeiaPaths (legacy /data/local/tmp/kazeia-et + // sur tablette dev, stockage externe app sur tablette de production). + private val ET_DIR: String get() = com.kazeia.KazeiaApplication.LLM_DIR + + // Racine GGUF (moteur kazeia-engine "lib"). Les GGUF (q35-lmq4, qwen2.5-7b…) + // vivent sous MODELS_DIR, pas LLM_DIR. + private val GGUF_DIR: String get() = com.kazeia.KazeiaApplication.MODELS_DIR + + enum class Role { SPEAKER, THINKER, BOTH } + + // PTE = runner ExecuTorch QNN HTP ; GGUF = moteur kazeia-engine "lib" (CPU). + enum class Backend { PTE, GGUF } + + // Template ChatML du modèle. QWEN35_THINKOFF injecte le bloc + // (thinking-off déterministe Qwen3.5). PLAIN_CHATML = ChatML standard sans + // (Qwen2.5 et autres modèles dense sans raisonnement). + enum class ChatTemplate { QWEN35_THINKOFF, PLAIN_CHATML } + + data class ModelInfo( + val id: String, + val displayName: String, + val pteFile: String, + val tokenizerFile: String, + val role: Role, + val sizeMb: Int, + val maxSeqLen: Int, + val notes: String = "", + val backend: Backend = Backend.PTE, + // Nom du fichier GGUF (sous GGUF_DIR) quand backend = GGUF. Le tokenizer + // est embarqué dans le GGUF, pas de fichier séparé. + val ggufFile: String = "", + val chatTemplate: ChatTemplate = ChatTemplate.QWEN35_THINKOFF + ) { + fun ptePath() = "$ET_DIR/$pteFile" + fun tokenizerPath() = "$ET_DIR/$tokenizerFile" + fun ggufPath() = "$GGUF_DIR/$ggufFile" + fun fileExists(): Boolean = when (backend) { + Backend.PTE -> File(ptePath()).exists() + Backend.GGUF -> File(ggufPath()).exists() + } + } + + val ALL: List = listOf( + ModelInfo( + id = "qwen3-4b-seq512", + displayName = "Qwen3-4B (seq=512)", + pteFile = "hybrid_llama_qnn_4b.pte", + tokenizerFile = "tokenizer.json", + role = Role.SPEAKER, + sizeMb = 3000, + maxSeqLen = 512, + notes = "Speaker historique, budget contexte serré" + ), + ModelInfo( + id = "qwen3-4b-seq1024", + displayName = "Qwen3-4B (seq=1024) ★", + pteFile = "hybrid_llama_qnn_4b_seq1024.pte", + tokenizerFile = "tokenizer.json", + role = Role.SPEAKER, + sizeMb = 3000, + maxSeqLen = 1024, + notes = "Speaker recommandé — production stable 2026-05-14" + ), + ModelInfo( + id = "qwen3.5-4b", + displayName = "Qwen3.5-4B (GGUF) ★", + pteFile = "", + tokenizerFile = "", + role = Role.SPEAKER, + sizeMb = 2380, + maxSeqLen = 4096, + notes = "Speaker GGUF par défaut du moteur lib (q35-lmq4). Thinking-off Qwen3.5.", + backend = Backend.GGUF, + ggufFile = "q35-lmq4.gguf", + chatTemplate = ChatTemplate.QWEN35_THINKOFF + ), + ModelInfo( + id = "qwen2.5-7b-instruct", + displayName = "Qwen2.5-7B Instruct (GGUF)", + pteFile = "", + tokenizerFile = "", + role = Role.SPEAKER, + sizeMb = 4683, + maxSeqLen = 4096, + notes = "Speaker GGUF dense (arch qwen2) via moteur lib CPU. Q4_K_M ~4.7 GB.", + backend = Backend.GGUF, + ggufFile = "Qwen2.5-7B-Instruct-Q4_K_M.gguf", + chatTemplate = ChatTemplate.PLAIN_CHATML + ), + ModelInfo( + id = "guard06b", + displayName = "Qwen3Guard 0.6B", + pteFile = "hybrid_llama_qnn_guard06b.pte", + tokenizerFile = "tokenizer_guard06b.json", + role = Role.THINKER, + sizeMb = 672, + maxSeqLen = 512, + notes = "Thinker léger, ION ~700 MB, cascade safe" + ), + ModelInfo( + id = "guard4b", + displayName = "Qwen3Guard 4B", + pteFile = "hybrid_llama_qnn_guard4b.pte", + tokenizerFile = "tokenizer_guard.json", + role = Role.THINKER, + sizeMb = 3000, + maxSeqLen = 512, + notes = "Thinker analytique. ⚠ Cascade Guard-4B + Speaker-4B = risque OOM." + ) + ) + + fun availableOnDisk(): List = ALL.filter { it.fileExists() } + + fun byId(id: String): ModelInfo? = ALL.firstOrNull { it.id == id } + + fun defaultSpeaker(): ModelInfo = byId("qwen3-4b-seq1024") + ?: ALL.first { it.role == Role.SPEAKER && it.fileExists() } + + fun defaultThinker(): ModelInfo = byId("guard06b") + ?: ALL.first { it.role == Role.THINKER && it.fileExists() } +} diff --git a/kazeia-android/app/src/main/java/com/kazeia/llm/EngineLlmAdapter.kt b/kazeia-android/app/src/main/java/com/kazeia/llm/EngineLlmAdapter.kt new file mode 100644 index 0000000..083de71 --- /dev/null +++ b/kazeia-android/app/src/main/java/com/kazeia/llm/EngineLlmAdapter.kt @@ -0,0 +1,69 @@ +package com.kazeia.llm + +import com.kazeia.core.GenerationResult +import com.kazeia.core.LlmConfig +import com.kazeia.core.LlmEngine +import com.kazeia.core.SamplingParams +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.withContext + +/** + * Adapter qui expose `com.kazeia.llm.EngineLlmEngine` (façade dist Kazeia-Engine, + * GGUF Qwen3.5/Qwen3 via libkazeia_engine.so + libllama fork ql + Hexagon) + * derrière l'interface prod `com.kazeia.core.LlmEngine`. Sert au swap réversible + * piloté par le flag config `llmEngine` ("prod" = ExecuTorch .pte | "lib" = engine GGUF). + * + * Pré-requis runtime : `libllama.so` fork ql + `libggml*.so` cohérents dans jniLibs. + * Si la chaîne libllama est encore en legacy (f86d), `System.loadLibrary("kazeia_engine")` + * jette UnsatisfiedLinkError au 1er accès à `EngineJni` → KazeiaService doit catch + * et tomber en repli sur le path prod. + * + * Modèle par défaut : Qwen3.5-4B `q35-lmq4.gguf` (upgrade vs Qwen3 dense en prod). + */ +class EngineLlmAdapter( + private val modelPath: String, + private val systemPrompt: String, + private val ctxLen: Int = 4096, + private val nThreads: Int = 6, + // true = ChatML standard sans (Qwen2.5…) via generateRaw ; + // false = bloc thinking-off Qwen3.5 injecté côté natif (generate). + private val plainChatml: Boolean = false, + private val onLog: ((String) -> Unit)? = null +) : LlmEngine { + private var engine: EngineLlmEngine? = null + + override suspend fun load(modelPath: String, config: LlmConfig) = withContext(Dispatchers.IO) { + if (engine != null) return@withContext + val t0 = System.currentTimeMillis() + engine = EngineLlmEngine(this@EngineLlmAdapter.modelPath, ctx = ctxLen, nThreads = nThreads) + onLog?.invoke("[ENGINE] load OK (${System.currentTimeMillis() - t0}ms) path=${this@EngineLlmAdapter.modelPath} t=${nThreads}") + } + + override fun isLoaded(): Boolean = engine != null + + override suspend fun generate( + prompt: String, + params: SamplingParams, + onToken: ((String) -> Boolean)? + ): GenerationResult = withContext(Dispatchers.IO) { + val e = engine ?: throw IllegalStateException("EngineLlmAdapter not loaded") + val t0 = System.currentTimeMillis() + val out = if (plainChatml) { + // ChatML standard, sans bloc (le natif generate l'injecterait). + val p = buildString { + append("<|im_start|>system\n").append(systemPrompt).append("<|im_end|>\n") + append("<|im_start|>user\n").append(prompt).append("<|im_end|>\n") + append("<|im_start|>assistant\n") + } + e.generateRaw(p, params.maxNewTokens).trim() + } else { + e.generate(systemPrompt, prompt, params.maxNewTokens).trim() + } + val ms = System.currentTimeMillis() - t0 + onToken?.invoke(out) + val n = out.split(Regex("\\s+")).size + GenerationResult(out, n, ms, if (ms > 0) n * 1000f / ms else 0f) + } + + override fun release() { engine?.release(); engine = null } +} diff --git a/kazeia-android/app/src/main/java/com/kazeia/service/KazeiaService.kt b/kazeia-android/app/src/main/java/com/kazeia/service/KazeiaService.kt index f39e3e7..0bee524 100644 --- a/kazeia-android/app/src/main/java/com/kazeia/service/KazeiaService.kt +++ b/kazeia-android/app/src/main/java/com/kazeia/service/KazeiaService.kt @@ -15,7 +15,6 @@ import androidx.core.app.NotificationCompat import androidx.core.content.ContextCompat import com.kazeia.KazeiaApplication import com.kazeia.R -import com.kazeia.audio.AudioCaptureManager import com.kazeia.audio.AudioPlaybackManager import com.kazeia.conversation.ConversationManager import com.kazeia.conversation.PromptBuilder @@ -23,35 +22,118 @@ import com.kazeia.conversation.StoppingCriteria import com.kazeia.core.* import com.kazeia.llm.ExecuTorchLlmEngine import com.kazeia.llm.GenieLlmEngine -import com.kazeia.stt.AndroidSttEngine +import com.kazeia.profiles.ConversationDao +import com.kazeia.profiles.ConversationDb +import com.kazeia.profiles.ProfileStore +import com.kazeia.profiles.Role +import com.kazeia.profiles.Turn import com.kazeia.stt.WhisperHybridEngine -import com.kazeia.stt.WhisperSttEngine -import com.kazeia.tts.AndroidTtsEngine -import com.kazeia.tts.ChatterboxTtsEngine import com.kazeia.ui.ChatActivity -import com.kazeia.vad.SileroVadEngine import kotlinx.coroutines.* import kotlinx.coroutines.flow.MutableStateFlow import kotlinx.coroutines.flow.StateFlow +import kotlinx.coroutines.sync.withLock class KazeiaService : Service() { companion object { private const val TAG = "KazeiaService" private const val NOTIFICATION_ID = 1 + // Diagnostic mode: when true, TTS is bypassed entirely. The chat + // bubble updates token-by-token as the LLM emits, so we can see + // the raw LLM speed (tok/s) without any TTS coupling. Flip back + // to false to re-enable the full STT→LLM→TTS pipeline. + const val LLM_ONLY_MODE = false + + // Prompt système Thinker (Qwen3Guard-Gen-4B) — verbatim PoC validé. + private const val SYS_THINKER = """Tu es un psychologue clinicien experimente qui ecoute attentivement ton patient. Avant de repondre, tu prepares une analyse silencieuse pour orienter ta reponse. Produis UNIQUEMENT le bloc clinique au format strict suivant : +- emotion : +- besoin : +- approche : +- axe : +Pas d'introduction, pas d'explication. Juste les 4 lignes en francais. /no_think""" } private lateinit var llm: LlmEngine - private lateinit var stt: SttEngine - private lateinit var tts: TtsEngine - private lateinit var vad: VadEngine + // Cascade Option E v2 : 2 ExecuTorchLlmEngine cohabitants in-process. + // - llm = Speaker Qwen3-4B (.pte ~3.3 GB ION, ~4.4 GB total) + // - thinkerEngine = Guard-0.6B (.pte ~700 MB ION) lazy-loaded au 1er PTT + // Total ION ~5.1 GB, marge OK pour TTS pic. QnnBackendUnifiedRegistry + // singleton partage le QnnBackendBundle (handle fastrpc DSP) entre les 2 + // contextes — pas de collision DSP, pas de subprocess. Cf + // project_kazeia_cascade_option_e dans la mémoire persistante. + private var thinkerEngine: com.kazeia.llm.ExecuTorchLlmEngine? = null + private val thinkerLock = Object() + + // Config runtime (cascade, prompts, modèles) pilotée par l'admin app via + // ContentProvider + ConfigStore. Snapshot mis à jour par le listener + // ConfigStore et par le BroadcastReceiver `com.kazeia.action.RELOAD_CONFIG`. + @Volatile + private var runtimeConfig: com.kazeia.config.ConfigStore.RuntimeConfig = + com.kazeia.config.ConfigStore.defaultConfig() + private var configReloadReceiver: android.content.BroadcastReceiver? = null + private lateinit var stt: SttEngine + + // Cycle de vie Whisper STT — lazy load + swap intelligent. + // Whisper consomme ~600 MB ION via QAIRT context binaries. Sur une tablette + // tendue mémoire, on le décharge pendant les phases LLM/TTS où il ne sert + // pas, et on le recharge en arrière-plan pendant le TTS playback (3-5 s + // disponibles) pour qu'il soit prêt au prochain PTT — zéro impact UX. + private var sttModelPath: String? = null + private val sttMutex = kotlinx.coroutines.sync.Mutex() + + private suspend fun ensureSttLoaded() = sttMutex.withLock { + if (stt.isLoaded()) return@withLock + val path = sttModelPath ?: return@withLock + val t0 = System.currentTimeMillis() + log("[STT] ensure load…") + stt.load(path) + log("[STT] loaded in ${System.currentTimeMillis() - t0}ms") + } + + private suspend fun releaseStt() = sttMutex.withLock { + if (!stt.isLoaded()) return@withLock + log("[STT] release (free ION ~600 MB)") + stt.release() + } + + // Cycle de vie Speaker LLM — release+reload entre tours. + // resetContext() est silencieusement ignoré sur certains .pte Qwen3 + // sharding=2 → cur_pos accumule → crash prefill après 3-5 tours. + // Solution robuste : reload complet pendant TTS playback (3-5s, + // caché derrière l'audio). Cost ~2-3s par turn, mais 0 latence user + // si le TTS playback dure plus long que le reload. + private val llmMutex = kotlinx.coroutines.sync.Mutex() + @Volatile private var llmReloadInFlight = false + + private suspend fun ensureLlmLoaded() = llmMutex.withLock { + if (llm.isLoaded()) return@withLock + log("[LLM] ensure load (post-reload wait)…") + val t0 = System.currentTimeMillis() + llm.load("", com.kazeia.core.LlmConfig()) + log("[LLM] loaded in ${System.currentTimeMillis() - t0}ms") + } + + private suspend fun reloadLlmAfterTurn() = llmMutex.withLock { + val ll = llm as? com.kazeia.llm.ExecuTorchLlmEngine ?: return@withLock + if (!ll.isLoaded()) return@withLock + val t0 = System.currentTimeMillis() + log("[LLM] reload pour reset cur_pos (état propre prochain tour)…") + ll.reloadAfterTurn() + log("[LLM] reloaded in ${System.currentTimeMillis() - t0}ms") + } + private lateinit var tts: TtsEngine - private lateinit var audioCapture: AudioCaptureManager private lateinit var audioPlayback: AudioPlaybackManager private lateinit var conversationManager: ConversationManager private lateinit var promptBuilder: PromptBuilder private lateinit var stoppingCriteria: StoppingCriteria + + // Per-interaction timing tracer. One instance, reset at every begin(). + // All stage timings for a single user turn (STT+LLM+TTS+playback) end + // up in a single grep-friendly log section tagged "[PIPELINE]". + private val metrics = com.kazeia.core.PipelineMetrics(TAG) private lateinit var voiceCommands: com.kazeia.conversation.VoiceCommandProcessor // New modular pipeline @@ -92,6 +174,8 @@ class KazeiaService : Service() { sealed class VisualizerSignal { object Idle : VisualizerSignal() data class Listening(val micRms: Float) : VisualizerSignal() + /** L'orbe pulse une couleur smooth (LLM réfléchit). */ + object Thinking : VisualizerSignal() data class Speaking( val rmsEnvelope: FloatArray, val spectrogram: Array, @@ -119,6 +203,60 @@ class KazeiaService : Service() { private val _loadingState = MutableStateFlow(LoadingState()) val loadingState: StateFlow = _loadingState + /** + * Cascade E v2 : crée la 2ᵉ ExecuTorchLlmEngine (Thinker = Guard-0.6B) dans + * le MÊME process app. Lazy au 1er PTT (~1-2 s de load 0.6B). Le + * QnnBackendUnifiedRegistry singleton réutilise le QnnBackendBundle du + * Speaker → un seul handle fastrpc DSP, 2 contextes/graphes coexistent. + */ + private fun ensureThinkerEngine() { + synchronized(thinkerLock) { + if (thinkerEngine != null && thinkerEngine!!.isLoaded()) return + val cfg = runtimeConfig + val info = com.kazeia.config.ModelRegistry.byId(cfg.thinker.modelId) + ?: com.kazeia.config.ModelRegistry.defaultThinker() + log("[THINKER] loading ${info.displayName} engine in-process (lazy first PTT)…") + val tStart = System.currentTimeMillis() + val memBefore = readMemAvailableMb() + val e = com.kazeia.llm.ExecuTorchLlmEngine( + context = this@KazeiaService, + onLog = { msg -> log("[THINKER] $msg") }, + modelPath = info.ptePath(), + tokenizerPath = info.tokenizerPath(), + systemPrompt = cfg.thinker.systemPrompt, + temperature = cfg.thinker.temperature, + displayName = "${info.displayName} Thinker" + ) + try { + runBlocking { e.load("", com.kazeia.core.LlmConfig()) } + if (!e.isLoaded()) throw IllegalStateException("Thinker load() returned but isLoaded=false") + thinkerEngine = e + val dt = System.currentTimeMillis() - tStart + val memAfter = readMemAvailableMb() + log("[THINKER] engine READY in ${dt}ms (Δmem ${memBefore - memAfter}MB)") + } catch (e: Exception) { + log("[THINKER] engine load failed: ${e.javaClass.simpleName}: ${e.message} — cascade off") + thinkerEngine = null + } + } + } + + private fun stopThinkerEngine() { + synchronized(thinkerLock) { + thinkerEngine?.let { try { it.release() } catch (_: Exception) {} } + thinkerEngine = null + } + } + + /** Lecture rapide MemAvailable pour traçage cascade memory footprint. */ + private fun readMemAvailableMb(): Long { + return try { + java.io.File("/proc/meminfo").readLines() + .firstOrNull { it.startsWith("MemAvailable:") } + ?.trim()?.split(Regex("\\s+"))?.get(1)?.toLong()?.div(1024) ?: 0L + } catch (_: Exception) { 0L } + } + private fun log(msg: String) { Log.i(TAG, msg) val time = java.text.SimpleDateFormat("HH:mm:ss.SSS", java.util.Locale.FRANCE).format(java.util.Date()) @@ -157,10 +295,31 @@ class KazeiaService : Service() { // inter-token pause so we can observe the "sentence N plays // while sentence N+1 is generating" behaviour end-to-end. Used // in debug builds where the LLM is disabled (echo mode). - log("Stream LLM mock: '${text.take(60)}${if (text.length>60) "..." else ""}'") + // Optional companion extra `tts_voice` selects voice prefix file + // before generation (used by chantier ggml-cpu reference dump). + val voiceOverride = intent.getStringExtra("tts_voice") + log("Stream LLM mock: '${text.take(60)}${if (text.length>60) "..." else ""}'${voiceOverride?.let { " voice=$it" } ?: ""}") serviceScope.launch { try { - val qwenTts = tts as? com.kazeia.tts.Qwen3TtsEngine ?: return@launch + val qwenTts = tts as? com.kazeia.tts.Qwen3TtsEngine + if (qwenTts == null) { + // Backend non-Qwen3 (ex. KazeiaTtsEngine lib) : pas de + // streaming par phrase, on synthétise le texte complet + // via l'interface et on joue. Sert au test E2E du wiring. + if (!::tts.isInitialized) { log("Stream mock: TTS not ready"); return@launch } + _pipelineState.value = PipelineState.Speaking + val tS = System.currentTimeMillis() + tts.synthesizeAndPlay(text, "fr", + onComplete = { + log("Stream mock (lib) done at ${System.currentTimeMillis() - tS}ms") + _pipelineState.value = PipelineState.Idle + }) + return@launch + } + if (voiceOverride != null) { + qwenTts.setVoice(voiceOverride) + log("Voice set to: $voiceOverride") + } qwenTts.startStreamingSession() val tStart = System.currentTimeMillis() var firstSentenceLogged = false @@ -400,9 +559,11 @@ class KazeiaService : Service() { intent?.getStringExtra("tts_text")?.let { text -> val savePath = intent.getStringExtra("save_wav") if (savePath != null) { - log("TTS generate WAV: '$text' → $savePath") + val voiceOverride = intent.getStringExtra("tts_voice") + log("TTS generate WAV: '$text' → $savePath${voiceOverride?.let { " voice=$it" } ?: ""}") serviceScope.launch { try { + if (voiceOverride != null) { setVoice(voiceOverride); kotlinx.coroutines.delay(50) } log("TTS synthesize starting for '$text'...") val result = tts.synthesize(text, "fr") log("TTS synthesize done: ${result.audioData.size} samples, ${result.sampleRate}Hz, ${result.durationMs}ms") @@ -442,11 +603,52 @@ class KazeiaService : Service() { processTextInput(text) } } + intent?.getStringExtra("llm_text")?.let { prompt -> + // Affordance de test LLM pur (sans TTS) : génère et logge la réponse. + // Sert à valider un Speaker (ex. Qwen2.5-7B GGUF) hors pipeline audio. + val maxTok = intent.getIntExtra("llm_max", 120) + log("LLM test: '${prompt.take(80)}' (max=$maxTok)") + serviceScope.launch { + if (!::llm.isInitialized || !llm.isLoaded()) { log("LLM test: not ready"); return@launch } + try { + val t0 = System.currentTimeMillis() + val r = llm.generate(prompt, com.kazeia.core.SamplingParams(maxNewTokens = maxTok, temperature = 0.7f), null) + val ms = System.currentTimeMillis() - t0 + log("LLM test OUT (${r.tokenCount} word/tok, ${ms}ms, cap=$maxTok, ${r.text.length} chars): ${r.text}") + } catch (e: Exception) { log("LLM test error: ${e.message}") } + } + } + if (intent?.hasExtra("cfg_set") == true) { + // Maintenance/bench : persiste engine+model+prompt sans passer par le + // ContentProvider CLI (qui casse sur les valeurs contenant ':'). Les + // extras passent par `am --es` (pas de split sur ':'). cfg_sys="DEFAULT" + // résout le prompt thérapeutique par défaut en interne (jamais transporté). + val store = com.kazeia.config.ConfigStore.get(this) + val cur = store.current() + val eng = intent.getStringExtra("cfg_engine") ?: cur.llmEngine + val mid = intent.getStringExtra("cfg_model") ?: cur.speaker.modelId + val sysRaw = intent.getStringExtra("cfg_sys") + val sys = when (sysRaw) { + null -> cur.speaker.systemPrompt + "DEFAULT" -> com.kazeia.config.ConfigStore.DEFAULT_SPEAKER_PROMPT + else -> sysRaw + } + val newCfg = cur.copy(llmEngine = eng, speaker = cur.speaker.copy(modelId = mid, systemPrompt = sys)) + store.save(newCfg) + log("cfg_set: engine=$eng model=$mid sysLen=${sys.length}") + } return START_STICKY } override fun onCreate() { super.onCreate() + // Charge la config runtime au plus tôt + s'abonne aux changements + // (admin app → ContentProvider.update → ConfigStore.save → listener) + val store = com.kazeia.config.ConfigStore.get(this) + runtimeConfig = store.current() + store.addListener { new -> handleConfigChange(new) } + registerConfigReloadReceiver() + val hasMicPermission = ContextCompat.checkSelfPermission( this, Manifest.permission.RECORD_AUDIO ) == PackageManager.PERMISSION_GRANTED @@ -484,121 +686,107 @@ class KazeiaService : Service() { // Initialize engines val modelsDir = KazeiaApplication.MODELS_DIR - _loadingState.value = LoadingState(5, "VAD…") - vad = SileroVadEngine() - vad.load(this@KazeiaService) - log("VAD loaded") - _loadingState.value = LoadingState(10, "VAD OK") + // VAD opérationnelle = RMS énergie inline dans startContinuousListening(). + // SileroVadEngine + AudioCaptureManager (legacy code mort) supprimés + // 2026-05-02 — cf /opt/Kazeia/kazeia-stt-perf-validated/. val nativeLibDir = applicationInfo.nativeLibraryDir - // TTS: try Qwen3-TTS (NPU Hexagon), fallback to Android TTS - _loadingState.value = LoadingState(15, "TTS Qwen3…") - try { + // TTS legacy gated par ttsEngine="prod". Si "lib" ou autre : + // skip l'init Qwen3TtsEngine (incompat ABI avec libllama fork-ql + // déployé pour LLM engine). Step B câblera l'adapter lib. + val ttsBackend = runtimeConfig.ttsEngine + if (ttsBackend == "lib") { + _loadingState.value = LoadingState(15, "TTS engine…") + val libTts = com.kazeia.tts.KazeiaTtsEngine(this@KazeiaService) { msg -> log("[TTS] $msg") } + libTts.load() + if (!libTts.isLoaded()) throw IllegalStateException("KazeiaTtsEngine load failed") + tts = libTts + log("TTS: KazeiaTtsEngine loaded (libkazeia_tts CPU-only)") + } else { + _loadingState.value = LoadingState(15, "TTS Qwen3…") val qwenTts = com.kazeia.tts.Qwen3TtsEngine(nativeLibDir, this@KazeiaService) { msg -> log("[TTS] $msg") } qwenTts.load("$modelsDir/qwen3-tts-npu") - if (qwenTts.isLoaded()) { - tts = qwenTts - log("TTS: Qwen3-TTS loaded (Hexagon NPU)") - } else { - tts = AndroidTtsEngine(this@KazeiaService) - tts.load(voiceId = "kazeia_fr") - log("TTS: Qwen3 failed, using Android TTS") - } - } catch (e: Exception) { - log("TTS: Qwen3 error: ${e.message}, using Android TTS") - tts = AndroidTtsEngine(this@KazeiaService) - tts.load(voiceId = "kazeia_fr") + if (!qwenTts.isLoaded()) throw IllegalStateException("Qwen3-TTS load failed (no fallback in frozen stack)") + tts = qwenTts + log("TTS: Qwen3-TTS loaded (Hexagon NPU + ggml-cpu)") } _loadingState.value = LoadingState(20, "TTS OK") - // STT: try NPU first, fallback to CPU - _loadingState.value = LoadingState(25, "STT Whisper…") - var sttLoaded = false + // STT = Whisper-Small NPU (HfWhisper Qualcomm) — voie validée + // (cf /opt/Kazeia/kazeia-stt-perf-validated/). Pas de fallback. + // Lazy-load + swap intelligent : on construit le moteur ici mais + // on diffère le load() au 1ᵉʳ PTT pour économiser ~600 MB ION + // au boot. Hooks sur _pipelineState gèrent reload anticipé + // pendant TTS (caché derrière le playback) et release pendant + // les phases LLM/TTS où Whisper ne sert pas. + _loadingState.value = LoadingState(25, "STT Whisper (lazy)…") + // Backend STT pilotable via ConfigStore.sttEngine. Default "prod" = + // WhisperHybridEngine gelé. "lib" = KazeiaSttAdapter (libkazeia_stt.so). + val sttBackend = runtimeConfig.sttEngine + val npuStt: com.kazeia.core.SttEngine = when (sttBackend) { + "lib" -> com.kazeia.stt.KazeiaSttAdapter( + "$modelsDir/whisper-small-sm8750" + ) { msg -> log("[STT] $msg") } + else -> WhisperHybridEngine(nativeLibDir) { msg -> log("[STT] $msg") } + } + sttModelPath = "$modelsDir/whisper-small-sm8750" + stt = npuStt + log("Whisper créé backend='$sttBackend' (lazy load au 1er PTT)") + addMessage(ChatMessage( + role = ChatMessage.Role.SYSTEM, + text = "STT: Whisper-Small NPU" + )) - // Try 1: Whisper NPU (Qualcomm ONNX QNN EP) + // LLM = ExecuTorch QNN. Modèle choisi dans la config runtime + // (admin app peut le changer à chaud). Fallback sur le + // Speaker par défaut du registre si l'id config est inconnu. + // LLM = Kazeia-Engine GGUF (llama.cpp upstream + Hexagon), Speaker + // Qwen3.5-9B prefill NPU / decode CPU. Plus de .pte (-30% RAM). + // Backend LLM pilotable via ConfigStore.llmEngine. Default "prod" = + // ExecuTorch .pte (Qwen3 dense). "lib" = EngineLlmAdapter (libkazeia_engine, + // GGUF Qwen3.5). Fallback gracieux sur prod si la lib échoue au load + // (libllama fork ql encore absente du jniLibs tant que swap coordonné non fait). + val llmBackend = runtimeConfig.llmEngine + _loadingState.value = LoadingState(50, "LLM backend=$llmBackend…") + val sp = com.kazeia.config.ModelRegistry.byId(runtimeConfig.speaker.modelId) ?: com.kazeia.config.ModelRegistry.defaultSpeaker() + val llmLib: com.kazeia.core.LlmEngine? = if (llmBackend == "lib") try { + // Si le Speaker sélectionné est un GGUF du registre (ex. Qwen2.5-7B), + // on le charge ; sinon fallback sur le GGUF moteur par défaut (q35-lmq4). + val gguf = if (sp.backend == com.kazeia.config.ModelRegistry.Backend.GGUF && sp.fileExists()) + sp.ggufPath() + else + "${KazeiaApplication.MODELS_DIR}/q35-lmq4.gguf" + val plain = sp.chatTemplate == com.kazeia.config.ModelRegistry.ChatTemplate.PLAIN_CHATML + log("[LLM] lib GGUF=$gguf (speaker=${sp.id}, plainChatml=$plain)") + com.kazeia.llm.EngineLlmAdapter(gguf, runtimeConfig.speaker.systemPrompt, + plainChatml = plain, onLog = { msg -> log(msg) }) + } catch (e: Throwable) { + log("[LLM] lib instantiation failed (${e.message}) — fallback .pte") + null + } else null + llm = llmLib ?: ExecuTorchLlmEngine(this@KazeiaService, { msg -> log(msg) }, + sp.ptePath(), sp.tokenizerPath(), runtimeConfig.speaker.systemPrompt, + runtimeConfig.speaker.temperature, sp.displayName) try { - val npuStt = WhisperHybridEngine(nativeLibDir) { msg -> log(msg) } - npuStt.load("$modelsDir/whisper-small-sm8750") - if (npuStt.isLoaded()) { - stt = npuStt - sttLoaded = true - log("Using Whisper ONNX") - addMessage(ChatMessage( - role = ChatMessage.Role.SYSTEM, - text = "STT: Whisper ONNX loaded" - )) - } + llm.load("", com.kazeia.core.LlmConfig()) + log("LLM créé backend='${if (llmLib != null) "lib" else "prod"}'") } catch (e: Exception) { - log("WARN: Whisper NPU failed: ${e.message}") - } - - // Try 2: whisper.cpp CPU - if (!sttLoaded) { - try { - val cpuStt = WhisperSttEngine() - cpuStt.load("$modelsDir/whisper-base") - if (cpuStt.isLoaded()) { - stt = cpuStt - sttLoaded = true - log("Using whisper.cpp (CPU)") - addMessage(ChatMessage( - role = ChatMessage.Role.SYSTEM, - text = "STT: Whisper-Base (CPU)" - )) - } - } catch (e: Exception) { - log("WARN: whisper.cpp failed: ${e.message}") - } - } - - if (!sttLoaded) { - stt = AndroidSttEngine(this@KazeiaService) - stt.load() - addMessage(ChatMessage( - role = ChatMessage.Role.SYSTEM, - text = "STT: Android natif" - )) - } - - // LLM backend selection : llama.cpp Qwen3.5-4B (new default for - // therapeutic conversation, measured ~11 tok/s on OPD2415 +57 % - // vs ExecuTorch path) or legacy ExecuTorch Qwen3-4B QNN. - // Override at runtime via `adb shell setprop debug.kazeia.llm_backend llamacpp`. - val backend = try { - val spClass = Class.forName("android.os.SystemProperties") - spClass.getMethod("get", String::class.java, String::class.java) - .invoke(null, "debug.kazeia.llm_backend", KazeiaApplication.LLM_BACKEND) as String - } catch (_: Throwable) { KazeiaApplication.LLM_BACKEND } - - _loadingState.value = LoadingState(50, "LLM $backend…") - when (backend) { - "llamacpp" -> { - val engine = com.kazeia.llm.LlamaCppLlmEngine() - llm = engine - try { - val gguf = "${KazeiaApplication.MODELS_DIR}/${KazeiaApplication.LLAMACPP_MODEL_FILENAME}" - engine.load(gguf, com.kazeia.core.LlmConfig(maxContextLength = 4096)) - log("LLM llama.cpp loaded: ${KazeiaApplication.LLAMACPP_MODEL_FILENAME}") - } catch (e: Exception) { - log("llama.cpp load failed: ${e.message} — falling back to echo mode") - } - } - else -> { - // "executorch" (default) — Qwen3-4B QNN on Hexagon V79. - llm = ExecuTorchLlmEngine(this@KazeiaService) { msg -> log(msg) } - try { - llm.load("${KazeiaApplication.MODELS_DIR}/qwen3-4b", com.kazeia.core.LlmConfig()) - } catch (e: Exception) { - log("LLM load failed: ${e.message} — falling back to echo mode") - } + log("LLM load failed (${e.message}) — fallback echo") + if (llmLib != null) { + // lib loadLibrary("kazeia_engine") a échoué (libllama fork ql absente) ; + // on retombe sur .pte pour ne pas perdre l'app + log("[LLM] retry on prod .pte") + llm = ExecuTorchLlmEngine(this@KazeiaService, { msg -> log(msg) }, + sp.ptePath(), sp.tokenizerPath(), runtimeConfig.speaker.systemPrompt, + runtimeConfig.speaker.temperature, sp.displayName) + try { llm.load("", com.kazeia.core.LlmConfig()) } catch (_: Exception) {} } } _loadingState.value = LoadingState(80, "Audio…") // Audio audioPlayback = AudioPlaybackManager() - audioCapture = AudioCaptureManager() // Conversation logic promptBuilder = PromptBuilder() @@ -635,6 +823,35 @@ class KazeiaService : Service() { _loadingState.value = LoadingState(100, "Prêt", done = true) log("All components initialized") log("Pipeline: STT(${stt::class.simpleName}) → [${pipeline.getProcessors().joinToString(" → ") { it.name }}] → TTS(${tts::class.simpleName})") + + // Cycle de vie STT (option A+B) : decharge Whisper pendant LLM, + // recharge anticipée pendant TTS playback (caché derrière 3-5s + // d'audio). Au prochain PTT, Whisper est déjà prêt → 0 latence. + serviceScope.launch { + var lastTriggered: String? = null + _pipelineState.collect { st -> + val key = when (st) { + is PipelineState.Thinking, is PipelineState.TokenGenerated -> "release" + is PipelineState.Speaking -> "ensure" + else -> null + } + if (key != null && key != lastTriggered) { + lastTriggered = key + launch { + if (key == "release") releaseStt() else ensureSttLoaded() + } + // Speaking = TTS playback démarre → on a 3-5s pour + // recharger le Speaker LLM en arrière-plan, garantir + // un cur_pos propre pour le prochain tour. + // 2026-05-13 : reloadLlmAfterTurn() désactivé. + // Le release+reload échoue avec + // "Failed to load llm runner: [1]" sur seq=1024 .pte + // (probable bug singleton QnnBackendUnifiedRegistry). + // On compte sur resetContext() en début de generate() + // pour cur_pos_=0 (qui appelle Runner::reset côté C++). + } + } + } } catch (e: Exception) { log("ERROR: Initialization error: ${e.message}") _loadingState.value = LoadingState(100, "Erreur: ${e.message}", done = true) @@ -648,14 +865,57 @@ class KazeiaService : Service() { } } - fun setVoice(voicePath: String) { - val chatterbox = tts as? ChatterboxTtsEngine - if (chatterbox != null) { - chatterbox.setVoice(voicePath) - log("Voice set to: $voicePath") - return + // ─── Historique conversation (2c.4) ──────────────────────────────────── + // DAO SQLCipher lazy : la base n'est ouverte qu'au 1er tour loggé. En + // mode invité (pas de profil actif) rien n'est ouvert ni écrit. + private val conversationDao: ConversationDao by lazy { + ConversationDao(ConversationDb.get(applicationContext)) + } + // Nom de la voix courante (basename du .wav), tracké pour le log de tour. + @Volatile private var currentVoiceName: String = "damien" + + /** + * Persiste un tour de conversation pour le profil actif. No-op en mode + * invité (pas de profil → pas de session). L'écriture SQLCipher part sur + * un coroutine IO pour ne pas bloquer le pipeline. + */ + private fun logConversationTurn( + role: Role, + text: String, + ttftMs: Long? = null, + totalMs: Long? = null + ) { + val trimmed = text.trim() + if (trimmed.isEmpty()) return + val store = ProfileStore.get(applicationContext) + val profile = store.activeProfile() ?: return // mode invité + val sessionId = store.currentSessionId ?: return + serviceScope.launch(Dispatchers.IO) { + try { + conversationDao.logTurn( + Turn( + profileId = profile.id, + sessionId = sessionId, + timestamp = System.currentTimeMillis(), + role = role, + text = trimmed, + ttftMs = ttftMs, + totalMs = totalMs, + voiceUsed = if (role == Role.KAZEIA) currentVoiceName else null, + modelUsed = if (role == Role.KAZEIA) runtimeConfig.speaker.modelId else null + ) + ) + } catch (e: Exception) { + log("[HISTORY] log turn failed: ${e.message}") + } } + } + + fun setVoice(voicePath: String) { + // Mémorise le nom de voix (basename sans extension) pour l'historique. + currentVoiceName = voicePath.substringAfterLast('/').substringBeforeLast('.') val qwen = tts as? com.kazeia.tts.Qwen3TtsEngine + val libTts = tts as? com.kazeia.tts.KazeiaTtsEngine if (qwen != null) { // Hot-swap prefix/suffix embeddings — no model reload. Takes // effect from the NEXT synthesized segment (current in-flight @@ -663,6 +923,10 @@ class KazeiaService : Service() { // are already in its closure). qwen.setVoice(voicePath) log("Voice set to: $voicePath") + } else if (libTts != null) { + // Engine lib : le natif dérive le x_vector du WAV (speaker encoder). + libTts.setVoice(voicePath) + log("Voice set to: $voicePath (lib)") } } @@ -697,7 +961,7 @@ class KazeiaService : Service() { } private fun startListening() { - log("startListening: stt=${stt::class.simpleName}, vad=${vad.isLoaded()}") + log("startListening: stt=${stt::class.simpleName} (VAD = RMS énergie inline)") _isListening.value = true _pipelineState.value = PipelineState.Listening @@ -827,9 +1091,22 @@ class KazeiaService : Service() { @android.annotation.SuppressLint("MissingPermission") private fun startContinuousListening() { serviceScope.launch(Dispatchers.IO) { + // Ensure Whisper STT loaded (lazy 1ᵉʳ PTT, ou re-load après swap). + // Bloquant côté coroutine pour garantir que stt.transcribe() ne soit + // pas appelé sur un moteur déchargé en plein milieu de la boucle. + ensureSttLoaded() + // Speaker LLM doit aussi être prêt — au cas où la phase TTS du tour + // précédent ait été plus courte que le reload (rare mais possible). + ensureLlmLoaded() + val sampleRate = 16000 val frameSize = 1600 // 100ms frames for VAD - val silenceThreshold = 150 // RMS threshold for speech detection + val silenceThreshold = 150 // RMS threshold for speech detection. + // Mesures terrain : bruit ambiant 86-91 RMS, click PTT 116 RMS, + // parole douce 500+ RMS, parole forte 700-3000 RMS. 80 était trop + // bas → bruit ambiant > seuil → silence jamais détecté → + // segment ne se finalise pas. 150 = au-dessus du bruit + click, + // bien en-dessous de toute parole humaine. val speechMinFrames = 3 // min 300ms of speech to trigger val silenceEndFrames = 8 // 800ms of silence to end segment val maxSegmentSamples = sampleRate * 30 // max 30s per segment @@ -880,12 +1157,19 @@ class KazeiaService : Service() { var silenceFrameCount = 0 var isSpeechActive = false var frameCount = 0 + // Warmup window : AudioRecord retourne du bruit non stabilisé dans les + // ~500 ms après start (click PTT, mic init). Sans warmup, le VAD + // déclenche sur ce bruit, Whisper hallucine "Bonsoir" sur silence, + // et la 1ère phrase utilisateur arrive après plusieurs faux départs. + val listenStartMs = System.currentTimeMillis() + val warmupMs = 500L var speechStartTime = 0L val streamingIntervalSamples = sampleRate * 2 // send to Whisper every 2s of speech var lastStreamedSamples = 0 var streamingResults = mutableListOf() var wasMuted = false + var lastBusyMs = 0L while (_isListening.value) { val read = recorder.read(frame, 0, frameSize) @@ -893,21 +1177,35 @@ class KazeiaService : Service() { val now = System.currentTimeMillis() - val isTtsSpeaking = _pipelineState.value is PipelineState.Speaking - val isTranscribing = _pipelineState.value is PipelineState.Transcribing + // Anti-feedback acoustique : le micro capte la sortie HP et + // Whisper transcrit ces échos en gibberish ("page beg, beg") + // → boucle infinie. On mute le mic dès que la machine + // travaille (Thinking, Speaking, Transcribing, etc.). Seuls + // Listening / Idle / SpeechDetected acceptent les frames. + val st = _pipelineState.value + val machineBusy = st is PipelineState.Speaking + || st is PipelineState.Thinking + || st is PipelineState.Transcribing + || st is PipelineState.Transcribed + || st is PipelineState.TokenGenerated + val nowMs = System.currentTimeMillis() + // Tail buffer : 800 ms après la fin du Speaking pour + // absorber la reverb / queue MediaPlayer qui peut continuer + // à émettre du son alors que pipelineState est passé Idle. + if (machineBusy) { + lastBusyMs = nowMs + } + val inTailBuffer = (nowMs - lastBusyMs) < 800 - // Mute during TTS (mic button to interrupt) - if (isTtsSpeaking) { + if (machineBusy || inTailBuffer) { if (isSpeechActive) { speechBuffer.clear(); speechFrameCount = 0; silenceFrameCount = 0 isSpeechActive = false; lastStreamedSamples = 0; streamingResults.clear() } + if (!wasMuted) { log("Mic muted (state=${st::class.simpleName} tail=${inTailBuffer})"); wasMuted = true } continue } - // Mute during transcription - if (isTranscribing) continue - if (wasMuted) { log("Mic unmuted"); wasMuted = false } // Compute RMS energy @@ -917,11 +1215,16 @@ class KazeiaService : Service() { // Drive the visualizer orb. Normalize with the same // sqrt squashing used for TTS so loud peaks don't - // saturate and quiet speech is still visible. The - // visualizer stays in Listening mode; it will swap - // to Speaking or Idle when pipelineState moves on. + // saturate and quiet speech is still visible. + // IMPORTANT : ne PAS écraser un signal Thinking/Speaking en + // cours (l'app est entre 2 tours, le LLM/TTS est en train + // de générer la réponse, l'orbe doit rester en cycle de + // couleurs/waveform et pas repasser en Listening). val rmsNorm = kotlin.math.sqrt((rms / 6000f).coerceIn(0f, 1f)) - _visualizerSignal.value = VisualizerSignal.Listening(rmsNorm) + val cur = _visualizerSignal.value + if (cur is VisualizerSignal.Idle || cur is VisualizerSignal.Listening) { + _visualizerSignal.value = VisualizerSignal.Listening(rmsNorm) + } // Log RMS every second for calibration if (frameCount % 10 == 0) { @@ -930,8 +1233,13 @@ class KazeiaService : Service() { frameCount++ val isSpeech = rms > silenceThreshold + // Pendant le warmup, on ignore tout candidat speech : audio + // pas encore stabilisé, RMS peut être élevé sur du bruit + // mécanique (click PTT, capacitive feedback) → Whisper + // hallucinerait. On laisse aussi le buffer vide. + val inWarmup = (System.currentTimeMillis() - listenStartMs) < warmupMs - if (isSpeech) { + if (isSpeech && !inWarmup) { silenceFrameCount = 0 speechFrameCount++ speechBuffer.add(frame.copyOf()) @@ -960,11 +1268,24 @@ class KazeiaService : Service() { val result = stt.transcribe(audioSnapshot, language = "fr") val elapsed = System.currentTimeMillis() - t if (result.text.isNotBlank()) { + val trimmed = result.text.trim() + val previousText = streamingResults.lastOrNull() streamingResults.clear() - streamingResults.add(result.text.trim()) - log("Streaming result (${elapsed}ms): \"${result.text.trim()}\"") - // Show intermediate result - _pipelineState.value = PipelineState.Transcribed(result.text.trim()) + streamingResults.add(trimmed) + log("Streaming result (${elapsed}ms): \"$trimmed\"") + _pipelineState.value = PipelineState.StreamingTranscript(trimmed) + // Finalisation par STABILITÉ : si Whisper sort 2× le même + // texte d'affilée, l'utilisateur a fini de parler. + // Permet de boucler la cascade sans attendre 800ms de + // silence VAD (qui ne se déclenche pas si bruit ambiant + // > seuil 80) ni le PTT release manuel. + if (previousText != null && previousText == trimmed) { + log("Streaming stable (2× '$trimmed') → finalize early") + // Force segment end : suffit d'incrémenter + // silenceFrameCount au-dessus du seuil pour + // déclencher la branche de finalisation. + silenceFrameCount = silenceEndFrames + 1 + } } } } @@ -1050,6 +1371,34 @@ class KazeiaService : Service() { recorder.release() log("Continuous listening stopped") + // PTT release : finaliser le segment en cours s'il y a du speech bufferisé. + // Sans ça, si le bruit ambiant flotte juste au-dessus du seuil VAD le + // segment ne se ferme jamais via silence detection → le LLM n'est jamais + // déclenché. On utilise le streaming result si dispo, sinon transcription + // complète du buffer. + if (isSpeechActive && speechBuffer.isNotEmpty()) { + val pendingMs = speechBuffer.sumOf { it.size } * 1000 / sampleRate + log("PTT release: forcing finalization of pending segment (${pendingMs}ms)") + val pending = ShortArray(speechBuffer.sumOf { it.size }) + var off = 0 + for (chunk in speechBuffer) { chunk.copyInto(pending, off); off += chunk.size } + val streamingText = streamingResults.joinToString(" ").trim() + serviceScope.launch(Dispatchers.IO) { + if (streamingText.isNotEmpty()) { + log("PTT-finalize using streaming result: \"$streamingText\"") + _pipelineState.value = PipelineState.Transcribed(streamingText) + if (!handleVoiceCommand(streamingText)) { + addMessage(ChatMessage(role = ChatMessage.Role.PATIENT, text = streamingText)) + processLlmResponse(streamingText) + } + } else { + log("PTT-finalize via full Whisper transcription (no streaming text)") + _pipelineState.value = PipelineState.Transcribing + processSpeechInput(pending) + } + } + } + } catch (e: Exception) { log("ERROR: Continuous listening error: ${e.message}") } @@ -1141,9 +1490,8 @@ class KazeiaService : Service() { private fun stopListening() { _isListening.value = false - val androidStt = stt as? AndroidSttEngine - androidStt?.stopListening() - audioCapture.stop() + // L'AudioRecord de startContinuousListening / startPushToTalkRecording + // est piloté par _isListening.value : la boucle while sort sur false. _pipelineState.value = PipelineState.Idle } @@ -1152,10 +1500,14 @@ class KazeiaService : Service() { val audioMs = audioData.size * 1000L / 16000 _aiWorkload.value = _aiWorkload.value.copy(sttActive = true) log(">>> STT START: ${audioMs}ms audio, ${audioData.size} samples, stt=${stt::class.simpleName}") + metrics.begin("speech", "${audioMs}ms audio") + metrics.mark("STT.start", "audio=${audioMs}ms engine=${stt::class.simpleName}") val t0 = System.currentTimeMillis() val transcription = stt.transcribe(audioData, language = "fr") val whisperMs = System.currentTimeMillis() - t0 + metrics.mark("STT.end", + "text=\"${transcription.text.take(60)}\" dur=${whisperMs}ms rtf=${"%.2f".format(whisperMs.toFloat() / audioMs)}") if (transcription.text.isBlank()) { log("Whisper: (vide) ${whisperMs}ms") @@ -1176,6 +1528,15 @@ class KazeiaService : Service() { fun processTextInput(text: String) { log("processTextInput: '$text'") + // For text input, the pipeline clock starts here (no STT stage). + // Speech path opens metrics itself in processSpeechInput before + // calling processLlmResponse, so we only begin() on the text path. + if (!text.trim().lowercase().let { + it.startsWith("pipeline") || it.startsWith("!pipeline") || + it.startsWith("\\!pipeline") || it == "go" + }) { + metrics.begin("text", text) + } serviceScope.launch { // Special test commands — cancel previous pipeline first if (text.trim().lowercase().let { it.startsWith("pipeline") || it.startsWith("!pipeline") || it.startsWith("\\!pipeline") || it == "go" }) { @@ -1225,6 +1586,15 @@ class KazeiaService : Service() { _pipelineState.value = PipelineState.Thinking conversationManager.onNewTurn() + // Historique : tour patient persisté avant toute génération (no-op + // en mode invité). Le tour Kazeia est loggé en fin de génération. + logConversationTurn(Role.PATIENT, patientMessage) + + // TTS désactivable à chaud depuis l'app admin (runtimeConfig.ttsEnabled) + // ou via la constante de dev LLM_ONLY_MODE. Si le TTS est coupé : aucune + // synthèse vocale, la bulle se remplit progressivement, token par token. + val llmOnly = LLM_ONLY_MODE || !runtimeConfig.ttsEnabled + // If LLM not loaded, use echo mode with TTS. Route through the // streaming session — in debug builds this is the fast path most // exercised, and it benefits from per-sentence playback like any @@ -1233,36 +1603,245 @@ class KazeiaService : Service() { log("Echo mode: '$patientMessage' → TTS (${tts::class.simpleName})") val echoResponse = patientMessage addMessage(ChatMessage(role = ChatMessage.Role.KAZEIA, text = echoResponse)) - pipeline.speakText(echoResponse) + metrics.mark("LLM.skipped", "echo_mode") + if (!llmOnly) pipeline.speakText(echoResponse) + metrics.end("echo") _pipelineState.value = if (_isListening.value) PipelineState.Listening else PipelineState.Idle return } + // Signal visuel Thinking ASAP (cycle de couleurs sur l'orbe) : couvre + // toute la cascade (Thinker pass + Speaker pass), repasse à Speaking + // automatiquement dès le 1er audio TTS. + _visualizerSignal.value = VisualizerSignal.Thinking + + // ─── Cascade Option E v2 : DÉSACTIVÉE 2026-05-04 ──────────────── + // Guard-0.6B produit des bullets que le Speaker régurgite verbatim + // dans sa réponse, malgré l'injection en system prompt + instruction + // "NE PAS RÉPÉTER". Cause : Guard est un classifieur safety, pas un + // Cascade pilotée par la config runtime (admin app). Quand cascade=ON, + // le Thinker analyse le message patient et produit des bullets injectées + // au Speaker. Quand OFF (défaut FROZEN), le Speaker reçoit le message + // directement avec son system prompt seul. + val cfg = runtimeConfig + val CASCADE_ENABLED = cfg.cascadeEnabled + val cascadeBullets: String = try { + if (CASCADE_ENABLED) ensureThinkerEngine() + val tE = if (CASCADE_ENABLED) thinkerEngine else null + if (tE != null && tE.isLoaded()) { + metrics.mark("THINKER.start", "msg_chars=${patientMessage.length}") + val tStart = System.currentTimeMillis() + val result = tE.generateWithSystem( + prompt = "Message du patient : $patientMessage", + systemPromptOverride = cfg.thinker.systemPrompt, + params = com.kazeia.core.SamplingParams( + maxNewTokens = 128, temperature = cfg.thinker.temperature + ), + onToken = null, + tag = "THINKER" + ) + val dt = System.currentTimeMillis() - tStart + // Tronque les bullets à ~400 chars (≈ 100 tokens) pour garantir que + // le prompt Speaker reste sous le budget prefill du .pte + // (context_len - ar_len ≈ 384 tokens). Sans cap, un Thinker qui + // "rame" (Qwen3-4B générique, pas un Guard spécialisé) peut produire + // 200-300 tokens de bullets et faire crasher Speaker au prefill. + val rawBullets = result.text + val capped = if (rawBullets.length > 250) { + log("[THINKER] WARN: bullets ${rawBullets.length} chars → truncated to 250") + rawBullets.take(250) + } else rawBullets + log("[THINKER] bullets (${dt}ms, ${result.tokenCount}tok ${"%.1f".format(result.tokensPerSecond)}t/s):\n$capped") + metrics.mark("THINKER.end", "wall=${dt}ms tok=${result.tokenCount} chars=${capped.length}") + capped + } else "" + } catch (e: Exception) { + log("[THINKER] failed: ${e.message} — fallback mono Speaker") + "" + } + + // Speaker : user prompt = juste le message patient. Les bullets vont + // dans le system prompt (override) pour empêcher le Speaker de les + // régurgiter dans sa sortie (vu en pratique : le 4B reformule "Analyse + // clinique : ...." comme partie de sa réponse, lue par le TTS). + // + // Sécurité prefill budget : context_len(512)-ar_len(128) = 384 tok max. + // SYS_KAZEIA + intro analyse ≈ 250 tok, bullets ≤ 250 chars (~65 tok), + // boilerplate ~15 tok → reste ~54 tok pour le message patient. + // À 4 chars/tok ≈ 200 chars max user message safe avec cascade. + // Si message plus long → on tombe en mono Speaker (pas de bullets). + val speakerSystemPrompt: String? = if (cascadeBullets.isNotBlank() && patientMessage.length <= 200) { + com.kazeia.llm.ExecuTorchLlmEngine.DEFAULT_SYSTEM_PROMPT + + "\n\nIndices d'analyse interne (NE PAS RÉPÉTER, NE PAS CITER, sert uniquement à orienter le ton de ta réponse) :\n" + + cascadeBullets + } else { + if (cascadeBullets.isNotBlank()) { + log("[CASCADE] patientMessage ${patientMessage.length} chars > 200 → fallback mono Speaker (skip bullets)") + } + null + } + val prompt = promptBuilder.build( message = patientMessage, history = _messages.value ) + // === Opt TTS-B: the TTS streaming session opens BEFORE llm.generate, + // so the first complete sentence emitted by the LLM (on terminal + // punctuation) is enqueued for synthesis while the LLM is still + // decoding the rest of the response. Saves ~LLM-wall-time of TTFA + // when the first sentence finishes before the LLM does (common for + // short conversational answers). Non-Qwen TTS backends that lack + // the enqueueSentence plumbing fall through to the legacy post-LLM + // pipeline.speakText batch call. + val qwen = tts as? com.kazeia.tts.Qwen3TtsEngine + // Bulle vide + signal Thinking sur le visualizer : la couleur de l'orbe + // pulse pendant que le LLM réfléchit, plus de "..." textuels parasites. + // L'écran reveal mot-par-mot s'enchaîne quand le premier audio joue. + val bubble = ChatMessage(role = ChatMessage.Role.KAZEIA, text = "") + val revealScope = kotlinx.coroutines.CoroutineScope(kotlinx.coroutines.Dispatchers.Default) + var revealedSoFar = "" + val revealJobs = mutableListOf() + val firstSegmentSeen = java.util.concurrent.atomic.AtomicBoolean(false) + // Annonce visuelle "thinking" : couleur orbe en pulsation jusqu'au 1er audio. + _visualizerSignal.value = VisualizerSignal.Thinking + // No-op job kept for compat with downstream typingJob.cancel() calls. + val typingJob = revealScope.launch { /* visual handled by orb */ } + var segCounter = 0 + val onSegPlay: (String, Long, FloatArray, Array) -> Unit = onSeg@{ + sentence, durationMs, envelope, spectrogram -> + val segIdx = segCounter++ + if (firstSegmentSeen.compareAndSet(false, true)) { + try { typingJob.cancel() } catch (_: Exception) {} + updateMessageText(bubble.id, "") + metrics.mark("PIPELINE.first_audio", + "seg=$segIdx dur=${durationMs}ms") + _pipelineState.value = PipelineState.Speaking + } + metrics.mark("TTS.seg.play_start", + "seg=$segIdx sentence=\"${sentence.take(40)}\" dur=${durationMs}ms") + val durCopy = durationMs + val segCopy = segIdx + kotlinx.coroutines.GlobalScope.launch { + kotlinx.coroutines.delay(durCopy) + metrics.mark("TTS.seg.play_end", + "seg=$segCopy reported_dur=${durCopy}ms") + } + _visualizerSignal.value = + VisualizerSignal.Speaking(envelope, spectrogram, durationMs) + val prefix = revealedSoFar + val words = sentence.split(Regex("\\s+")).filter { it.isNotBlank() } + revealedSoFar = + if (prefix.isEmpty()) sentence + else "$prefix $sentence" + if (words.isEmpty()) return@onSeg + val perWordMs = (durationMs / words.size).coerceAtLeast(40L) + val job = revealScope.launch { + val sb = StringBuilder(prefix) + if (prefix.isNotEmpty()) sb.append(' ') + sb.append(words[0]) + updateMessageText(bubble.id, sb.toString()) + for (i in 1 until words.size) { + kotlinx.coroutines.delay(perWordMs) + sb.append(' ').append(words[i]) + updateMessageText(bubble.id, sb.toString()) + } + } + revealJobs.add(job) + } + + // Bubble + TTS session open now so the animated typing dots appear + // as soon as the user finishes speaking, covering the LLM's TTFT. + addMessage(bubble) + val llmStreamStart = System.currentTimeMillis() + val sentenceCounter = java.util.concurrent.atomic.AtomicInteger(0) + // TTS coupé (llmOnly) → pas de session de synthèse ouverte ; la bulle + // se remplit token par token dans onToken (voir plus bas). + val ttsStreamer: com.kazeia.tts.SentenceStreamer? = + if (qwen != null && !llmOnly) { + qwen.onSegmentPlaying = onSegPlay + qwen.startStreamingSession() + com.kazeia.tts.SentenceStreamer { raw -> + val sIdx = sentenceCounter.getAndIncrement() + val dt = System.currentTimeMillis() - llmStreamStart + val spoken = pipeline.stripNonSpeakable(raw).trim() + log("[LLM.sentence] #$sIdx +${dt}ms → '$spoken'") + if (spoken.isNotEmpty()) qwen.enqueueSentence(spoken) + } + } else null + + // In LLM_ONLY_MODE, kill the typing dots as soon as the first token + // arrives so the bubble starts revealing text immediately. Outside + // this mode, the dots continue until the first TTS segment plays. + val killDotsOnFirstToken = llmOnly + try { stoppingCriteria.reset() val responseBuilder = StringBuilder() _aiWorkload.value = _aiWorkload.value.copy(llmActive = true) - val result = llm.generate( - prompt = prompt, - params = SamplingParams( - maxNewTokens = 450, - temperature = conversationManager.currentTemperature() - ), - onToken = { token -> - responseBuilder.append(token) + metrics.mark("LLM.start", + "engine=${llm::class.simpleName} prompt_chars=${prompt.length}") + var firstTokenSeen = false + var firstTokenAtMs = -1L + val tokenIdx = java.util.concurrent.atomic.AtomicInteger(0) + + val onTokenLambda: (String) -> Boolean = { token -> + if (!firstTokenSeen) { + firstTokenSeen = true + firstTokenAtMs = System.currentTimeMillis() - llmStreamStart + metrics.mark("LLM.first_token", "piece=\"${token.take(16).replace("\n", "\\n")}\"") + if (killDotsOnFirstToken && firstSegmentSeen.compareAndSet(false, true)) { + try { typingJob.cancel() } catch (_: Exception) {} + updateMessageText(bubble.id, "") + } + } + responseBuilder.append(token) + ttsStreamer?.append(token) + // TTS coupé : la bulle se met à jour à chaque token, l'usager + // voit la réponse s'écrire progressivement en temps réel. + if (llmOnly) { + updateMessageText(bubble.id, responseBuilder.toString()) + } + val ti = tokenIdx.getAndIncrement() + val dt = System.currentTimeMillis() - llmStreamStart + val safe = token.replace("\n", "\\n").replace("\r", "") + log("[LLM.tok] #$ti +${dt}ms '$safe'") + // Suppress TokenGenerated state updates once the first + // segment plays — the state stays Speaking so the mic + // loop reliably drops frames during TTS playback. + if (!firstSegmentSeen.get()) { _pipelineState.value = PipelineState.TokenGenerated( token = token, fullText = responseBuilder.toString() ) - !stoppingCriteria.shouldStop(responseBuilder.toString()) } + !stoppingCriteria.shouldStop(responseBuilder.toString()) + } + val sParams = SamplingParams( + maxNewTokens = 450, + temperature = conversationManager.currentTemperature() ) + // Si bullets cascade dispo → injection dans system prompt via + // generateWithSystem (cast vers ExecuTorchLlmEngine). Sinon + // generateWithSystem aussi pour propager le prompt système Speaker + // de la config runtime (hot-reload depuis admin app). + val llmRef = llm + val speakerPromptToUse = speakerSystemPrompt ?: cfg.speaker.systemPrompt + val result = if (llmRef is com.kazeia.llm.ExecuTorchLlmEngine) { + llmRef.generateWithSystem( + prompt = prompt, + systemPromptOverride = speakerPromptToUse, + params = sParams, + onToken = onTokenLambda, + tag = "SPEAKER" + ) + } else { + llmRef.generate(prompt = prompt, params = sParams, onToken = onTokenLambda) + } + + metrics.mark("LLM.end", + "tokens=${result.tokenCount} wall=${result.timeMs}ms tps=${"%.2f".format(result.tokensPerSecond)}") _aiWorkload.value = _aiWorkload.value.copy(llmActive = false) val responseText = result.text.trim() @@ -1272,109 +1851,61 @@ class KazeiaService : Service() { log("LLM stats: ${result.tokenCount} tokens in ${result.timeMs}ms (${result.tokensPerSecond} tok/s)") if (responseText.isNotEmpty()) { - // Mark the pipeline as Speaking for the duration of TTS so - // the continuous-listening mic loop drops frames and we - // don't feed our own speaker output back into STT. - _pipelineState.value = PipelineState.Speaking - // Create a KAZEIA bubble up-front. Until the first TTS - // segment actually starts playing the bubble shows an - // animated "." → ".." → "..." typing indicator so the - // user knows Kazeia is thinking/synthesising; once the - // first segment plays the dots are cleared and the - // per-sentence word reveal takes over. - val bubble = ChatMessage(role = ChatMessage.Role.KAZEIA, text = ".") - addMessage(bubble) - val revealScope = kotlinx.coroutines.CoroutineScope(kotlinx.coroutines.Dispatchers.Default) - var revealedSoFar = "" - val revealJobs = mutableListOf() - val firstSegmentSeen = java.util.concurrent.atomic.AtomicBoolean(false) - val typingJob = revealScope.launch { - var tick = 0 - while (!firstSegmentSeen.get()) { - val dots = ".".repeat(1 + (tick % 3)) // . → .. → ... - updateMessageText(bubble.id, dots) - tick++ - kotlinx.coroutines.delay(400) - } - } - try { - pipeline.speakText(responseText) { sentence, durationMs, envelope, spectrogram -> - // First segment: stop the typing indicator and - // reset the bubble to empty so the word reveal - // doesn't collide with the dots. - if (firstSegmentSeen.compareAndSet(false, true)) { - try { typingJob.cancel() } catch (_: Exception) {} - updateMessageText(bubble.id, "") - } - // Push the envelope + spectrogram to the - // visualizer at the same moment the MediaPlayer - // starts playing so the orb reacts to this - // segment's actual energy and the in-sphere - // spectrum bars match the audio content. - _visualizerSignal.value = - VisualizerSignal.Speaking(envelope, spectrogram, durationMs) - // Start a coroutine that appends one word at a time - // over the segment's audio duration. Words are - // separated on whitespace; punctuation rides with - // the trailing word. The prefix (= text already - // revealed from previous sentences) carries over so - // earlier sentences stay on screen. - val prefix = revealedSoFar - val words = sentence.split(Regex("\\s+")).filter { it.isNotBlank() } - revealedSoFar = - if (prefix.isEmpty()) sentence - else "$prefix $sentence" - if (words.isEmpty()) return@speakText - val perWordMs = (durationMs / words.size).coerceAtLeast(40L) - val job = revealScope.launch { - val sb = StringBuilder(prefix) - if (prefix.isNotEmpty()) sb.append(' ') - // Immediately reveal the first word so there's - // no visible gap between audio start and text. - sb.append(words[0]) - updateMessageText(bubble.id, sb.toString()) - for (i in 1 until words.size) { - kotlinx.coroutines.delay(perWordMs) - sb.append(' ').append(words[i]) - updateMessageText(bubble.id, sb.toString()) - } - } - revealJobs.add(job) - } - // After all segments finished playing, ensure the full - // text is visible even if a reveal job was racing. + if (llmOnly) { + // TTS coupé — la bulle a été remplie token par token dans + // onToken ; on finalise avec la réponse nettoyée. + updateMessageText(bubble.id, responseText) + metrics.end("llm_only") + } else if (ttsStreamer != null && qwen != null) { + // Streaming path: tail fragment (if any) + drain playback. + ttsStreamer.flush() + qwen.endStreamingSession() revealJobs.forEach { try { it.join() } catch (_: Exception) {} } updateMessageText(bubble.id, responseText) - } finally { - // Defensive: cancel the typing dots in case no - // segment ever fired (e.g. the response was entirely - // emojis and got stripped empty). - firstSegmentSeen.set(true) - try { typingJob.cancel() } catch (_: Exception) {} - _pipelineState.value = if (_isListening.value) - PipelineState.Listening else PipelineState.Idle - // If we're going back to mic listening, the VAD loop - // will keep pushing Listening signals; otherwise drop - // to Idle so the orb settles back to its breathing - // baseline. - if (!_isListening.value) { - _visualizerSignal.value = VisualizerSignal.Idle + metrics.end("ok") + } else { + // Legacy batch path for non-Qwen TTS engines. + _pipelineState.value = PipelineState.Speaking + pipeline.speakText(responseText) { sentence, durationMs, envelope, spectrogram -> + onSegPlay(sentence, durationMs, envelope, spectrogram) } + revealJobs.forEach { try { it.join() } catch (_: Exception) {} } + updateMessageText(bubble.id, responseText) + metrics.end("ok") } + // Historique : tour Kazeia persisté avec TTFT + durée LLM. + logConversationTurn( + role = Role.KAZEIA, + text = responseText, + ttftMs = firstTokenAtMs.takeIf { it >= 0 }, + totalMs = result.timeMs + ) } else { - _pipelineState.value = if (_isListening.value) - PipelineState.Listening else PipelineState.Idle + // Empty response — close any open streaming session cleanly + // and drop the placeholder bubble. + try { qwen?.endStreamingSession() } catch (_: Exception) {} + _messages.value = _messages.value.filter { it.id != bubble.id } + metrics.end("empty_response") } - } catch (e: Exception) { _aiWorkload.value = _aiWorkload.value.copy(llmActive = false) log("ERROR: LLM generation error: ${e.message}") + metrics.mark("LLM.error", "msg=\"${e.message?.take(80) ?: ""}\"") + metrics.end("error") + try { qwen?.endStreamingSession() } catch (_: Exception) {} _pipelineState.value = PipelineState.Error(e.message ?: "Erreur LLM") addMessage(ChatMessage( role = ChatMessage.Role.SYSTEM, text = "Erreur lors de la génération: ${e.message}" )) - _pipelineState.value = PipelineState.Idle + } finally { + firstSegmentSeen.set(true) + try { typingJob.cancel() } catch (_: Exception) {} + _pipelineState.value = if (_isListening.value) + PipelineState.Listening else PipelineState.Idle + if (!_isListening.value) { + _visualizerSignal.value = VisualizerSignal.Idle + } } } @@ -1414,10 +1945,103 @@ class KazeiaService : Service() { override fun onDestroy() { super.onDestroy() serviceScope.cancel() - if (_isListening.value) audioCapture.stop() + _isListening.value = false // pilote l'arrêt des boucles AudioRecord + configReloadReceiver?.let { try { unregisterReceiver(it) } catch (_: Exception) {} } + stopThinkerEngine() try { llm.release() } catch (_: Exception) {} try { stt.release() } catch (_: Exception) {} try { tts.release() } catch (_: Exception) {} - try { vad.release() } catch (_: Exception) {} + } + + private fun registerConfigReloadReceiver() { + val r = object : android.content.BroadcastReceiver() { + override fun onReceive(c: android.content.Context?, i: Intent?) { + log("[CONFIG] RELOAD_CONFIG broadcast reçu — re-lecture du store") + com.kazeia.config.ConfigStore.get(this@KazeiaService).load() + } + } + configReloadReceiver = r + val filter = android.content.IntentFilter( + com.kazeia.telemetry.KazeiaTelemetryProvider.ACTION_RELOAD_CONFIG + ) + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) { + registerReceiver(r, filter, RECEIVER_NOT_EXPORTED) + } else { + @Suppress("UnspecifiedRegisterReceiverFlag") + registerReceiver(r, filter) + } + } + + /** + * Appliqué quand l'admin app (via ContentProvider.update) ou la + * relecture broadcast écrasent la config. + * + * - Prompts : pris en compte au prochain tour (l'engine lit la config + * à chaque generate). Pas de reload. + * - Speaker model_id changé → restart complet du process Kazeia. + * L'in-process release+recreate échoue à la 2ᵉ exécution + * (`Failed to load llm runner: [1]`) à cause d'un bug singleton dans + * `QnnBackendUnifiedRegistry` (cf project_kazeia_reset_context_bug). + * Un restart propre relit la config et instancie l'engine avec le + * nouveau .pte sans état QNN résiduel. + * - Cascade toggle ON → Thinker chargé au prochain tour ; OFF → + * libération immédiate. + * - Thinker model_id changé en cascade ON → release, sera lazy-loadé. + */ + private fun handleConfigChange(new: com.kazeia.config.ConfigStore.RuntimeConfig) { + val old = runtimeConfig + runtimeConfig = new + log("[CONFIG] applied : cascade ${old.cascadeEnabled}→${new.cascadeEnabled}, " + + "speaker ${old.speaker.modelId}→${new.speaker.modelId}, " + + "thinker ${old.thinker.modelId}→${new.thinker.modelId}") + + // Speaker model swap → full restart (contourne bug QNN double-swap) + if (new.speaker.modelId != old.speaker.modelId) { + log("[CONFIG] Speaker model changé → restart Kazeia dans 1s pour state QNN propre") + scheduleSelfRestart() + return // pas la peine de traiter le Thinker, on va redémarrer + } + + // Thinker lifecycle suivant le toggle + le model_id + if (!new.cascadeEnabled && old.cascadeEnabled) { + log("[CONFIG] cascade OFF → libération Thinker") + stopThinkerEngine() + } else if (new.cascadeEnabled && old.cascadeEnabled && + new.thinker.modelId != old.thinker.modelId) { + log("[CONFIG] Thinker model changé → release + lazy reload au prochain tour") + stopThinkerEngine() + } + // ON après OFF : le Thinker sera lazy-loadé au prochain tour via ensureThinkerEngine() + } + + /** + * Programme un redémarrage propre du process Kazeia : AlarmManager + * relance ChatActivity dans 1 seconde, puis on tue le process actuel. + * Au cold start, ConfigStore.load() lit la nouvelle config et le LLM + * est instancié avec le bon .pte — sans état QNN résiduel. + */ + private fun scheduleSelfRestart() { + val ctx = applicationContext + val launchIntent = Intent(ctx, com.kazeia.ui.SplashActivity::class.java).apply { + addFlags(Intent.FLAG_ACTIVITY_NEW_TASK or Intent.FLAG_ACTIVITY_CLEAR_TASK) + } + val pi = android.app.PendingIntent.getActivity( + ctx, 0, launchIntent, + android.app.PendingIntent.FLAG_IMMUTABLE or + android.app.PendingIntent.FLAG_CANCEL_CURRENT + ) + val am = ctx.getSystemService(android.app.AlarmManager::class.java) + am?.set( + android.app.AlarmManager.RTC, + System.currentTimeMillis() + 1000, + pi + ) + log("[CONFIG] self-restart programmé (alarm +1s) — process killé dans 500ms") + // Petit délai pour que la réponse ContentProvider.update parte avant + // le suicide du process. + serviceScope.launch { + kotlinx.coroutines.delay(500) + android.os.Process.killProcess(android.os.Process.myPid()) + } } } diff --git a/kazeia-android/app/src/main/java/com/kazeia/telemetry/KazeiaTelemetryProvider.kt b/kazeia-android/app/src/main/java/com/kazeia/telemetry/KazeiaTelemetryProvider.kt new file mode 100644 index 0000000..8f6934f --- /dev/null +++ b/kazeia-android/app/src/main/java/com/kazeia/telemetry/KazeiaTelemetryProvider.kt @@ -0,0 +1,576 @@ +package com.kazeia.telemetry + +import android.content.ContentProvider +import android.content.ContentValues +import android.content.UriMatcher +import android.database.Cursor +import android.database.MatrixCursor +import android.net.Uri +import android.os.Debug +import android.os.Process +import java.io.File + +/** + * Expose à l'admin app (com.kazeia.admin) les données runtime du process + * Kazeia patient via une URI standard. + * + * Tables (read-only sauf voices delete) : + * - content://com.kazeia.provider/state — 1 row : état process courant + * - content://com.kazeia.provider/turns — N rows : derniers tours pipeline + * - content://com.kazeia.provider/crashes — N rows : derniers crashes + * - content://com.kazeia.provider/voices — N rows : voix présentes sur disque + * - content://com.kazeia.provider/voices/ — delete : supprime la voix + * + * Pas de permission custom en MVP — données non sensibles (système + meta voix). + * À sécuriser en production via signature-based permission une fois le keystore + * partagé mis en place (cf project_kazeia_admin_app.md). + */ +class KazeiaTelemetryProvider : ContentProvider() { + + companion object { + const val AUTHORITY = "com.kazeia.provider" + const val PATH_STATE = "state" + const val PATH_TURNS = "turns" + const val PATH_CRASHES = "crashes" + const val PATH_VOICES = "voices" + const val PATH_CONFIG = "config" + const val PATH_MODELS = "models" + const val PATH_PROFILES = "profiles" + const val PATH_PROFILE_ACTIVE = "profiles/active" + const val PATH_CONVERSATIONS = "conversations" + const val PATH_DIST_CONFIG = "dist_config" + const val ACTION_RELOAD_CONFIG = "com.kazeia.action.RELOAD_CONFIG" + const val ACTION_RELOAD_PROFILES = "com.kazeia.action.RELOAD_PROFILES" + + private const val CODE_STATE = 1 + private const val CODE_TURNS = 2 + private const val CODE_CRASHES = 3 + private const val CODE_VOICES = 4 + private const val CODE_VOICE_ITEM = 5 + private const val CODE_CONFIG = 6 + private const val CODE_MODELS = 7 + private const val CODE_PROFILES = 8 + private const val CODE_PROFILE_ITEM = 9 + private const val CODE_PROFILE_ACTIVE = 10 + private const val CODE_CONVERSATIONS = 11 + private const val CODE_CONVERSATION_SESSION = 12 + private const val CODE_CONVERSATION_PROFILE = 13 + private const val CODE_DIST_CONFIG = 14 + + // Emplacements canoniques — résolus par KazeiaPaths (legacy /data/local/tmp + // sur tablette dev, stockage externe app sinon). Non-const. + private val VOICE_WAV_DIR: String + get() = "${com.kazeia.KazeiaApplication.MODELS_DIR}/../voix/voix" + private val VOICE_EMB_DIR: String + get() = "${com.kazeia.KazeiaApplication.MODELS_DIR}/qwen3-tts-npu" + private const val PREFIX_SUFFIX = "_voice_prefix.bin" + private const val SUFFIX_SUFFIX = "_voice_suffix.bin" + } + + private val matcher = UriMatcher(UriMatcher.NO_MATCH).apply { + addURI(AUTHORITY, PATH_STATE, CODE_STATE) + addURI(AUTHORITY, PATH_TURNS, CODE_TURNS) + addURI(AUTHORITY, PATH_CRASHES, CODE_CRASHES) + addURI(AUTHORITY, PATH_VOICES, CODE_VOICES) + addURI(AUTHORITY, "$PATH_VOICES/*", CODE_VOICE_ITEM) + addURI(AUTHORITY, PATH_CONFIG, CODE_CONFIG) + addURI(AUTHORITY, PATH_MODELS, CODE_MODELS) + // Profiles : enum + active + per-id (CRUD) + addURI(AUTHORITY, PATH_PROFILES, CODE_PROFILES) + addURI(AUTHORITY, PATH_PROFILE_ACTIVE, CODE_PROFILE_ACTIVE) + addURI(AUTHORITY, "$PATH_PROFILES/*", CODE_PROFILE_ITEM) + // Conversations : timeline + filter by session/profile + addURI(AUTHORITY, PATH_CONVERSATIONS, CODE_CONVERSATIONS) + addURI(AUTHORITY, "$PATH_CONVERSATIONS/session/*", CODE_CONVERSATION_SESSION) + addURI(AUTHORITY, "$PATH_CONVERSATIONS/profile/*", CODE_CONVERSATION_PROFILE) + // Config WebDAV de distribution (éditable depuis l'app admin) + addURI(AUTHORITY, PATH_DIST_CONFIG, CODE_DIST_CONFIG) + } + + override fun onCreate(): Boolean = true + + override fun query( + uri: Uri, + projection: Array?, + selection: String?, + selectionArgs: Array?, + sortOrder: String? + ): Cursor? = when (matcher.match(uri)) { + CODE_STATE -> stateCursor() + CODE_TURNS -> turnsCursor() + CODE_CRASHES -> crashesCursor() + CODE_VOICES -> voicesCursor() + CODE_CONFIG -> configCursor() + CODE_MODELS -> modelsCursor() + CODE_PROFILES -> profilesCursor() + CODE_PROFILE_ACTIVE -> activeProfileCursor() + CODE_CONVERSATIONS -> sessionsCursor(profileId = null) + CODE_CONVERSATION_PROFILE -> sessionsCursor(profileId = uri.lastPathSegment) + CODE_CONVERSATION_SESSION -> turnsForSessionCursor(uri.lastPathSegment ?: "") + CODE_DIST_CONFIG -> distConfigCursor() + else -> null + } + + private fun profilesCursor(): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val store = com.kazeia.profiles.ProfileStore.get(ctx) + val cols = arrayOf( + "id", "display_name", "avatar_color", "voice_id", + "system_prompt_override", "has_pin", "created_at", + "last_used_at", "notes", "is_default", "turn_count" + ) + val cursor = MatrixCursor(cols) + val dao = com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ) + for (p in store.all()) { + val count = try { dao.countTurns(p.id) } catch (_: Exception) { 0 } + cursor.addRow(arrayOf( + p.id, p.displayName, p.avatarColor, p.voiceId, + p.systemPromptOverride, if (p.hasPin()) 1 else 0, + p.createdAt, p.lastUsedAt, p.notes, + if (p.isDefault) 1 else 0, count + )) + } + return cursor + } + + private fun activeProfileCursor(): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val cursor = MatrixCursor(arrayOf("active_profile_id")) + val active = com.kazeia.profiles.ProfileStore.get(ctx).activeProfile() + cursor.addRow(arrayOf(active?.id)) + return cursor + } + + private fun sessionsCursor(profileId: String?): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val cols = arrayOf( + "profile_id", "session_id", "started_at", "ended_at", "turn_count" + ) + val cursor = MatrixCursor(cols) + try { + val dao = com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ) + for (s in dao.listSessions(profileId)) { + cursor.addRow(arrayOf(s.profileId, s.sessionId, + s.startedAt, s.endedAt, s.turnCount)) + } + } catch (_: Exception) { /* DB locked or new : retourne vide */ } + return cursor + } + + private fun turnsForSessionCursor(sessionId: String): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val cols = arrayOf( + "id", "profile_id", "session_id", "timestamp", "role", + "text", "ttft_ms", "total_ms", "voice_used", "model_used" + ) + val cursor = MatrixCursor(cols) + try { + val dao = com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ) + for (t in dao.listTurns(sessionId = sessionId)) { + cursor.addRow(arrayOf( + t.id, t.profileId, t.sessionId, t.timestamp, t.role.name, + t.text, t.ttftMs, t.totalMs, t.voiceUsed, t.modelUsed + )) + } + } catch (_: Exception) { } + return cursor + } + + private fun configCursor(): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val cfg = com.kazeia.config.ConfigStore.get(ctx).current() + val cols = arrayOf( + "cascade_enabled", "tts_enabled", + "stt_engine", "llm_engine", "tts_engine", + "speaker_model_id", "speaker_system_prompt", "speaker_temperature", + "thinker_model_id", "thinker_system_prompt", "thinker_temperature" + ) + val cursor = MatrixCursor(cols) + cursor.addRow(arrayOf( + if (cfg.cascadeEnabled) 1 else 0, + if (cfg.ttsEnabled) 1 else 0, + cfg.sttEngine, cfg.llmEngine, cfg.ttsEngine, + cfg.speaker.modelId, cfg.speaker.systemPrompt, cfg.speaker.temperature, + cfg.thinker.modelId, cfg.thinker.systemPrompt, cfg.thinker.temperature + )) + return cursor + } + + private fun modelsCursor(): Cursor { + val cols = arrayOf( + "id", "display_name", "pte_path", "tokenizer_path", + "role", "size_mb", "max_seq_len", "file_exists", "notes" + ) + val cursor = MatrixCursor(cols) + for (m in com.kazeia.config.ModelRegistry.ALL) { + val path = if (m.backend == com.kazeia.config.ModelRegistry.Backend.GGUF) m.ggufPath() else m.ptePath() + cursor.addRow(arrayOf( + m.id, m.displayName, path, m.tokenizerPath(), + m.role.name, m.sizeMb, m.maxSeqLen, + if (m.fileExists()) 1 else 0, m.notes + )) + } + return cursor + } + + override fun update( + uri: Uri, + values: ContentValues?, + selection: String?, + selectionArgs: Array? + ): Int { + if (values == null) return 0 + return when (matcher.match(uri)) { + CODE_CONFIG -> updateConfig(uri, values) + CODE_PROFILES -> upsertProfile(values) + CODE_PROFILE_ITEM -> upsertProfile(values.apply { + if (!containsKey("id")) put("id", uri.lastPathSegment) + }) + CODE_PROFILE_ACTIVE -> setActiveProfile(values.getAsString("active_profile_id")) + CODE_DIST_CONFIG -> updateDistConfig(uri, values) + else -> 0 + } + } + + /** + * Config WebDAV : 1 ligne. Le mot de passe n'est jamais renvoyé en clair — + * seul `has_password` indique s'il est défini. + */ + private fun distConfigCursor(): Cursor { + val ctx = context ?: return MatrixCursor(arrayOf()) + val cfg = com.kazeia.dist.DistConfigStore.get(ctx) + val cursor = MatrixCursor(arrayOf( + "webdav_base", "webdav_user", "has_password", "is_customized")) + cursor.addRow(arrayOf( + cfg.base, + cfg.user, + if (cfg.pass.isNotBlank()) 1 else 0, + if (com.kazeia.dist.DistConfigStore.isCustomized(ctx)) 1 else 0 + )) + return cursor + } + + private fun updateDistConfig(uri: Uri, values: ContentValues): Int { + val ctx = context ?: return 0 + // reset=true → retour au défaut d'usine (BuildConfig). + if (values.getAsBoolean("reset") == true) { + com.kazeia.dist.DistConfigStore.reset(ctx) + ctx.contentResolver.notifyChange(uri, null) + return 1 + } + val cur = com.kazeia.dist.DistConfigStore.get(ctx) + val base = values.getAsString("webdav_base") ?: cur.base + val user = values.getAsString("webdav_user") ?: cur.user + // Mot de passe : absent ou vide → inchangé ; non-vide → nouveau. + val pass = values.getAsString("webdav_pass")?.takeIf { it.isNotEmpty() } + com.kazeia.dist.DistConfigStore.update(ctx, base, user, pass) + ctx.contentResolver.notifyChange(uri, null) + return 1 + } + + private fun updateConfig(uri: Uri, values: ContentValues): Int { + val ctx = context ?: return 0 + val store = com.kazeia.config.ConfigStore.get(ctx) + val cur = store.current() + val newCfg = com.kazeia.config.ConfigStore.RuntimeConfig( + cascadeEnabled = if (values.containsKey("cascade_enabled")) + values.getAsBoolean("cascade_enabled") else cur.cascadeEnabled, + ttsEnabled = if (values.containsKey("tts_enabled")) + values.getAsBoolean("tts_enabled") else cur.ttsEnabled, + sttEngine = values.getAsString("stt_engine") ?: cur.sttEngine, + llmEngine = values.getAsString("llm_engine") ?: cur.llmEngine, + ttsEngine = values.getAsString("tts_engine") ?: cur.ttsEngine, + speaker = com.kazeia.config.ConfigStore.ModelConfig( + modelId = values.getAsString("speaker_model_id") ?: cur.speaker.modelId, + systemPrompt = values.getAsString("speaker_system_prompt") + ?: cur.speaker.systemPrompt, + temperature = values.getAsFloat("speaker_temperature") + ?: cur.speaker.temperature + ), + thinker = com.kazeia.config.ConfigStore.ModelConfig( + modelId = values.getAsString("thinker_model_id") ?: cur.thinker.modelId, + systemPrompt = values.getAsString("thinker_system_prompt") + ?: cur.thinker.systemPrompt, + temperature = values.getAsFloat("thinker_temperature") + ?: cur.thinker.temperature + ) + ) + store.save(newCfg) + ctx.contentResolver.notifyChange(uri, null) + ctx.sendBroadcast(android.content.Intent(ACTION_RELOAD_CONFIG)) + return 1 + } + + private fun upsertProfile(values: ContentValues): Int { + val ctx = context ?: return 0 + val store = com.kazeia.profiles.ProfileStore.get(ctx) + val id = values.getAsString("id") ?: com.kazeia.profiles.ProfileStore.newProfileId() + val existing = store.byId(id) + val displayName = values.getAsString("display_name") ?: existing?.displayName ?: id + val pinRaw = values.getAsString("pin") // PIN en clair pour set/clear + val newPinHash = when { + pinRaw == null -> existing?.pinHash // pas de changement + pinRaw.isEmpty() -> null // efface PIN + else -> com.kazeia.profiles.Profile.hashPin(pinRaw) + } + val p = com.kazeia.profiles.Profile( + id = id, + displayName = displayName, + avatarColor = values.getAsInteger("avatar_color") + ?: existing?.avatarColor + ?: com.kazeia.profiles.Profile.defaultColorFor(id), + voiceId = values.getAsString("voice_id") ?: existing?.voiceId, + systemPromptOverride = values.getAsString("system_prompt_override") + ?: existing?.systemPromptOverride, + pinHash = newPinHash, + createdAt = existing?.createdAt ?: System.currentTimeMillis(), + lastUsedAt = existing?.lastUsedAt ?: System.currentTimeMillis(), + notes = values.getAsString("notes") ?: existing?.notes, + isDefault = values.getAsBoolean("is_default") ?: existing?.isDefault ?: false + ) + store.upsert(p) + ctx.sendBroadcast(android.content.Intent(ACTION_RELOAD_PROFILES)) + return 1 + } + + private fun setActiveProfile(profileId: String?): Int { + val ctx = context ?: return 0 + com.kazeia.profiles.ProfileStore.get(ctx).setActiveProfile(profileId) + ctx.sendBroadcast(android.content.Intent(ACTION_RELOAD_PROFILES)) + return 1 + } + + private fun voicesCursor(): Cursor { + val cols = arrayOf( + "id", "name", "state", + "wav_exists", "prefix_exists", "suffix_exists", + "wav_size_bytes", "wav_duration_seconds", "wav_path", "created_at" + ) + val cursor = MatrixCursor(cols) + + // Set des ids = union (wav basenames) ∪ (prefix bin basenames) + val wavDir = File(VOICE_WAV_DIR) + val embDir = File(VOICE_EMB_DIR) + val wavIds = (wavDir.listFiles { f -> f.name.endsWith(".wav") } ?: emptyArray()) + .associateBy { it.nameWithoutExtension.lowercase() } + val prefixIds = (embDir.listFiles { f -> f.name.endsWith(PREFIX_SUFFIX) } ?: emptyArray()) + .associateBy { it.name.removeSuffix(PREFIX_SUFFIX).lowercase() } + val ids = (wavIds.keys + prefixIds.keys).sorted() + + for (id in ids) { + val wav = wavIds[id] + val prefix = File(VOICE_EMB_DIR, "${id}$PREFIX_SUFFIX") + val suffix = File(VOICE_EMB_DIR, "${id}$SUFFIX_SUFFIX") + val state = when { + prefix.exists() && suffix.exists() -> "ready" + wav != null -> "recorded" + else -> "error" + } + val durSec = wav?.let { wavDurationSeconds(it) } ?: 0f + cursor.addRow(arrayOf( + id, + id.replaceFirstChar { it.uppercase() }, + state, + if (wav != null) 1 else 0, + if (prefix.exists()) 1 else 0, + if (suffix.exists()) 1 else 0, + wav?.length() ?: 0L, + durSec, + wav?.absolutePath ?: "", + wav?.lastModified() ?: prefix.lastModified() + )) + } + return cursor + } + + /** Décode l'en-tête WAV pour estimer la durée. Très tolérant. */ + private fun wavDurationSeconds(f: File): Float = try { + java.io.RandomAccessFile(f, "r").use { raf -> + // Saute le RIFF header (12 bytes), parcourt les chunks + raf.seek(12) + var sampleRate = 0 + var channels = 0 + var bitsPerSample = 0 + var dataSize = 0L + while (raf.filePointer < raf.length() - 8) { + val chunkId = ByteArray(4).also { raf.readFully(it) } + val sizeBytes = ByteArray(4).also { raf.readFully(it) } + val size = ((sizeBytes[0].toInt() and 0xff) + or ((sizeBytes[1].toInt() and 0xff) shl 8) + or ((sizeBytes[2].toInt() and 0xff) shl 16) + or ((sizeBytes[3].toInt() and 0xff) shl 24)).toLong() and 0xffffffffL + val id = String(chunkId) + if (id == "fmt ") { + val pos = raf.filePointer + raf.skipBytes(2) // audio format + channels = raf.readUnsignedByte() or (raf.readUnsignedByte() shl 8) + sampleRate = raf.readUnsignedByte() or + (raf.readUnsignedByte() shl 8) or + (raf.readUnsignedByte() shl 16) or + (raf.readUnsignedByte() shl 24) + raf.seek(pos + size) + } else if (id == "data") { + dataSize = size + break + } else { + raf.skipBytes(size.toInt()) + } + if (raf.filePointer % 2L != 0L) raf.skipBytes(1) + } + // bitsPerSample : lu rapidement par estimation sinon 16 par défaut + if (sampleRate <= 0 || channels <= 0) 0f + else { + // bytes per sample = 2 par défaut + val bytesPerSample = 2 + val totalSamples = dataSize / (channels * bytesPerSample) + totalSamples.toFloat() / sampleRate + } + } + } catch (_: Exception) { 0f } + + override fun delete(uri: Uri, selection: String?, selectionArgs: Array?): Int { + val matched = matcher.match(uri) + android.util.Log.i("KazeiaProvider", "delete uri=$uri matched=$matched lastSeg=${uri.lastPathSegment}") + when (matched) { + CODE_VOICE_ITEM -> return deleteVoiceFiles(uri.lastPathSegment?.lowercase() ?: return 0) + CODE_PROFILE_ITEM -> return deleteProfile(uri.lastPathSegment ?: return 0) + CODE_CONVERSATION_PROFILE -> return deleteConversationsForProfile( + uri.lastPathSegment ?: return 0 + ) + CODE_CONVERSATION_SESSION -> return deleteConversationSession( + uri.lastPathSegment ?: return 0 + ) + } + return 0 + } + + private fun deleteVoiceFiles(id: String): Int { + var removed = 0 + listOf( + File(VOICE_EMB_DIR, "${id}$PREFIX_SUFFIX"), + File(VOICE_EMB_DIR, "${id}$SUFFIX_SUFFIX"), + File(VOICE_WAV_DIR, "${id}.wav") + ).forEach { f -> + val exists = f.exists() + val ok = if (exists) f.delete() else false + android.util.Log.i("KazeiaProvider", " ${f.absolutePath} exists=$exists deleted=$ok") + if (exists && ok) removed++ + } + context?.contentResolver?.notifyChange( + Uri.parse("content://$AUTHORITY/$PATH_VOICES"), null + ) + context?.sendBroadcast(android.content.Intent("com.kazeia.action.RELOAD_VOICES")) + return removed + } + + private fun deleteProfile(id: String): Int { + val ctx = context ?: return 0 + // Supprime aussi tout l'historique de conversation associé + try { + com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ).deleteProfile(id) + } catch (_: Exception) { } + val ok = com.kazeia.profiles.ProfileStore.get(ctx).delete(id) + if (ok) ctx.sendBroadcast(android.content.Intent(ACTION_RELOAD_PROFILES)) + return if (ok) 1 else 0 + } + + private fun deleteConversationsForProfile(id: String): Int { + val ctx = context ?: return 0 + return try { + com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ).deleteProfile(id) + } catch (_: Exception) { 0 } + } + + private fun deleteConversationSession(id: String): Int { + val ctx = context ?: return 0 + return try { + com.kazeia.profiles.ConversationDao( + com.kazeia.profiles.ConversationDb.get(ctx) + ).deleteSession(id) + } catch (_: Exception) { 0 } + } + + private fun stateCursor(): Cursor { + val cols = arrayOf( + "pid", "uptime_seconds", "pss_mb", "ion_mb", + "anon_pages_mb", "mem_available_mb" + ) + val cursor = MatrixCursor(cols) + + val info = Debug.MemoryInfo() + Debug.getMemoryInfo(info) + val pssMb = info.totalPss / 1024 + + val ionMb = readMeminfoKb("IonTotalUsed") / 1024 + val anonMb = readMeminfoKb("AnonPages") / 1024 + val availMb = readMeminfoKb("MemAvailable") / 1024 + + val uptimeSec = (System.currentTimeMillis() - TelemetryHolder.processStartedAt) / 1000 + + cursor.addRow(arrayOf( + Process.myPid(), + uptimeSec, + pssMb, + ionMb, + anonMb, + availMb + )) + return cursor + } + + private fun turnsCursor(): Cursor { + val cols = arrayOf( + "timestamp", "input_kind", "status", "total_ms", + "stt_ms", "llm_ttft_ms", "llm_gen_ms", "llm_tokens", + "time_to_first_audio_ms", "audio_playback_ms" + ) + val cursor = MatrixCursor(cols) + for (t in TelemetryHolder.recentTurns()) { + cursor.addRow(arrayOf( + t.timestamp, t.inputKind, t.status, t.totalMs, + t.sttDurationMs, t.llmTtftMs, t.llmGenerationMs, t.llmTokens, + t.timeToFirstAudioMs, t.audioPlaybackMs + )) + } + return cursor + } + + private fun crashesCursor(): Cursor { + val cols = arrayOf("timestamp", "component", "message") + val cursor = MatrixCursor(cols) + for (c in TelemetryHolder.recentCrashes()) { + cursor.addRow(arrayOf(c.timestamp, c.component, c.message)) + } + return cursor + } + + private fun readMeminfoKb(key: String): Long = try { + File("/proc/meminfo").bufferedReader().useLines { lines -> + lines.firstOrNull { it.startsWith("$key:") } + ?.split(Regex("\\s+"))?.get(1)?.toLong() ?: 0L + } + } catch (_: Exception) { 0L } + + override fun getType(uri: Uri): String? = when (matcher.match(uri)) { + CODE_STATE -> "vnd.android.cursor.item/vnd.com.kazeia.state" + CODE_TURNS -> "vnd.android.cursor.dir/vnd.com.kazeia.turns" + CODE_CRASHES -> "vnd.android.cursor.dir/vnd.com.kazeia.crashes" + CODE_VOICES -> "vnd.android.cursor.dir/vnd.com.kazeia.voices" + CODE_VOICE_ITEM -> "vnd.android.cursor.item/vnd.com.kazeia.voice" + CODE_CONFIG -> "vnd.android.cursor.item/vnd.com.kazeia.config" + CODE_MODELS -> "vnd.android.cursor.dir/vnd.com.kazeia.models" + else -> null + } + + override fun insert(uri: Uri, values: ContentValues?): Uri? = null +}