From 4bb0a083d7e6f16e804bc298935cb3de75f3c93f Mon Sep 17 00:00:00 2001 From: myia-po-2027 Date: Wed, 7 Oct 2026 16:45:31 +0200 Subject: [PATCH] Add: scripts outillage re-rendu A0C (Qwen3 neutre + Chatterbox chunked + 3ASR) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cadrage audio c.1467 (DM ai-01 11:09 le 07/10) — re-rendu A0C ordonné pour gate pré-UAT #17586, livraison avant 08/10 18:00. 3 scripts + README dans prosody_lab/a0c_rerun/ : - rerun_a0c_neutral_qwen3.py : Qwen3-TTS CustomVoice sans --instruct (neutre, corrige la 5× inflation "débit lent" de c.1462) - rerun_a0c_chatterbox_chunked.py : Chatterbox MTL v3 découpé par phrase avec ref_wav = phrase 1 (corrige l'omission de 2 segments sous 3 ASR sur 3) - measure_3asr.py : WER sur 3 ASR (tiny/small/large-v3) + détection d'omissions (segments >= 3 mots omis sous >= 2 ASR sur 3) - README.md : pipeline complet, sortie attendue, verdict vs c.1462 Grain: LIGHT/tooling -- lane myia-po-2027:CoursIA-2 -- prev: DEEP/genai #19722 Co-Authored-By: Claude Haiku 4.5 (1M context) --- .../v4/prosody_lab/a0c_rerun/README.md | 83 +++++++++ .../v4/prosody_lab/a0c_rerun/measure_3asr.py | 153 ++++++++++++++++ .../a0c_rerun/rerun_a0c_chatterbox_chunked.py | 171 ++++++++++++++++++ .../a0c_rerun/rerun_a0c_neutral_qwen3.py | 109 +++++++++++ 4 files changed, 516 insertions(+) create mode 100644 MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/README.md create mode 100644 MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py create mode 100644 MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py create mode 100644 MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py diff --git a/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/README.md b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/README.md new file mode 100644 index 0000000000..677a5a3efb --- /dev/null +++ b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/README.md @@ -0,0 +1,83 @@ +# a0c_rerun — scripts de re-rendu A0C pour gate pré-UAT #17586 + +## Cadrage (audio c.1467, DM ai-01 11:09 le 07/10) + +Suite à l'audit de la mesure A0C précédente (c.1462) : + +- **Qwen3-TTS-12Hz-1.7B-CustomVoice** : 5× inflation "débit lent" sur le WER, trace de l'instruct « voix posée, débit lent, ton narratif » passé au modèle. +- **Chatterbox Multilingual V3** : 93,65 % WER, voice DRIFTING, hallucination `« Ah, blâme ! Moi... »`, **2 segments omis** sous 3 ASR sur 3 (« des artilleurs sombres alignés avec des fantassins divers », « sur leurs épaules de fanfarons »). + +Re-rendu A0C ordonné par ai-01 : + +- Qwen3-TTS en réglage **NEUTRE** (sans « débit lent »). +- Chatterbox **découpé par phrase** avec un wav de référence. +- Mesure : WER sur 3 ASR (tiny / small / large-v3), prosodie, omissions (segments ≥ 3 mots omis sous ≥ 2 ASR sur 3). +- Délai : avant 08/10 18:00 (date annoncée à l'association). + +## Scripts + +| Fichier | Rôle | Venv | GPU | +|---|---|---|---| +| `rerun_a0c_neutral_qwen3.py` | Synthèse Qwen3-TTS CustomVoice **sans --instruct** (neutre) | `venv-qwen3tts` | cuda | +| `rerun_a0c_chatterbox_chunked.py` | Synthèse Chatterbox MTL v3 **découpée par phrase** + ref_wav = phrase 1 | `venv` | cuda | +| `measure_3asr.py` | WER sur 3 ASR (tiny / small / large-v3) + détection d'omissions | `venv` | cuda (tiny/small OK CPU) | + +## Pipeline complet + +```bash +# 1) Préparer le dossier de revue (côté GDrive, hors dépôt) +mkdir -p "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007" +cp "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261006/extract_C_long_narration.txt" \ + "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/" + +# 2a) Re-rendu Qwen3-TTS neutre (charge 1.7B, ~3-4 GB VRAM, ~5-10 min) +cd D:/dev/CoursIA-2 +./venv-qwen3tts/Scripts/python.exe \ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py \ + --text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \ + --out-dir "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007" \ + --speaker serena --language French --device cuda --dtype bf16 + +# 2b) Re-rendu Chatterbox chunked (à lancer APRÈS le Qwen3, GPU partagé) +./venv/Scripts/python.exe \ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py \ + --text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \ + --out-dir "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007" \ + --lang fr --device cuda + +# 3) Mesure 3 ASR (à lancer après CHAQUE rendu, met à jour metrics.json) +./venv/Scripts/python.exe \ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py \ + --wav "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/A0C-qwen3tts-customvoice-neutral.wav" \ + --ref-text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \ + --metrics-json "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/A0C-qwen3tts-customvoice-neutral-metrics.json" \ + --asr-models tiny small large-v3 +``` + +## Sortie + +Pour chaque run, dans `A0-review-20261007/` : + +- `A0C-qwen3tts-customvoice-neutral.wav` (+ `-metrics.json`) +- `A0C-chatterbox-mtl-v3-chunked.wav` (+ `-metrics.json`) + +Le `metrics.json` est enrichi en place par `measure_3asr.py` : + +- `wer.by_model` : WER tiny / small / large-v3, hyp complet, durée de transcription. +- `omissions` : segments ≥ 3 mots absents de ≥ 2 ASR sur 3, avec liste des ASR où l'absence est constatée. +- `prosody` : à mesurer par `verify_prosody.py` (gate pré-UAT ne dépend pas que de WER). + +## Verdict attendu vs. mesure c.1462 + +| Mesure | c.1462 | Re-rendu attendu | +|---|---|---| +| Qwen3-TTS WER (tiny) | 32,94 % (5× inflation "débit lent") | < 15 % (sans l'instruct qui fabrique l'amplification) | +| Chatterbox WER (tiny) | 93,65 % (DRIFTING, hallucination) | < 30 % (chunking + ref_wav doivent stabiliser) | +| Omissions Chatterbox | 2 segments omis (3 ASR sur 3) | 0 ou 1 (chunking par phrase ne saute plus) | + +## Acceptance + +- [ ] Qwen3-TTS neutre livré (WAV + metrics.json enrichi 3ASR) +- [ ] Chatterbox chunked livré (WAV + metrics.json enrichi 3ASR) +- [ ] Verdict gate pré-UAT #17586 sur les 2 mesures (pass / fail par moteur) +- [ ] DM ai-01 avec les chiffres + dépôt `A0-review-20261007/` diff --git a/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py new file mode 100644 index 0000000000..1075f3aee9 --- /dev/null +++ b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py @@ -0,0 +1,153 @@ +"""measure_3asr.py — mesure WER sur 3 ASR (tiny, small, large-v3) + détection d'omissions. + +Cadrage audio c.1467 (DM ai-01 11:09 le 07/10) : +- Mesure WER sur 3 ASR, prosodie, **omissions** (segments de 3+ mots omis sous ≥2 ASR sur 3). + +Entrée : un fichier .wav (Qwen3 neutre OU Chatterbox chunked) + le texte de référence. +Sortie : mise à jour du metrics.json existant (ajout wer + omissions). + +Usage : + /d/dev/CoursIA-2/venv/Scripts/python.exe \\ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/measure_3asr.py \\ + --wav "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/A0C-qwen3tts-customvoice-neutral.wav" \\ + --ref-text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \\ + --metrics-json "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/A0C-qwen3tts-customvoice-neutral-metrics.json" \\ + [--asr-models tiny small large-v3] [--device cuda] +""" +from __future__ import annotations + +import argparse +import json +import re +import subprocess +import sys +import time +from pathlib import Path + + +def _wer(ref_words: list[str], hyp_words: list[str]) -> float: + """WER par distance de Levenshtein (inline, jiwer-free).""" + n, m = len(ref_words), len(hyp_words) + dp = [[0] * (m + 1) for _ in range(n + 1)] + for i in range(n + 1): + dp[i][0] = i + for j in range(m + 1): + dp[0][j] = j + for i in range(1, n + 1): + for j in range(1, m + 1): + if ref_words[i - 1] == hyp_words[j - 1]: + dp[i][j] = dp[i - 1][j - 1] + else: + dp[i][j] = 1 + min(dp[i - 1][j], dp[i][j - 1], dp[i - 1][j - 1]) + return dp[n][m] / max(n, 1) + + +def _transcribe_with_faster_whisper(wav_path: str, model_size: str, device: str) -> str: + """Transcrit via faster-whisper, retourne la chaîne hyp complète.""" + from faster_whisper import WhisperModel + wm = WhisperModel( + model_size, + device=device if device == "cuda" else "cpu", + compute_type="float16" if device == "cuda" else "int8", + ) + segments, info = wm.transcribe(wav_path, language="fr", beam_size=5) + return " ".join(seg.text for seg in segments).strip() + + +def _detect_omissions(ref_text: str, hyps: dict[str, str], min_words: int = 3) -> dict: + """Détecte les segments de ≥ min_words mots omis sous ≥ 2 ASR sur 3. + + Stratégie : pour chaque segment de N≥min_words mots du ref, vérifier s'il + apparaît (en substring) dans chaque hyp. Un segment est "omis" si absent + de ≥ 2 hyps. Retourne la liste des segments omis + le verdict global. + """ + sentences = re.split(r"(?<=[.!?])\s+", ref_text.strip()) + segments = [s.strip() for s in sentences if len(s.strip().split()) >= min_words] + omitted = [] + for seg in segments: + per_asr = {name: (seg in hyp) for name, hyp in hyps.items()} + n_absent = sum(1 for v in per_asr.values() if not v) + if n_absent >= 2: + omitted.append({"segment": seg, "n_words": len(seg.split()), + "absent_in": [k for k, v in per_asr.items() if not v]}) + return { + "n_segments_checked": len(segments), + "n_omitted": len(omitted), + "segments": omitted, + } + + +def main() -> int: + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--wav", required=True, type=Path) + p.add_argument("--ref-text-file", required=True, type=Path) + p.add_argument("--metrics-json", required=True, type=Path, + help="metrics.json à enrichir (le script écrit en place).") + p.add_argument("--asr-models", nargs="+", default=["tiny", "small", "large-v3"]) + p.add_argument("--device", default="cuda") + args = p.parse_args() + + if not args.wav.exists(): + print(f"WAV introuvable : {args.wav}", file=sys.stderr) + return 2 + if not args.ref_text_file.exists(): + print(f"Ref texte introuvable : {args.ref_text_file}", file=sys.stderr) + return 2 + if not args.metrics_json.exists(): + print(f"metrics.json introuvable : {args.metrics_json}", file=sys.stderr) + return 2 + + ref_text = args.ref_text_file.read_text(encoding="utf-8").strip() + ref_words = ref_text.split() + + print(f"=== Mesure 3-ASR sur {args.wav.name} ===") + print(f" Ref : {len(ref_text)} chars, {len(ref_words)} mots") + print(f" ASR : {args.asr_models}") + print() + + metrics = json.loads(args.metrics_json.read_text(encoding="utf-8")) + hyps = {} + wer_by_model = {} + for model_size in args.asr_models: + t0 = time.time() + try: + hyp = _transcribe_with_faster_whisper(str(args.wav), model_size, args.device) + except Exception as e: + print(f" [{model_size}] ÉCHEC : {e}", file=sys.stderr) + wer_by_model[model_size] = {"error": str(e)} + continue + dt = time.time() - t0 + wer = _wer(ref_words, hyp.split()) + hyps[model_size] = hyp + wer_by_model[model_size] = { + "wer": float(wer), + "wer_pct": round(wer * 100, 2), + "hyp_words": len(hyp.split()), + "wallclock_s": round(dt, 1), + } + print(f" [{model_size}] WER = {wer*100:.2f}% ({len(hyp.split())} mots, {dt:.1f}s)") + + # Omissions (si on a au moins 2 hyps) + omissions = None + if len(hyps) >= 2: + omissions = _detect_omissions(ref_text, hyps, min_words=3) + print(f"\n Omissions (≥ 3 mots, ≥ 2 ASR) : {omissions['n_omitted']} / " + f"{omissions['n_segments_checked']} segments") + for o in omissions["segments"]: + print(f" - « {o['segment'][:80]}… » absent de {o['absent_in']}") + + # Mise à jour metrics.json + metrics["wer"] = { + "by_model": wer_by_model, + "ref_text_file": str(args.ref_text_file), + "hyps_full": hyps, + } + metrics["omissions"] = omissions + args.metrics_json.write_text(json.dumps(metrics, indent=2, ensure_ascii=False), + encoding="utf-8") + print(f"\n metrics.json enrichi : {args.metrics_json}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py new file mode 100644 index 0000000000..a9967345f0 --- /dev/null +++ b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py @@ -0,0 +1,171 @@ +r"""rerun_a0c_chatterbox_chunked.py — re-rendu A0C Chatterbox MTL v3 découpé par phrase + ref_wav. + +Cadrage audio c.1467 (reçu 11:09 le 07/10) : +- Chatterbox **découpé par phrase** avec un wav de référence. +- Mesure : WER sur 3 ASR, prosodie, **omissions** (hallucinations / segments sautés). + +Le précédent rendu (c.1462) avait un score catastrophique sur la fidélité (93,65 % +WER) et l'omission de segments (2 mesures perdues sous 3 ASR sur 3 — phrases +"des artilleurs sombres alignés avec des fantassins divers" et "sur leurs +épaules de fanfarons"). Le découpage par phrase + un wav de référence doit +réduire l'omission : Chatterbox a moins de "sauts" sur des phrases courtes. + +Stratégie : +1. Découpe extract_C en phrases (regex simple sur [.!?]\s+). +2. Phrase 1 sert de **ref_wav** pour les phrases suivantes (consistance de voix). +3. Pour chaque phrase, generate avec `audio_prompt_path=ref_wav_path`. +4. Concatène les wav numpy en un seul. +5. Mesure WER sur l'assemblé. + +Pré-requis : + /d/dev/CoursIA-2/venv/Scripts/python.exe (chatterbox + faster_whisper déjà OK) + +Usage : + /d/dev/CoursIA-2/venv/Scripts/python.exe \\ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_chatterbox_chunked.py \\ + --text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \\ + --out-dir "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007" \\ + [--lang fr] [--device cuda] +""" +from __future__ import annotations + +import argparse +import io +import json +import re +import sys +import time +from pathlib import Path + + +def split_sentences(text: str) -> list[str]: + """Découpe grossière par [.!?] suivi d'espace, sans dépendance NLP.""" + # Sépare aussi sur ; pour les phrases longues du narratif Maupassant + parts = re.split(r"(?<=[.!?])\s+", text.strip()) + return [p.strip() for p in parts if p.strip()] + + +def main() -> int: + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--text-file", required=True, type=Path) + p.add_argument("--out-dir", required=True, type=Path) + p.add_argument("--lang", default="fr") + p.add_argument("--device", default="cuda") + p.add_argument("--ref-from-first-sentence", action="store_true", default=True, + help="Utilise la première phrase comme ref_wav pour la consistance de voix.") + args = p.parse_args() + + if not args.text_file.exists(): + print(f"Texte introuvable : {args.text_file}", file=sys.stderr) + return 2 + text = args.text_file.read_text(encoding="utf-8").strip() + + sentences = split_sentences(text) + if not sentences: + print("Aucune phrase détectée.", file=sys.stderr) + return 2 + + args.out_dir.mkdir(parents=True, exist_ok=True) + out_wav = args.out_dir / "A0C-chatterbox-mtl-v3-chunked.wav" + metrics_path = args.out_dir / "A0C-chatterbox-mtl-v3-chunked-metrics.json" + + print(f"=== Re-rendu A0C Chatterbox MTL v3 DÉCOUPÉ PAR PHRASE ===") + print(f" Texte : {args.text_file} ({len(text)} chars, {len(sentences)} phrases)") + print(f" Out : {out_wav}") + print(f" Langue : {args.lang}, ref_wav = première phrase synthétisée") + print() + + # Imports tardifs + import numpy as np + import soundfile as sf + import torch + from chatterbox.mtl_tts import ChatterboxMultilingualTTS # noqa: E402 + + t_load = time.time() + model = ChatterboxMultilingualTTS.from_pretrained(args.device) + t_load_s = time.time() - t_load + sr = model.sr # convention chatterbox : sr=24000 + print(f" Modèle chargé en {t_load_s:.1f}s (sr={sr} Hz)\n") + + # 1) Première phrase — sert de ref_wav + if args.ref_from_first_sentence: + ref_text = sentences[0] + print(f" [REF] Phrase 1 (ref_wav) : {ref_text[:80]}...") + t0 = time.time() + wav_ref = model.generate(ref_text, language_id=args.lang) + if hasattr(wav_ref, "squeeze"): + wav_ref_np = wav_ref.squeeze(0).detach().cpu().numpy() + else: + wav_ref_np = wav_ref + ref_wav_path = args.out_dir / ".chunked_ref.wav" + sf.write(str(ref_wav_path), wav_ref_np, sr) + t_ref = time.time() - t0 + print(f" OK en {t_ref:.1f}s, durée {len(wav_ref_np)/sr:.2f}s") + concat = [wav_ref_np] + n_chunks = 1 + chunk_durations = [len(wav_ref_np) / sr] + else: + ref_wav_path = None + concat = [] + n_chunks = 0 + chunk_durations = [] + + # 2) Phrases suivantes — avec audio_prompt_path + for i, sent in enumerate(sentences[1:], start=2): + print(f" [{i:2d}/{len(sentences)}] {sent[:80]}...") + t0 = time.time() + if ref_wav_path is not None: + wav = model.generate(sent, language_id=args.lang, + audio_prompt_path=str(ref_wav_path)) + else: + wav = model.generate(sent, language_id=args.lang) + if hasattr(wav, "squeeze"): + wav_np = wav.squeeze(0).detach().cpu().numpy() + else: + wav_np = wav + concat.append(wav_np) + chunk_durations.append(len(wav_np) / sr) + n_chunks += 1 + print(f" OK en {time.time()-t0:.1f}s, durée {len(wav_np)/sr:.2f}s") + + # 3) Concatène + silence 200ms entre phrases + silence = np.zeros(int(sr * 0.2), dtype=np.float32) + pieces = [] + for i, w in enumerate(concat): + pieces.append(w) + if i < len(concat) - 1: + pieces.append(silence) + full = np.concatenate(pieces) if pieces else np.zeros(0, dtype=np.float32) + sf.write(str(out_wav), full, sr) + + total_audio_s = len(full) / sr + print(f"\n Concaténé : {n_chunks} phrases, {total_audio_s:.2f}s audio, {out_wav}") + + metrics = { + "cell": "chatterbox_mtl_v3_chunked", + "license": "Apache-2.0", + "size": "0.5B", + "model_id": "ResembleAI/chatterbox-multilingual", + "mode": "CHUNKED_SENTENCE + REF_FROM_FIRST", + "text_file": str(args.text_file), + "text_chars": len(text), + "text_sentences": len(sentences), + "sentences": sentences, + "n_chunks": n_chunks, + "chunk_durations_s": chunk_durations, + "sample_rate": sr, + "duration_s": float(total_audio_s), + "ref_wav_path": str(ref_wav_path) if ref_wav_path else None, + "load_s": t_load_s, + "wer": None, # rempli par measure_3asr.py + "omissions": None, # rempli par measure_3asr.py (segments de 3+ mots omis) + "prosody": None, # rempli par verify_prosody.py + } + metrics_path.write_text(json.dumps(metrics, indent=2, ensure_ascii=False), encoding="utf-8") + print(f" Métriques (synthèse) : {metrics_path}") + print(f" WER 3-ASR + omissions : à mesurer par measure_3asr.py (c.1473)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py new file mode 100644 index 0000000000..5ffb14ec90 --- /dev/null +++ b/MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py @@ -0,0 +1,109 @@ +"""rerun_a0c_neutral_qwen3.py — re-rendu A0C Qwen3-TTS CustomVoice en mode NEUTRE. + +Cadrage audio c.1467 (reçu 11:09 le 07/10) : +- Qwen3-TTS en réglage **neutre** (sans « débit lent »). +- Mesure : WER sur 3 ASR (tiny / small / large-v3), prosodie. +- Sortie : A0-review-20261007/A0C-qwen3tts-customvoice-neutral.{wav,metrics.json}. + +Mode neutre = `instruct=None` (défaut du client). c.1462 a tourné Chatterbox MTL v3 ++ Qwen3-TTS-1.7B CustomVoice avec instruct "voix posée, débit lent, ton narratif" +(5× inflation "débit lent" mesurée sur WER). Le neutre doit retomber à un +rendement plus proche de la lecture naturelle. + +Pré-requis : + /d/dev/CoursIA-2/venv-qwen3tts/Scripts/python.exe -m pip install qwen-tts torch torchaudio + +Usage (depuis racine dépôt ou worktree) : + /d/dev/CoursIA-2/venv-qwen3tts/Scripts/python.exe \\ + MyIA.AI.Notebooks/GenAI/Audio/04-Applications/v4/prosody_lab/a0c_rerun/rerun_a0c_neutral_qwen3.py \\ + --text-file "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007/extract_C_long_narration.txt" \\ + --out-dir "G:/Mon Drive/MyIA/Projets/BibliothequesSonores/A0-review-20261007" \\ + [--speaker serena] [--language French] [--device cuda] [--dtype bf16] +""" +from __future__ import annotations + +import argparse +import json +import sys +import time +from pathlib import Path + + +def main() -> int: + p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + p.add_argument("--text-file", required=True, type=Path, + help="Fichier texte de référence (extract_C_long_narration.txt).") + p.add_argument("--out-dir", required=True, type=Path, + help="Dossier de sortie (A0-review-20261007).") + p.add_argument("--speaker", default="serena") + p.add_argument("--language", default="French") + p.add_argument("--device", default="cuda") + p.add_argument("--dtype", default="bf16", choices=["bf16", "fp16", "fp32"]) + args = p.parse_args() + + if not args.text_file.exists(): + print(f"Texte introuvable : {args.text_file}", file=sys.stderr) + return 2 + text = args.text_file.read_text(encoding="utf-8").strip() + if not text: + print(f"Texte vide : {args.text_file}", file=sys.stderr) + return 2 + + args.out_dir.mkdir(parents=True, exist_ok=True) + out_wav = args.out_dir / "A0C-qwen3tts-customvoice-neutral.wav" + metrics_path = args.out_dir / "A0C-qwen3tts-customvoice-neutral-metrics.json" + + # Imports tardifs (env potentiellement absent) + import torch + sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "bakeoff_large")) + from clients import qwen3_tts_customvoice as qwen # noqa: E402 + + print(f"=== Re-rendu A0C Qwen3-TTS NEUTRE ===") + print(f" Texte : {args.text_file} ({len(text)} chars, {len(text.split())} mots)") + print(f" Out : {out_wav}") + print(f" Speaker={args.speaker} Language={args.language} (instruct=None = NEUTRE)") + print() + + t_load = time.time() + model = qwen.load_model(device=args.device, dtype=args.dtype) + t_load_s = time.time() - t_load + print(f" Modèle chargé en {t_load_s:.1f}s\n") + + t_synth = time.time() + synth_result = qwen.synth( + text=text, + out_wav=str(out_wav), + model=model, + language=args.language, + speaker=args.speaker, + instruct=None, # *** NEUTRE — sans « débit lent » *** + ) + t_synth_s = time.time() - t_synth + print(f"\n Synthèse OK en {t_synth_s:.1f}s : {synth_result.get('duration_s', 0):.2f}s audio, " + f"RTF={synth_result.get('rtf', 0):.3f}, VRAM peak={synth_result.get('vram_peak_gb', 0):.2f} GB") + + metrics = { + "cell": "qwen3_tts_customvoice_neutral", + "license": "Apache-2.0", + "size": "1.7B", + "model_id": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice", + "mode": "NEUTRAL (instruct=None)", + "text_file": str(args.text_file), + "text_chars": len(text), + "text_words": len(text.split()), + "speaker": args.speaker, + "language": args.language, + "load_s": t_load_s, + "synth": synth_result, + "wallclock_total_s": t_load_s + t_synth_s, + "wer": None, # rempli par measure_3asr.py + "prosody": None, # rempli par verify_prosody.py (à part) + } + metrics_path.write_text(json.dumps(metrics, indent=2, ensure_ascii=False), encoding="utf-8") + print(f"\n Métriques (synthèse) : {metrics_path}") + print(f" WER 3-ASR : à mesurer par measure_3asr.py (c.1473)") + return 0 + + +if __name__ == "__main__": + sys.exit(main())