From 9c424ac6165505acfddf4d20065196fcfc73fba3 Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Fri, 25 Sep 2026 14:52:03 -0400 Subject: [PATCH] fix(tts/kokoro-ane): use fp32-compute KokoroProsody_v2 (#947) The shipped fp16 KokoroProsody miscomputes F0/N over the first ~1-3 s of the utterance for many T_a >= 400 frames (~10 s of audio) on the Core ML CPU and ANE paths (GPU fp16 and fp32 are exact). The corrupted F0 is flat ~120 Hz and N is compressed, so the opening words come out ~12-15 dB quiet (af_heart 'Self-attention' -38 dB vs -24 dB body). Not length-monotonic: 57/99 lengths in T=20..1980 broken for en/ja (identical weights), 56/99 zh. Isolation: same phonemes into PyTorch vs Core ML; swapping torch F0/N into the Core ML chain restores the onset; LSTM alone, palettization and each sub-op exposed as an output are all clean, so this is a fused fp16 kernel issue inside the upsampling AdainResBlk1d. KokoroProsody_v2 keeps the fp16 I/O and int8 palettization, switches compute to fp32: 0/99 broken lengths, worst F0 MAE 0.3 Hz, +~3.7 ms per call at T=450. Renamed (not overwritten) so cached clients re-download. Co-Authored-By: Claude Opus 5.5 (1M context) --- Sources/FluidAudio/ModelNames.swift | 5 ++++- .../Pipeline/KokoroAneSynthesizer+Types.swift | 2 +- .../TTS/KokoroAne/KokoroAneSynthesizerTests.swift | 15 +++++++++++++++ 3 files changed, 20 insertions(+), 2 deletions(-) diff --git a/Sources/FluidAudio/ModelNames.swift b/Sources/FluidAudio/ModelNames.swift index d60fa6198..f77123c7a 100644 --- a/Sources/FluidAudio/ModelNames.swift +++ b/Sources/FluidAudio/ModelNames.swift @@ -1571,7 +1571,10 @@ public enum ModelNames { public static let albert = "KokoroAlbert.mlmodelc" public static let postAlbert = "KokoroPostAlbert.mlmodelc" public static let alignment = "KokoroAlignment.mlmodelc" - public static let prosody = "KokoroProsody.mlmodelc" + // v2: fp32 compute. The fp16 CPU/ANE path corrupts F0/N at the start of + // the utterance for many T_a >= 400 (quiet/garbled onset). Renamed (not + // overwritten) so cached clients re-download. See issue #947. + public static let prosody = "KokoroProsody_v2.mlmodelc" // v2: atan2 phase-correction in the noise-source STFT (removes broad-spectrum // HF noise / "sharpness"). Renamed (not overwritten) so cached clients // re-download. See mobius laishere-coreml docs/trials-and-errors.md. diff --git a/Sources/FluidAudio/TTS/KokoroAne/Pipeline/KokoroAneSynthesizer+Types.swift b/Sources/FluidAudio/TTS/KokoroAne/Pipeline/KokoroAneSynthesizer+Types.swift index 3d16d4ddc..971c62b65 100644 --- a/Sources/FluidAudio/TTS/KokoroAne/Pipeline/KokoroAneSynthesizer+Types.swift +++ b/Sources/FluidAudio/TTS/KokoroAne/Pipeline/KokoroAneSynthesizer+Types.swift @@ -112,7 +112,7 @@ public enum KokoroAneStage: String, CaseIterable, Sendable { case .albert: return "KokoroAlbert.mlmodelc" case .postAlbert: return "KokoroPostAlbert.mlmodelc" case .alignment: return "KokoroAlignment.mlmodelc" - case .prosody: return "KokoroProsody.mlmodelc" + case .prosody: return "KokoroProsody_v2.mlmodelc" // v2: fp32 compute (long-utterance onset fix, #947) case .noise: return "KokoroNoise_v2.mlmodelc" // v2: atan2 phase-correction (HF-noise fix) case .vocoder: return "KokoroVocoder.mlmodelc" case .tail: return "KokoroTail_v2.mlmodelc" // v2: COLA-normalized iSTFT (level fix, #852) diff --git a/Tests/FluidAudioTests/TTS/KokoroAne/KokoroAneSynthesizerTests.swift b/Tests/FluidAudioTests/TTS/KokoroAne/KokoroAneSynthesizerTests.swift index 815f92b2f..5f4efb5cd 100644 --- a/Tests/FluidAudioTests/TTS/KokoroAne/KokoroAneSynthesizerTests.swift +++ b/Tests/FluidAudioTests/TTS/KokoroAne/KokoroAneSynthesizerTests.swift @@ -3,6 +3,21 @@ import XCTest @testable import FluidAudio +/// Stage bundle names must match the download set, including the renamed v2 bundles. +final class KokoroAneStageBundleNameTests: XCTestCase { + + func testStageBundlesMatchRequiredDownloadSet() { + let bundles = Set(KokoroAneStage.allCases.map(\.bundleName)) + XCTAssertEqual(bundles, ModelNames.KokoroAne.requiredCoreMLModels) + } + + func testProsodyUsesFp32ComputeBundle() { + // v1 fp16 Prosody corrupts F0/N at the utterance onset for many T_a >= 400 (#947). + XCTAssertEqual(KokoroAneStage.prosody.bundleName, "KokoroProsody_v2.mlmodelc") + XCTAssertEqual(ModelNames.KokoroAne.prosody, "KokoroProsody_v2.mlmodelc") + } +} + /// Lightweight tests for the pure duration-rounding helper (no models needed). final class KokoroAnePredictedDurationTests: XCTestCase {