diff --git a/models/tts/kokoro-v1.1-zh/coreml/scripts/convert-coreml.py b/models/tts/kokoro-v1.1-zh/coreml/scripts/convert-coreml.py index c191f790..b65f9903 100644 --- a/models/tts/kokoro-v1.1-zh/coreml/scripts/convert-coreml.py +++ b/models/tts/kokoro-v1.1-zh/coreml/scripts/convert-coreml.py @@ -778,7 +778,10 @@ def bench(path, feed, n_runs=10): else: print(f'\n[3/7] Alignment — SKIP (reusing {align_path.name})') - # ═══ 4/7: Prosody (fp16+int8pal) ═══ + # ═══ 4/7: Prosody (fp16 I/O, fp32 compute, int8pal) ═══ + # fp32 compute: the Core ML CPU/ANE fp16 path corrupts F0/N at the start + # of the sequence for many T_a >= 400 (FluidAudio #947); GPU fp16 and + # fp32 are exact. KokoroProsody_v2 on HF. pros_path = OUTDIR / 'KokoroProsody.mlpackage' pros_feed = {"en": en.numpy().astype(np.float16), "style_s": s.numpy().astype(np.float16)} @@ -789,8 +792,8 @@ def bench(path, feed, n_runs=10): with torch.no_grad(): traced = torch.jit.trace(prosody, (en, s), strict=False) ml = ct.convert(traced, - inputs=[ct.TensorType(name="en", shape=(1, 640, T_a_dim), dtype=np.float32), - ct.TensorType(name="style_s", shape=(1, 128), dtype=np.float32)], + inputs=[ct.TensorType(name="en", shape=(1, 640, T_a_dim), dtype=np.float16), + ct.TensorType(name="style_s", shape=(1, 128), dtype=np.float16)], outputs=[ct.TensorType(name="F0"), ct.TensorType(name="N")], convert_to="mlprogram", minimum_deployment_target=ct.target.iOS17, compute_precision=ct.precision.FLOAT32, compute_units=ct.ComputeUnit.ALL) diff --git a/models/tts/kokoro/laishere-coreml/convert-coreml.py b/models/tts/kokoro/laishere-coreml/convert-coreml.py index b14f4269..0479bd72 100644 --- a/models/tts/kokoro/laishere-coreml/convert-coreml.py +++ b/models/tts/kokoro/laishere-coreml/convert-coreml.py @@ -715,12 +715,15 @@ def bench(path, feed, n_runs=10): else: print(f'\n[3/7] Alignment — SKIP (reusing {align_path.name})') - # ═══ 4/7: Prosody (fp16+int8pal) ═══ + # ═══ 4/7: Prosody (fp16 I/O, fp32 compute, int8pal) ═══ + # fp32 compute: the Core ML CPU/ANE fp16 path corrupts F0/N at the start + # of the sequence for many T_a >= 400 (FluidAudio #947); GPU fp16 and + # fp32 are exact. KokoroProsody_v2 on HF. pros_path = OUTDIR / 'KokoroProsody.mlpackage' pros_feed = {"en": en.numpy().astype(np.float16), "style_s": s.numpy().astype(np.float16)} if 'prosody' in selected: - print('\n[4/7] Prosody (fp16+int8pal)...') + print('\n[4/7] Prosody (fp32 compute, int8pal)...') prosody = CoreMLProsodyF0N(model.predictor) prosody.eval() with torch.no_grad(): @@ -730,7 +733,7 @@ def bench(path, feed, n_runs=10): ct.TensorType(name="style_s", shape=(1, 128), dtype=np.float16)], outputs=[ct.TensorType(name="F0"), ct.TensorType(name="N")], convert_to="mlprogram", minimum_deployment_target=ct.target.iOS17, - compute_precision=ct.precision.FLOAT16, compute_units=ct.ComputeUnit.ALL) + compute_precision=ct.precision.FLOAT32, compute_units=ct.ComputeUnit.ALL) ml = cto.palettize_weights(ml, pal_config) ml.save(str(pros_path)) bench(pros_path, pros_feed)