Files
omnivoice-cpp/tests/debug-tts-cossim.py
T
2026-07-05 18:11:23 +07:00

409 lines
17 KiB
Python
Executable File

#!/usr/bin/env python3
"""Cossim debug : C++ omnivoice-tts vs Python OmniVoice on voice design.
Inputs (relative to CWD = tests/) :
../examples/prompt.txt target text fed to both pipelines
Both sides run with :
instruct=male, language=English, seed=42, F32 weights, no pre or post
process. Defaults match : num_step=32, guidance_scale=2.0, t_shift=0.1,
layer_penalty_factor=5.0, position_temperature=5.0, class_temperature=0.0.
Dumps land in cpp/ (C++) and python/ (Python). The script compares each
matching .bin pair via cosine similarity over the f32 payload, plus exact
match rate for tensors that originated as int (mg-tokens). All paths are
relative, no absolute paths anywhere.
"""
import argparse
import os
import struct
import subprocess
import sys
# Strict F32 matmul on both sides. NVIDIA_TF32_OVERRIDE=0 forces full FP32
# mantissa in cuBLAS for both PyTorch and the C++ child via inheritance.
# Must be set BEFORE torch imports so the cuBLAS handle reads it on init.
os.environ["NVIDIA_TF32_OVERRIDE"] = "0"
import numpy as np
import soundfile as sf
import torch
# Belt and suspenders : disable PyTorch's own TF32 toggles too. Some code
# paths bypass NVIDIA_TF32_OVERRIDE through cudnn or torch internal flags.
torch.backends.cuda.matmul.allow_tf32 = False
torch.backends.cudnn.allow_tf32 = False
torch.backends.cuda.matmul.allow_fp16_reduced_precision_reduction = False
torch.backends.cuda.matmul.allow_bf16_reduced_precision_reduction = False
torch.set_float32_matmul_precision("highest")
from omnivoice import OmniVoice
from omnivoice.utils.common import fix_random_seed
BIN = "../build/omnivoice-tts"
MODEL_LM_T = "../models/omnivoice-base-{q}.gguf"
MODEL_CDC_T = "../models/omnivoice-tokenizer-{q}.gguf"
CKPT = "../checkpoints/OmniVoice"
DUMP_CPP = "cpp"
DUMP_PT = "python"
def ensure_dir(path):
os.makedirs(path, exist_ok=True)
def save_dump(path, data):
"""Write a tensor in the C++ debug.h format :
[ndim:i32] [shape:i32 x ndim] [data:f32 x numel]
"""
if isinstance(data, torch.Tensor):
data = data.detach().to(torch.float32).cpu().numpy()
data = np.ascontiguousarray(data.astype(np.float32))
shape = data.shape
with open(path, "wb") as f:
f.write(struct.pack("i", len(shape)))
for s in shape:
f.write(struct.pack("i", s))
f.write(data.tobytes())
def load_dump(path):
"""Inverse of save_dump : returns (data:f32 numpy, shape:tuple)."""
raw = np.fromfile(path, dtype=np.uint8)
ndim = int(np.frombuffer(raw[0:4], dtype=np.int32)[0])
shape = tuple(int(x) for x in np.frombuffer(raw[4:4 + 4 * ndim], dtype=np.int32))
body = np.frombuffer(raw[4 + 4 * ndim:], dtype=np.float32)
return body.reshape(shape), shape
def cos(a, b):
a = a.astype(np.float64).ravel()
b = b.astype(np.float64).ravel()
n = min(len(a), len(b))
a, b = a[:n], b[:n]
d = float(np.linalg.norm(a) * np.linalg.norm(b))
return float(np.dot(a, b) / d) if d > 1e-10 else 0.0
def stft_cos(a, b, win=2048, hop=512):
# STFT magnitude cosine. Drops phase, so a constant time shift between
# the two waveforms does not collapse the score. The plain cos() on
# raw samples falls to ~0 the moment chunks land a few samples apart.
a = a.astype(np.float64).ravel()
b = b.astype(np.float64).ravel()
n = min(len(a), len(b))
a, b = a[:n], b[:n]
window = np.hanning(win)
frames = (n - win) // hop + 1
if frames <= 0:
return 0.0
sa = np.zeros((frames, win // 2 + 1))
sb = np.zeros((frames, win // 2 + 1))
for i in range(frames):
s = i * hop
sa[i] = np.abs(np.fft.rfft(a[s:s + win] * window))
sb[i] = np.abs(np.fft.rfft(b[s:s + win] * window))
return cos(sa.ravel(), sb.ravel())
def install_hooks(model, dump_dir):
# First call to _prepare_embed_inputs at step 0 returns the input embedding
# right before layer 0 of the LLM, mirroring the C++ inputs_embeds dump.
# Also captures the raw input_ids row k=0 for cond and uncond, so any
# token-level divergence localizes upstream of the embed lookup.
seen_embed = {"done": False}
orig_prepare = model._prepare_embed_inputs
def hooked_prepare(input_ids, audio_mask):
out = orig_prepare(input_ids, audio_mask)
if not seen_embed["done"] and out.dim() == 3 and out.shape[0] >= 2:
cond = out[0].detach().to(torch.float32).cpu().numpy()
uncond = out[1].detach().to(torch.float32).cpu().numpy()
save_dump(os.path.join(dump_dir, "lm-hidden-step0-cond-embed.bin"), cond)
save_dump(os.path.join(dump_dir, "lm-hidden-step0-uncond-embed.bin"), uncond)
# Style and text tokens duplicate across all K codebooks so k=0
# carries the full sequence for diagnostic. Cast to f32 keeps the
# debug.h binary format identical on both sides.
cond_ids = input_ids[0, 0, :].detach().to(torch.float32).cpu().numpy()
uncond_ids = input_ids[1, 0, :].detach().to(torch.float32).cpu().numpy()
save_dump(os.path.join(dump_dir, "prompt-cond-ids.bin"), cond_ids)
save_dump(os.path.join(dump_dir, "prompt-uncond-ids.bin"), uncond_ids)
seen_embed["done"] = True
return out
model._prepare_embed_inputs = hooked_prepare
# Bisection : dump cond and uncond hidden states after a few layers so a
# mismatch can be localized within the 28 layer Qwen3 stack.
bisect_layers = [0, 6, 13, 20]
seen_layers = {idx: False for idx in bisect_layers}
def make_layer_hook(layer_idx):
def hook(module, inputs, output):
if seen_layers[layer_idx]:
return
h = output[0] if isinstance(output, tuple) else output
if h.dim() == 3 and h.shape[0] >= 2:
cond = h[0].detach().to(torch.float32).cpu().numpy()
uncond = h[1].detach().to(torch.float32).cpu().numpy()
save_dump(os.path.join(dump_dir, f"lm-hidden-step0-cond-l{layer_idx}.bin"), cond)
save_dump(os.path.join(dump_dir, f"lm-hidden-step0-uncond-l{layer_idx}.bin"), uncond)
seen_layers[layer_idx] = True
return hook
for layer_idx in bisect_layers:
model.llm.layers[layer_idx].register_forward_hook(make_layer_hook(layer_idx))
# First call to audio_heads corresponds to step 0 of the MaskGIT loop.
# The input to audio_heads is the final hidden state, shape [B, S, D],
# mirroring what the C++ side reads back via dump_hidden_dir before the
# lm_head matmul. We dump cond (b=0) and uncond (b=1) separately.
seen_hidden = {"done": False}
def pre_audio_heads(module, inputs):
if not seen_hidden["done"]:
h = inputs[0]
if h.dim() == 3 and h.shape[0] >= 2:
cond = h[0].detach().to(torch.float32).cpu().numpy()
uncond = h[1].detach().to(torch.float32).cpu().numpy()
save_dump(os.path.join(dump_dir, "lm-hidden-step0-cond.bin"), cond)
save_dump(os.path.join(dump_dir, "lm-hidden-step0-uncond.bin"), uncond)
seen_hidden["done"] = True
model.audio_heads.register_forward_pre_hook(pre_audio_heads)
# First call to _predict_tokens_with_scoring corresponds to step 0 of the
# MaskGIT loop. Capture cond and uncond logits in [K, T, V] layout (squeeze
# the batch axis to match the C++ dump shape). Replicate the predict math
# locally so log_probs / pred_tokens / scores can all be dumped from a
# single source of truth on the Python side.
seen = {"step0": False, "mg_tokens": False, "audio": False}
orig_pred = model._predict_tokens_with_scoring
def hooked_pred(c_logits, u_logits, gen_config):
pred_tokens, scores = orig_pred(c_logits, u_logits, gen_config)
if not seen["step0"]:
if gen_config.guidance_scale != 0:
c_lp = torch.nn.functional.log_softmax(c_logits, dim=-1)
u_lp = torch.nn.functional.log_softmax(u_logits, dim=-1)
log_probs = torch.log_softmax(
c_lp + gen_config.guidance_scale * (c_lp - u_lp), dim=-1)
else:
log_probs = torch.nn.functional.log_softmax(c_logits, dim=-1)
log_probs = log_probs.clone()
log_probs[..., model.config.audio_mask_id] = float("-inf")
c = c_logits.detach().to(torch.float32).cpu().numpy()
u = u_logits.detach().to(torch.float32).cpu().numpy()
if c.ndim == 4:
c = c[0]
if u.ndim == 4:
u = u[0]
save_dump(os.path.join(dump_dir, "lm-logits-step0-cond.bin"), c)
save_dump(os.path.join(dump_dir, "lm-logits-step0-uncond.bin"), u)
lp_arr = log_probs.detach().to(torch.float32).cpu().numpy()
if lp_arr.ndim == 4:
lp_arr = lp_arr[0]
save_dump(os.path.join(dump_dir, "mg-log-probs-step0.bin"), lp_arr)
pt_arr = pred_tokens.detach().to(torch.float32).cpu().numpy()
sc_arr = scores.detach().to(torch.float32).cpu().numpy()
if pt_arr.ndim == 3:
pt_arr = pt_arr[0]
if sc_arr.ndim == 3:
sc_arr = sc_arr[0]
save_dump(os.path.join(dump_dir, "mg-pred-tokens-step0.bin"), pt_arr)
save_dump(os.path.join(dump_dir, "mg-scores-step0.bin"), sc_arr)
seen["step0"] = True
return pred_tokens, scores
model._predict_tokens_with_scoring = hooked_pred
orig_generate = model._generate_iterative
def hooked_generate(task, gen_config):
out = orig_generate(task, gen_config)
# out is a list of (K, T_i) long tensors, one per batch item.
# In chunked mode this hook fires once per chunk; only the first
# call mirrors the chunk 0 dump on the C++ side.
if not seen["mg_tokens"]:
save_dump(os.path.join(dump_dir, "mg-tokens.bin"), out[0])
seen["mg_tokens"] = True
return out
model._generate_iterative = hooked_generate
orig_decode = model.audio_tokenizer.decode
def hooked_decode(*args, **kwargs):
out = orig_decode(*args, **kwargs)
# The audio tokenizer returns either a tensor or a wrapper holding
# audio_values shape [B, C, N]. Unwrap and dump the first item mono.
# Same first-chunk guard as the mg-tokens hook above.
if seen["audio"]:
return out
wav = getattr(out, "audio_values", out)
if isinstance(wav, torch.Tensor):
arr = wav.detach().to(torch.float32).cpu().numpy()
else:
arr = np.asarray(wav, dtype=np.float32)
if arr.ndim == 3:
arr = arr[0, 0]
elif arr.ndim == 2:
arr = arr[0]
save_dump(os.path.join(dump_dir, "output-audio.bin"), arr)
seen["audio"] = True
return out
model.audio_tokenizer.decode = hooked_decode
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--prompt", default="../examples/prompt.txt")
ap.add_argument("--seed", type=int, default=42)
ap.add_argument("--instruct", default="male, young adult, moderate pitch")
ap.add_argument("--lang", default="English")
ap.add_argument("--duration", type=float, default=None)
ap.add_argument("--quant", default="F32",
help="quantization suffix for GGUF (default: F32, e.g. BF16, Q8_0, Q4_K_M)")
ap.add_argument("--out-cpp", default="cpp/tts-cpp.wav")
ap.add_argument("--out-pt", default="python/tts-python.wav")
args = ap.parse_args()
model_lm = MODEL_LM_T.format(q=args.quant)
model_cdc = MODEL_CDC_T.format(q=args.quant)
for p in (model_lm, model_cdc):
if not os.path.isfile(p):
print(f"[Error] GGUF not found: {p}")
sys.exit(1)
print(f"[Quant] {args.quant} -> {model_lm} + {model_cdc}")
ensure_dir(DUMP_CPP)
ensure_dir(DUMP_PT)
os.makedirs(os.path.dirname(args.out_cpp) or ".", exist_ok=True)
with open(args.prompt, "r", encoding="utf-8") as f:
text = f.read().strip()
print(f"[Input] Prompt: {len(text)} chars: {text[:60]}{'...' if len(text) > 60 else ''}")
print(f"[Input] Instruct: {args.instruct}")
print(f"[Input] Language: {args.lang}")
print(f"[Input] Seed: {args.seed}")
# Python reference path : F32, voice design male, no pre or post process.
fix_random_seed(args.seed)
device = "cuda" if torch.cuda.is_available() else "cpu"
model = OmniVoice.from_pretrained(
CKPT,
torch_dtype=torch.float32,
attn_implementation="eager",
).to(device).eval()
install_hooks(model, DUMP_PT)
gen_kwargs = dict(
text=text,
language=args.lang,
instruct=args.instruct,
duration=args.duration,
)
audios = model.generate(**gen_kwargs)
audio_pt = np.asarray(audios[0], dtype=np.float32)
sf.write(args.out_pt, audio_pt, 24000, subtype="FLOAT")
print(f"[Python] Audio: {audio_pt.shape[0]} samples {audio_pt.shape[0] / 24000:.2f}s -> {args.out_pt}")
# Free the GPU before launching the C++ binary so it has room to load
# the F32 GGUFs without fighting for VRAM.
del model
torch.cuda.empty_cache()
# C++ path : same text, same instruct, same seed, F32 GGUF weights,
# dumps under cpp/.
cmd = [
BIN,
"--model", model_lm,
"--codec", model_cdc,
"--seed", str(args.seed),
"--instruct", args.instruct,
"--lang", args.lang,
"--format", "wav32",
"--dump", DUMP_CPP,
"--no-fa",
"-o", args.out_cpp,
]
if args.duration:
cmd += ["--duration", str(args.duration)]
print(f"[GGML] Cmd: {' '.join(cmd)}")
r = subprocess.run(cmd, input=text, text=True)
if r.returncode != 0:
sys.exit(r.returncode)
audio_cpp, sr = sf.read(args.out_cpp)
if audio_cpp.ndim > 1:
audio_cpp = audio_cpp[:, 0]
audio_cpp = audio_cpp.astype(np.float32)
print(f"[GGML] Audio: {audio_cpp.shape[0]} samples {sr} Hz {audio_cpp.shape[0] / sr:.2f}s -> {args.out_cpp}")
# Cossim in pipeline order: prompt-ids -> embed -> l0 -> l6 -> l13 -> l20
# -> final -> logits -> tokens -> audio. Cond and uncond on the same line
# so a drift localizes immediately to the originating stage.
def pair(name):
a, _ = load_dump(os.path.join(DUMP_CPP, name))
b, _ = load_dump(os.path.join(DUMP_PT, name))
return a, b
def ids_exact(a, b):
n = min(a.size, b.size)
ai = a.astype(np.int64).ravel()[:n]
bi = b.astype(np.int64).ravel()[:n]
diffs = np.where(ai != bi)[0]
return 100.0 * float(np.mean(ai == bi)), diffs, ai, bi
ca, cb = pair("prompt-cond-ids.bin")
ua, ub = pair("prompt-uncond-ids.bin")
cm, cd, cai, cbi = ids_exact(ca, cb)
um, ud, uai, ubi = ids_exact(ua, ub)
print(f"[Cossim] PromptIDs cond exact: {cm:.2f}% uncond exact: {um:.2f}%")
for s in cd[:20]:
print(f"[Cossim] PromptIDs cond diff at s={s}: ggml={cai[s]} python={cbi[s]}")
for s in ud[:20]:
print(f"[Cossim] PromptIDs uncond diff at s={s}: ggml={uai[s]} python={ubi[s]}")
stages = [
("Embed", "lm-hidden-step0-{}-embed.bin"),
("L0", "lm-hidden-step0-{}-l0.bin"),
("L6", "lm-hidden-step0-{}-l6.bin"),
("L13", "lm-hidden-step0-{}-l13.bin"),
("L20", "lm-hidden-step0-{}-l20.bin"),
("Final", "lm-hidden-step0-{}.bin"),
("Logits", "lm-logits-step0-{}.bin"),
]
def metric(a, b):
n = min(a.size, b.size)
af = a.astype(np.float64).ravel()[:n]
bf = b.astype(np.float64).ravel()[:n]
d = np.abs(af - bf)
nrm_a = float(np.linalg.norm(af))
nrm_b = float(np.linalg.norm(bf))
c = float(np.dot(af, bf) / (nrm_a * nrm_b)) if nrm_a > 1e-10 and nrm_b > 1e-10 else 0.0
return c, float(d.max()), float(d.mean())
for label, fmt in stages:
ca, cb = pair(fmt.format("cond"))
ua, ub = pair(fmt.format("uncond"))
cc, cmax, cmean = metric(ca, cb)
uc, umax, umean = metric(ua, ub)
print(f"[Cossim] {label} cond cos: {cc:.6f} max: {cmax:.4e} mean: {cmean:.4e} uncond cos: {uc:.6f} max: {umax:.4e} mean: {umean:.4e}")
pa, pb = pair("mg-pred-tokens-step0.bin")
n = min(pa.size, pb.size)
ai = pa.astype(np.int64).ravel()[:n]
bi = pb.astype(np.int64).ravel()[:n]
diffs = np.where(ai != bi)[0]
print(f"[Cossim] Step0Tokens exact: {100.0 * float((ai == bi).mean()):.2f}% diffs: {diffs.size}")
sa, sb = pair("mg-scores-step0.bin")
sd = np.abs(sa - sb)
print(f"[Cossim] Step0Scores cos: {cos(sa, sb):.6f} max_abs_diff: {sd.max():.6f} mean_abs_diff: {sd.mean():.6f}")
la, lb = pair("mg-log-probs-step0.bin")
finite = np.isfinite(la) & np.isfinite(lb)
laf = la[finite]; lbf = lb[finite]
ld = np.abs(laf - lbf)
print(f"[Cossim] Step0LogProbs cos: {cos(laf, lbf):.6f} max_abs_diff: {ld.max():.6f} mean_abs_diff: {ld.mean():.6f}")
ta, tb = pair("mg-tokens.bin")
n = min(ta.size, tb.size)
ai = ta.astype(np.int64).ravel()[:n]
bi = tb.astype(np.int64).ravel()[:n]
print(f"[Cossim] Tokens: {cos(ta, tb):.6f} exact: {100.0 * float(np.mean(ai == bi)):.2f}%")
aa, ab = pair("output-audio.bin")
print(f"[Cossim] Audio: {cos(aa, ab):.6f}")
n = min(audio_cpp.size, audio_pt.size)
print(f"[Cossim] WAV stft_cos: {stft_cos(audio_cpp[:n], audio_pt[:n]):.6f} samples: {n}")
if __name__ == "__main__":
main()