A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
68 lines
2.3 KiB
Python
68 lines
2.3 KiB
Python
"""fp32 ONNX export of a NeMo transducer for sherpa-onnx, reproducing k2-fsa's own export verbatim
|
|
(k2-fsa/sherpa-onnx @040afe36, scripts/nemo/{parakeet-tdt-0.6b-v3,parakeet-unified-en-0.6b}/export_onnx.py):
|
|
same encoder/decoder/joint .export() calls, same tokens.txt, same metadata, encoder weights as external data.
|
|
The ONLY omission is their final quantize_dynamic() step: the point is the non-int8 graph on the CUDA EP.
|
|
usage: export_onnx.py NEMO_PATH OUT_DIR URL COMMENT
|
|
"""
|
|
import os
|
|
import sys
|
|
|
|
import onnx
|
|
import torch
|
|
import nemo.collections.asr as nemo_asr
|
|
|
|
|
|
def add_meta_data(filename, meta_data):
|
|
model = onnx.load(filename)
|
|
while len(model.metadata_props):
|
|
model.metadata_props.pop()
|
|
for key, value in meta_data.items():
|
|
meta = model.metadata_props.add()
|
|
meta.key = key
|
|
meta.value = str(value)
|
|
if os.path.basename(filename) == "encoder.onnx":
|
|
onnx.save(model, filename, save_as_external_data=True, all_tensors_to_one_file=True, location="encoder.weights")
|
|
else:
|
|
onnx.save(model, filename)
|
|
|
|
|
|
@torch.no_grad()
|
|
def main():
|
|
nemo_path, out, url, comment = sys.argv[1:5]
|
|
os.makedirs(out, exist_ok=True)
|
|
os.chdir(out)
|
|
m = nemo_asr.models.ASRModel.restore_from(restore_path=nemo_path, map_location="cpu")
|
|
m.eval()
|
|
if m.cfg.get("validation_ds") is None:
|
|
m.cfg.validation_ds = dict()
|
|
with open("./tokens.txt", "w", encoding="utf-8") as f:
|
|
for i, s in enumerate(m.joint.vocabulary):
|
|
f.write(f"{s} {i}\n")
|
|
f.write(f"<blk> {i+1}\n")
|
|
m.encoder.export("encoder.onnx")
|
|
m.decoder.export("decoder.onnx")
|
|
m.joint.export("joiner.onnx")
|
|
normalize_type = m.cfg.preprocessor.normalize
|
|
if normalize_type == "NA":
|
|
normalize_type = ""
|
|
meta = {
|
|
"vocab_size": m.decoder.vocab_size,
|
|
"normalize_type": normalize_type,
|
|
"pred_rnn_layers": m.decoder.pred_rnn_layers,
|
|
"pred_hidden": m.decoder.pred_hidden,
|
|
"subsampling_factor": m.encoder.subsampling_factor,
|
|
"model_type": "EncDecRNNTBPEModel",
|
|
"version": "2",
|
|
"model_author": "NeMo",
|
|
"url": url,
|
|
"comment": comment,
|
|
"feat_dim": 128,
|
|
}
|
|
add_meta_data("encoder.onnx", meta)
|
|
print("meta", meta)
|
|
os.system("ls -la")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|