@misc{xu2026vibevoice-asr-bitnet, author = {Xu, Songcheng and Song, Ting and Huang, Shaohan and Peng, Zhiliang and Xia, Yan and Tu, Yujie and Huang, Xin and Wu, Xun and Wang, Wenhui and Chang, Yaoyao and Yu, Jianwei and Dong, Li and Wei, Furu}, title = {VibeVoice-ASR-BitNet Technical Report}, howpublished = {arXiv}, year = {2026}, month = {July}, abstract = {We present VibeVoice-ASR-BitNet, a compressed variant of VibeVoice-ASR optimized for real-time inference on edge CPUs. We apply heterogeneous quantization tailored to the computational characteristics of each stage: the VAE acoustic tokenizer uses full-pipeline INT8 quantization (I8_S) with kernel fusion and SIMD optimization, while the autoregressive language model adopts BitNet-style ternary weights (I2_S). To preserve accuracy under aggressive compression, we employ a progressive quantization-aware training strategy. For inference, we implement custom SIMD kernels and fused operators within the ggml framework targeting both ARM and x86 platforms, achieving real-time recognition (RTF<1) on low-thread-count CPUs. VibeVoice-ASR-BitNet is 1.6--2.3x faster than Whisper.cpp at comparable model sizes (~1.6 GB), with only modest accuracy degradation compared to the FP16 baseline.}, url = {http://approjects.co.za/?big=en-us/research/publication/vibevoice-asr-bitnet-technical-report/}, }