@inproceedings{yan2026proact-vl, author = {Yan, Weicai and Dai, Yuhong and Ran, Qimu and Li, Haodong and Lin, Wang and Liao, Haodong and Xie, Xing and Jin, Tao and Lian, Jianxun}, title = {Proact-VL: A Proactive VideoLLM for Real-Time AI Companions}, booktitle = {ICML 2026}, year = {2026}, month = {May}, abstract = {Proactive and real-time interactive experiences are essential for human-like AI companions, yet face three key challenges: (1) achieving low-latency inference under continuous streaming inputs, (2) autonomously deciding when to respond, and (3) controlling both quality and quantity of generated content to meet real-time constraints. In this work, we instantiate AI companions through two gaming scenarios, commentator and guide, selected for their suitability for automatic evaluation. We introduce the Live Gaming Benchmark, a large-scale dataset with three representative scenarios: solo commentary, co-commentary, and user guidance, and present Proact-VL, a general framework that shapes multimodal language models into proactive, real-time interactive agents capable of human-like environment perception and interaction. Extensive experiments show Proact-VL achieves superior response latency and quality while maintaining strong video understanding capabilities, demonstrating its practicality for real-time interactive applications.}, url = {http://approjects.co.za/?big=en-us/research/publication/proact-vl-a-proactive-videollm-for-real-time-ai-companions/}, }