@misc{indiciae105eac65a7f6, title = {TempoGround: State-Aware Streaming Visual Grounding with Vision-Language Models}, author = {Leqian Ding and Junning Qiu and Manwen Yang and Yu Guo and Fei Wang}, year = {2026}, url = {https://arxiv.org/abs/2609.02359}, note = {Source identifier: 2609.02359} }