@misc{indiciae1d3dcfa42ecc, title = {Joint Training Is Not Enough: Conditioned Cross-Granularity Training for Multimodal Document Understanding}, author = {Chengguang Gan and Yunhao Liang and Hanjun Wei and Qinghao Zhang and Shiwen Ni}, year = {2026}, url = {https://arxiv.org/abs/2609.00756}, note = {Source identifier: 2609.00756} }