@misc{indiciae38efc18687a6, title = {Aligning where to see and what to tell: image caption with region-based attention and scene factorization}, author = {Junqi Jin and Kun Fu and Runpeng Cui and Fei Sha and Changshui Zhang}, year = {2015}, url = {https://arxiv.org/abs/1506.06272}, note = {Source identifier: 1506.06272} }