@article{d3netaspeakerlistenerarchitecturefor, title = {D3Net: A Unified Speaker-Listener Architecture for 3D Dense Captioning and Visual Grounding}, author = {Dave Zhenyu Chen and Qirui Wu and Matthias Nießner and Angel X. Chang}, year = {2021}, eprint = {2112.01551}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2112.01551v2}, }