@article{instructsequnifyingvisiontaskswith, title = {InstructSeq: Unifying Vision Tasks with Instruction-conditioned Multi-modal Sequence Generation}, author = {Rongyao Fang and Shilin Yan and Zhaoyang Huang and Jingqiu Zhou and Hao Tian and Jifeng Dai and Hongsheng Li}, year = {2023}, eprint = {2311.18835}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2311.18835v1}, }