{"task":"Chat-based Image Retrieval","dataset":"VisDial","metric_names":["Recall@10 on 1 rounds","Recall@10 on 2 rounds","Recall@10 on 3 rounds","Hits@10 on 10 Round"],"rows":[{"id":49152,"task":"Chat-based Image Retrieval","parent_task":"Image Retrieval","dataset":"VisDial","model_name":"ChatGPT & BLIP2","metrics":{"Recall@10 on 1 rounds":"70","Recall@10 on 2 rounds":"73.5","Recall@10 on 3 rounds":"75.75"},"paper_url":null,"paper_title":null,"paper_date":null,"code_links":[],"metrics_order":"[\"Recall@10 on 1 rounds\", \"Recall@10 on 2 rounds\", \"Recall@10 on 3 rounds\", \"Hits@10 on 10 Round\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":49153,"task":"Chat-based Image Retrieval","parent_task":"Image Retrieval","dataset":"VisDial","model_name":"Human & BLIP2","metrics":{"Recall@10 on 1 rounds":"67","Recall@10 on 2 rounds":"70","Recall@10 on 3 rounds":"71.8"},"paper_url":null,"paper_title":null,"paper_date":null,"code_links":[],"metrics_order":"[\"Recall@10 on 1 rounds\", \"Recall@10 on 2 rounds\", \"Recall@10 on 3 rounds\", \"Hits@10 on 10 Round\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":49154,"task":"Chat-based Image Retrieval","parent_task":"Image Retrieval","dataset":"VisDial","model_name":"ImageScope (CLIP-ViT-L/14)","metrics":{"Hits@10 on 10 Round":"79.89"},"paper_url":"https://arxiv.org/abs/2503.10166v1","paper_title":"ImageScope: Unifying Language-Guided Image Retrieval via Large Multimodal Model Collective Reasoning","paper_date":"2025-03-13","code_links":[{"title":"pengfei-luo/ImageScope","url":"https://github.com/pengfei-luo/ImageScope"}],"metrics_order":"[\"Recall@10 on 1 rounds\", \"Recall@10 on 2 rounds\", \"Recall@10 on 3 rounds\", \"Hits@10 on 10 Round\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]}]}