{"task":"Object Detection","dataset":"COCO 2017","metric_names":["AP","mAP","Mean mAP","AP50","AP75","APM","APM50","APM75"],"rows":[{"id":62591,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"MaxViT-B","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62592,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"MaxViT-S","metrics":{"AP":"53.1","AP50":"72.5","AP75":"58.1","APM":"45.4","APM50":"69.8","APM75":"49.5"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62593,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"MaxViT-T","metrics":{"AP":"52.1","AP50":"71.9","AP75":"56.8","APM":"44.6","APM50":"69.1","APM75":"48.4"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62594,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DAT-S++","metrics":{"AP":"50.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62595,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DAT-T++","metrics":{"AP":"49.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62596,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DyHead (SAP)","metrics":{"AP":"42.1","AP50":"59.4","AP75":"45.9"},"paper_url":"https://arxiv.org/abs/2409.16630v1","paper_title":"Stochastic Subsampling With Average Pooling","paper_date":"2024-09-25","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62597,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"Faster R-CNN (ideal number of groups)","metrics":{"AP":"40.7","AP50":"61.2","AP75":"44.6"},"paper_url":"https://arxiv.org/abs/2302.03193v1","paper_title":"On the Ideal Number of Groups for Isometric Gradient Propagation","paper_date":"2023-02-07","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62598,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-XL++","metrics":{"mAP":"56.4"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62599,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-L++","metrics":{"mAP":"55.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62600,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-B++","metrics":{"mAP":"54.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62601,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-S++","metrics":{"mAP":"54.3"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62602,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"MixMIM-L","metrics":{"mAP":"54.1"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62603,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-S","metrics":{"mAP":"53"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62604,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"MixMIM-B","metrics":{"mAP":"52.2"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62605,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"UniRepLKNet-T","metrics":{"mAP":"51.7"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62606,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"BiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.6"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62607,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62608,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"BiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.8"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62609,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62610,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, Retina)","metrics":{"mAP":"47.1"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62611,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, Retina)","metrics":{"mAP":"45.6"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62612,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"YOLO-Drone","metrics":{"mAP":"35.45"},"paper_url":"https://arxiv.org/abs/2304.06925v2","paper_title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","paper_date":"2023-04-14","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62613,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"retinanet","metrics":{"Mean mAP":"3153"},"paper_url":"https://arxiv.org/abs/1912.09476v2","paper_title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","paper_date":"2019-12-19","code_links":[{"title":"ParallelDots/generic-sku-detection-benchmark","url":"https://github.com/ParallelDots/generic-sku-detection-benchmark"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":62614,"task":"Object Detection","parent_task":null,"dataset":"COCO 2017","model_name":"Lpixel","metrics":{"Mean mAP":"4.2"},"paper_url":"https://arxiv.org/abs/2108.03798v2","paper_title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","paper_date":"2021-08-09","code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76570,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"MaxViT-B","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76571,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"MaxViT-S","metrics":{"AP":"53.1","AP50":"72.5","AP75":"58.1","APM":"45.4","APM50":"69.8","APM75":"49.5"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76572,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"MaxViT-T","metrics":{"AP":"52.1","AP50":"71.9","AP75":"56.8","APM":"44.6","APM50":"69.1","APM75":"48.4"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76573,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DAT-S++","metrics":{"AP":"50.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76574,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DAT-T++","metrics":{"AP":"49.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76575,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DyHead (SAP)","metrics":{"AP":"42.1","AP50":"59.4","AP75":"45.9"},"paper_url":"https://arxiv.org/abs/2409.16630v1","paper_title":"Stochastic Subsampling With Average Pooling","paper_date":"2024-09-25","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76576,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"Faster R-CNN (ideal number of groups)","metrics":{"AP":"40.7","AP50":"61.2","AP75":"44.6"},"paper_url":"https://arxiv.org/abs/2302.03193v1","paper_title":"On the Ideal Number of Groups for Isometric Gradient Propagation","paper_date":"2023-02-07","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76577,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-XL++","metrics":{"mAP":"56.4"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76578,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-L++","metrics":{"mAP":"55.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76579,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-B++","metrics":{"mAP":"54.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76580,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-S++","metrics":{"mAP":"54.3"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76581,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"MixMIM-L","metrics":{"mAP":"54.1"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76582,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-S","metrics":{"mAP":"53"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76583,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"MixMIM-B","metrics":{"mAP":"52.2"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76584,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"UniRepLKNet-T","metrics":{"mAP":"51.7"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76585,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"BiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.6"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76586,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76587,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"BiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.8"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76588,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76589,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, Retina)","metrics":{"mAP":"47.1"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76590,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, Retina)","metrics":{"mAP":"45.6"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76591,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"YOLO-Drone","metrics":{"mAP":"35.45"},"paper_url":"https://arxiv.org/abs/2304.06925v2","paper_title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","paper_date":"2023-04-14","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76592,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"retinanet","metrics":{"Mean mAP":"3153"},"paper_url":"https://arxiv.org/abs/1912.09476v2","paper_title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","paper_date":"2019-12-19","code_links":[{"title":"ParallelDots/generic-sku-detection-benchmark","url":"https://github.com/ParallelDots/generic-sku-detection-benchmark"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":76593,"task":"Object Detection","parent_task":"3D","dataset":"COCO 2017","model_name":"Lpixel","metrics":{"Mean mAP":"4.2"},"paper_url":"https://arxiv.org/abs/2108.03798v2","paper_title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","paper_date":"2021-08-09","code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117097,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"MaxViT-B","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117098,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"MaxViT-S","metrics":{"AP":"53.1","AP50":"72.5","AP75":"58.1","APM":"45.4","APM50":"69.8","APM75":"49.5"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117099,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"MaxViT-T","metrics":{"AP":"52.1","AP50":"71.9","AP75":"56.8","APM":"44.6","APM50":"69.1","APM75":"48.4"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117100,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DAT-S++","metrics":{"AP":"50.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117101,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DAT-T++","metrics":{"AP":"49.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117102,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DyHead (SAP)","metrics":{"AP":"42.1","AP50":"59.4","AP75":"45.9"},"paper_url":"https://arxiv.org/abs/2409.16630v1","paper_title":"Stochastic Subsampling With Average Pooling","paper_date":"2024-09-25","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117103,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"Faster R-CNN (ideal number of groups)","metrics":{"AP":"40.7","AP50":"61.2","AP75":"44.6"},"paper_url":"https://arxiv.org/abs/2302.03193v1","paper_title":"On the Ideal Number of Groups for Isometric Gradient Propagation","paper_date":"2023-02-07","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117104,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-XL++","metrics":{"mAP":"56.4"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117105,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-L++","metrics":{"mAP":"55.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117106,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-B++","metrics":{"mAP":"54.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117107,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-S++","metrics":{"mAP":"54.3"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117108,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"MixMIM-L","metrics":{"mAP":"54.1"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117109,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-S","metrics":{"mAP":"53"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117110,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"MixMIM-B","metrics":{"mAP":"52.2"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117111,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"UniRepLKNet-T","metrics":{"mAP":"51.7"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117112,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"BiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.6"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117113,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117114,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"BiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.8"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117115,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117116,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, Retina)","metrics":{"mAP":"47.1"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117117,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, Retina)","metrics":{"mAP":"45.6"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117118,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"YOLO-Drone","metrics":{"mAP":"35.45"},"paper_url":"https://arxiv.org/abs/2304.06925v2","paper_title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","paper_date":"2023-04-14","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117119,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"retinanet","metrics":{"Mean mAP":"3153"},"paper_url":"https://arxiv.org/abs/1912.09476v2","paper_title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","paper_date":"2019-12-19","code_links":[{"title":"ParallelDots/generic-sku-detection-benchmark","url":"https://github.com/ParallelDots/generic-sku-detection-benchmark"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":117120,"task":"Object Detection","parent_task":"2D Classification","dataset":"COCO 2017","model_name":"Lpixel","metrics":{"Mean mAP":"4.2"},"paper_url":"https://arxiv.org/abs/2108.03798v2","paper_title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","paper_date":"2021-08-09","code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124173,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"MaxViT-B","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124174,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"MaxViT-S","metrics":{"AP":"53.1","AP50":"72.5","AP75":"58.1","APM":"45.4","APM50":"69.8","APM75":"49.5"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124175,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"MaxViT-T","metrics":{"AP":"52.1","AP50":"71.9","AP75":"56.8","APM":"44.6","APM50":"69.1","APM75":"48.4"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124176,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DAT-S++","metrics":{"AP":"50.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124177,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DAT-T++","metrics":{"AP":"49.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124178,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DyHead (SAP)","metrics":{"AP":"42.1","AP50":"59.4","AP75":"45.9"},"paper_url":"https://arxiv.org/abs/2409.16630v1","paper_title":"Stochastic Subsampling With Average Pooling","paper_date":"2024-09-25","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124179,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"Faster R-CNN (ideal number of groups)","metrics":{"AP":"40.7","AP50":"61.2","AP75":"44.6"},"paper_url":"https://arxiv.org/abs/2302.03193v1","paper_title":"On the Ideal Number of Groups for Isometric Gradient Propagation","paper_date":"2023-02-07","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124180,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-XL++","metrics":{"mAP":"56.4"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124181,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-L++","metrics":{"mAP":"55.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124182,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-B++","metrics":{"mAP":"54.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124183,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-S++","metrics":{"mAP":"54.3"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124184,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"MixMIM-L","metrics":{"mAP":"54.1"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124185,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-S","metrics":{"mAP":"53"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124186,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"MixMIM-B","metrics":{"mAP":"52.2"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124187,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"UniRepLKNet-T","metrics":{"mAP":"51.7"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124188,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"BiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.6"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124189,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124190,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"BiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.8"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124191,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124192,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, Retina)","metrics":{"mAP":"47.1"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124193,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, Retina)","metrics":{"mAP":"45.6"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124194,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"YOLO-Drone","metrics":{"mAP":"35.45"},"paper_url":"https://arxiv.org/abs/2304.06925v2","paper_title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","paper_date":"2023-04-14","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124195,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"retinanet","metrics":{"Mean mAP":"3153"},"paper_url":"https://arxiv.org/abs/1912.09476v2","paper_title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","paper_date":"2019-12-19","code_links":[{"title":"ParallelDots/generic-sku-detection-benchmark","url":"https://github.com/ParallelDots/generic-sku-detection-benchmark"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":124196,"task":"Object Detection","parent_task":"2D Object Detection","dataset":"COCO 2017","model_name":"Lpixel","metrics":{"Mean mAP":"4.2"},"paper_url":"https://arxiv.org/abs/2108.03798v2","paper_title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","paper_date":"2021-08-09","code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149892,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"MaxViT-B","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149893,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"MaxViT-S","metrics":{"AP":"53.1","AP50":"72.5","AP75":"58.1","APM":"45.4","APM50":"69.8","APM75":"49.5"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149894,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"MaxViT-T","metrics":{"AP":"52.1","AP50":"71.9","AP75":"56.8","APM":"44.6","APM50":"69.1","APM75":"48.4"},"paper_url":"https://arxiv.org/abs/2204.01697v4","paper_title":"MaxViT: Multi-Axis Vision Transformer","paper_date":"2022-04-04","code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149895,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DAT-S++","metrics":{"AP":"50.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149896,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DAT-T++","metrics":{"AP":"49.2"},"paper_url":"https://arxiv.org/abs/2309.01430v1","paper_title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","paper_date":"2023-09-04","code_links":[{"title":"leaplabthu/dat","url":"https://github.com/leaplabthu/dat"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149897,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DyHead (SAP)","metrics":{"AP":"42.1","AP50":"59.4","AP75":"45.9"},"paper_url":"https://arxiv.org/abs/2409.16630v1","paper_title":"Stochastic Subsampling With Average Pooling","paper_date":"2024-09-25","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149898,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"Faster R-CNN (ideal number of groups)","metrics":{"AP":"40.7","AP50":"61.2","AP75":"44.6"},"paper_url":"https://arxiv.org/abs/2302.03193v1","paper_title":"On the Ideal Number of Groups for Isometric Gradient Propagation","paper_date":"2023-02-07","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149899,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-XL++","metrics":{"mAP":"56.4"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149900,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-L++","metrics":{"mAP":"55.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149901,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-B++","metrics":{"mAP":"54.8"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149902,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-S++","metrics":{"mAP":"54.3"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149903,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"MixMIM-L","metrics":{"mAP":"54.1"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149904,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-S","metrics":{"mAP":"53"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149905,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"MixMIM-B","metrics":{"mAP":"52.2"},"paper_url":"https://arxiv.org/abs/2205.13137v4","paper_title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","paper_date":"2022-05-26","code_links":[{"title":"sense-x/mixmim","url":"https://github.com/sense-x/mixmim"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149906,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"UniRepLKNet-T","metrics":{"mAP":"51.7"},"paper_url":"https://arxiv.org/abs/2311.15599v2","paper_title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","paper_date":"2023-11-27","code_links":[{"title":"ailab-cvc/unireplknet","url":"https://github.com/ailab-cvc/unireplknet"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149907,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"BiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.6"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149908,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"48.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149909,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"BiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.8"},"paper_url":"https://arxiv.org/abs/2303.08810v1","paper_title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","paper_date":"2023-03-15","code_links":[{"title":"rayleizhu/biformer","url":"https://github.com/rayleizhu/biformer"},{"title":"chenller/mmseg-extension","url":"https://github.com/chenller/mmseg-extension"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149910,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, MaskRCNN 12ep)","metrics":{"mAP":"47.5"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149911,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DeBiFormer-B (IN1k pretrain, Retina)","metrics":{"mAP":"47.1"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149912,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"DeBiFormer-S (IN1k pretrain, Retina)","metrics":{"mAP":"45.6"},"paper_url":"https://arxiv.org/abs/2410.08582v1","paper_title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","paper_date":"2024-10-11","code_links":[{"title":"maclong01/DeBiFormer","url":"https://github.com/maclong01/DeBiFormer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149913,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"YOLO-Drone","metrics":{"mAP":"35.45"},"paper_url":"https://arxiv.org/abs/2304.06925v2","paper_title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","paper_date":"2023-04-14","code_links":[],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149914,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"retinanet","metrics":{"Mean mAP":"3153"},"paper_url":"https://arxiv.org/abs/1912.09476v2","paper_title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","paper_date":"2019-12-19","code_links":[{"title":"ParallelDots/generic-sku-detection-benchmark","url":"https://github.com/ParallelDots/generic-sku-detection-benchmark"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]},{"id":149915,"task":"Object Detection","parent_task":"16k","dataset":"COCO 2017","model_name":"Lpixel","metrics":{"Mean mAP":"4.2"},"paper_url":"https://arxiv.org/abs/2108.03798v2","paper_title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","paper_date":"2021-08-09","code_links":[{"title":"huage001/painttransformer","url":"https://github.com/huage001/painttransformer"},{"title":"wzmsltw/painttransformer","url":"https://github.com/wzmsltw/painttransformer"}],"metrics_order":"[\"AP\", \"mAP\", \"Mean mAP\", \"AP50\", \"AP75\", \"APM\", \"APM50\", \"APM75\"]","area":"Computer Vision","uses_additional_data":0,"source":"archive","tags":[]}]}