Monocular Depth Estimation on KITTI
0.04Abs RelBetterDepth
Evaluation Results
| Method | Links | |||||||||
|---|---|---|---|---|---|---|---|---|---|---|
| BetterDepthGT aligned=true2024.11 | 0.04 | — | — | — | 95 | — | — | — | — | |
| OursArchitecture=Autoregressive2026.03 | 0.044 | 0.132 | 1.712 | 0.069 | 98 | 99.7 | 99.9 | — | — | |
| DepthAnythingArchitecture=ViT-L2026.03 | 0.046 | — | 1.896 | 0.069 | 98.2 | 99.8 | 100 | — | — | |
| DAR-BaseArchitecture=Autoregressive2026.03 | 0.046 | 0.114 | 1.823 | 0.069 | 98.5 | 99.9 | 100 | — | — | |
| EcoDepthArchitecture=ViT-L2026.03 | 0.048 | 0.139 | 2.039 | 0.074 | 97.9 | 99.8 | 100 | — | — | |
| ScaleDepth-NKParams=216M, Pre-trained on extra depth datasets=false2024.07 | 0.049 | — | — | — | — | — | — | — | — | |
| MoGe22026.02 | 0.049 | — | — | — | 97.9 | — | — | — | — | |
| MoGe v22026.03 | 0.049 | — | — | — | 97.9 | — | — | — | — | |
| MoGe2025.07 | 0.049 | — | — | — | 97.9 | — | — | — | — | |
| MoGeMethod version=v22025.07 | 0.049 | — | — | — | 97.9 | — | — | — | — | |
| WorDepthArchitecture=Swin-Large2026.03 | 0.049 | — | 2.039 | 0.074 | 97.9 | 99.8 | 99.9 | — | — | |
| MIMArchitecture=SwinV2-L2023.06 | 0.05 | — | 1.966 | — | 97.7 | — | — | — | — | |
| UniDepthGT aligned=false2024.11 | 0.05 | — | — | — | 98 | — | — | — | — | |
| Metric3Dv2GT aligned=false2024.11 | 0.05 | — | — | — | 98 | — | — | — | — | |
| VA-DepthNetArchitecture=Swin-Large2026.03 | 0.05 | 0.148 | 2.093 | 0.076 | 97.7 | 99.7 | 99.9 | — | — | |
| IEBinsArchitecture=Swin-Large2026.03 | 0.05 | 0.142 | 2.011 | — | 97.8 | 99.8 | 99.9 | — | — | |
| iDiscArchitecture=Swin-Large2026.03 | 0.05 | 0.145 | 2.067 | 0.077 | 97.7 | 99.7 | 99.9 | — | — | |
| DiffusionDepthArchitecture=Diffusion2026.03 | 0.05 | 0.141 | 2.016 | 0.074 | 97.7 | 99.8 | 99.9 | — | — | |
| PixelFormerArchitecture=Swin-Large, Supervised Pretraining=true2023.06 | 0.051 | — | 2.081 | — | 97.6 | — | — | — | — | |
| PixelFormerArchitecture=Swin-Large2026.03 | 0.051 | 0.149 | 2.081 | 0.077 | 97.6 | 99.7 | 99.9 | — | — | |
| DCDepthArchitecture=Swin-Large2026.03 | 0.051 | 0.145 | 2.044 | 0.076 | 97.7 | 99.7 | 99.9 | — | — | |
| BinsFormerArchitecture=Swin-Large, Supervised Pretraining=true2023.06 | 0.052 | — | 2.098 | — | 97.4 | — | — | — | — | |
| NeWCRFsArchitecture=Swin-Large2026.03 | 0.052 | 0.155 | 2.129 | 0.079 | 97.4 | 99.7 | 99.9 | — | — | |
| BinsFormerArchitecture=Swin-Large2026.03 | 0.052 | 0.151 | 2.098 | 0.079 | 97.4 | 99.7 | 99.9 | — | — | |
| MoGe12026.02 | 0.054 | — | — | — | 97.7 | — | — | — | — | |
| MoGe v12026.03 | 0.054 | — | — | — | 97.7 | — | — | — | — | |
| MoGeMethod version=v12025.07 | 0.054 | — | — | — | 97.7 | — | — | — | — | |
| ZoeDepthArchitecture=ViT-L2026.03 | 0.054 | 0.189 | 2.44 | 0.083 | 97.7 | 99.6 | 99.9 | — | — | |
| DDVMArchitecture=Efficient U-Net, Unsupervised Pretraining=true, Auxiliary Supervised Depth Data=true, samples=22023.06 | 0.055 | — | 2.66 | — | 96.5 | — | — | — | — | |
| DDVMArchitecture=Efficient U-Net, Unsupervised Pretraining=true, Auxiliary Supervised Depth Data=true, samples=42023.06 | 0.055 | — | 2.613 | — | 96.5 | — | — | — | — | |
| DDVMArchitecture=Efficient U-Net, Unsupervised Pretraining=true, Auxiliary Supervised Depth Data=true, samples=12023.06 | 0.056 | — | 2.7 | — | 96.4 | — | — | — | — | |
| AdaBinsArchitecture=E-B5+Mini-ViT, Supervised Pretraining=true2023.06 | 0.058 | — | 2.36 | — | 96.4 | — | — | — | — | |
| TrajVG2026.02 | 0.058 | — | — | — | 96.7 | — | — | — | — | |
| BTSArchitecture=DenseNet-161, Supervised Pretraining=true2023.06 | 0.059 | — | 2.756 | — | 95.6 | — | — | — | — | |
| pi^32026.03 | 0.059 | — | — | — | 97.2 | — | — | — | — | |
| ZoeDepthGT aligned=false2024.11 | 0.06 | — | — | — | 97 | — | — | — | — | |
| SharpDepthGT aligned=false2024.11 | 0.06 | — | — | — | 97 | — | — | — | — | |
| π32026.02 | 0.06 | — | — | — | 97.1 | — | — | — | — | |
| pi^32025.07 | 0.06 | — | — | — | 97.1 | — | — | — | — | |
| DPTArchitecture=VIT-L2026.03 | 0.06 | — | 2.573 | 0.092 | 95.9 | 99.5 | 99.6 | — | — | |
| π³Scale=Scale-Invariant2026.05 | 0.06 | — | 3.182 | — | 97.1 | — | — | — | — | |
| ZoeD-M12-NK+Params=345M, Pre-trained on extra depth datasets=true2024.07 | 0.061 | — | — | — | — | — | — | — | — | |
| UNITScale=Scale-Invariant2026.05 | 0.061 | — | 3.976 | — | 96.6 | — | — | — | — | |
| DPTArchitecture=Res-50+ViT-B, Supervised Pretraining=true, Auxiliary Supervised Depth Data=true2023.06 | 0.062 | — | 2.573 | — | 95.9 | — | — | — | — | |
| Cross-Context DistillationZero-shot=true, Teacher Model=MiDaS v3.12025.02 | 0.063 | — | — | — | 97.2 | — | — | — | — | |
| ZipMap2026.03 | 0.063 | — | — | — | 96 | — | — | — | — | |
| SnDBackbone=ViT-Base2026.02 | 0.0631 | — | 2.1741 | — | — | — | — | — | — | |
| TransDepthArchitecture=Res-50+ViT-B, Supervised Pretraining=true2023.06 | 0.064 | — | 2.755 | — | 95.6 | — | — | — | — | |
| AdaBinsArchitecture=E-B5+mini-ViT2026.03 | 0.067 | 0.19 | 2.96 | 0.088 | 94.9 | 99.2 | 99.8 | — | — | |
| Fit3DBackbone=ViT-Base2026.02 | 0.0679 | — | 2.2485 | — | — | — | — | — | — | |
| DINOv2Backbone=ViT-Base2026.02 | 0.0686 | — | 2.3558 | — | — | — | — | — | — | |
| Cross-Context DistillationZero-shot=true, Teacher Model=DepthAnythingv2-Large2025.02 | 0.07 | — | — | — | 94.9 | — | — | — | — | |
| P3DepthArchitecture=ResNet-1012026.03 | 0.071 | 0.27 | 2.842 | 0.103 | 95.3 | 99.3 | 99.8 | — | — | |
| DORNResolution=513 × 385, Supervision=D2021.05 | 0.072 | 0.307 | 2.727 | 0.12 | 93.2 | 98.4 | 99.4 | — | — | |
| VNLResolution=385 × 385, Supervision=D2021.05 | 0.072 | — | 3.258 | 0.117 | 93.8 | 99 | 99.8 | — | — | |
| ZoeD-X-NKParams=345M, Pre-trained on extra depth datasets=false2024.07 | 0.074 | — | — | — | — | — | — | — | — | |
| DepthAnything v2Zero-shot=true2025.02 | 0.074 | — | — | — | 94.6 | — | — | — | — | |
| DepthAnything3Scale=Scale-Invariant2026.05 | 0.074 | — | 3.25 | — | 95.6 | — | — | — | — | |
| NeWCRFs-X-NKParams=270M, Pre-trained on extra depth datasets=false2024.07 | 0.076 | — | — | — | — | — | — | — | — | |
| MASt3R2026.02 | 0.077 | — | — | — | 94.8 | — | — | — | — | |
| MASt3R2026.03 | 0.077 | — | — | — | 94.8 | — | — | — | — | |
| VGGTScale=Scale-Invariant2026.05 | 0.077 | — | 4.217 | — | 93.1 | — | — | — | — | |
| StreamVGGTScale=Scale-Invariant2026.05 | 0.077 | — | 3.959 | — | 94 | — | — | — | — | |
| MEFBackbone=ViT-Base2026.02 | 0.0772 | — | 2.416 | — | — | — | — | — | — | |
| GenPerceptZero-shot=true2025.02 | 0.08 | — | — | — | 93.4 | — | — | — | — | |
| LotusArchitecture=Diffusion2026.03 | 0.081 | — | — | — | 93.1 | 98.7 | — | — | — | |
| VGGT2026.02 | 0.082 | — | — | — | 94.7 | — | — | — | — | |
| VGGT2026.03 | 0.089 | — | — | — | 93.9 | — | — | — | — | |
| MarigoldGT aligned=true2024.11 | 0.09 | — | — | — | 92 | — | — | — | — | |
| DA V22025.07 | 0.09 | — | — | — | 91.9 | — | — | — | — | |
| DA V2Evaluation variant=metric outdoor2025.07 | 0.09 | — | — | — | 91.9 | — | — | — | — | |
| DepthFM DiffusionArchitecture=Diffusion2026.03 | 0.091 | — | — | — | 92 | — | — | — | — | |
| MoGe2026.04 | 0.094 | — | — | — | — | — | — | 90.4 | 2.43 | |
| DynamicDepthsemantic information usage=Training and Testing, Resolution=192 x 640, multiple test frames=true2023.12 | 0.096 | 0.72 | 4.458 | 0.175 | 89.7 | 96.4 | 98.4 | — | — | |
| Ours-MonoViTsemantic information usage=Training, Resolution=192 x 6402023.12 | 0.096 | 0.696 | 4.327 | 0.174 | 90.4 | 96.8 | 98.5 | — | — | |
| MapAnythingScale=Scale-Invariant2026.05 | 0.096 | — | 4.062 | — | 93.2 | — | — | — | — | |
| CUT3R2026.02 | 0.097 | — | — | — | 91.4 | — | — | — | — | |
| CUT3R2026.03 | 0.097 | — | — | — | 91.4 | — | — | — | — | |
| DA V2Evaluation variant=metric indoor2025.07 | 0.097 | — | — | — | 91.2 | — | — | — | — | |
| MonST3R2026.02 | 0.098 | — | — | — | 89.5 | — | — | — | — | |
| MonST3R2026.03 | 0.098 | — | — | — | 89.5 | — | — | — | — | |
| MoGe v22026.04 | 0.098 | — | — | — | — | — | — | 90.8 | 2.79 | |
| D3VOResolution=512 × 256, Supervision=S+M2021.05 | 0.099 | 0.763 | 4.485 | 0.185 | 88.5 | 95.8 | 97.9 | — | — | |
| MonoViTResolution=192 x 6402023.12 | 0.099 | 0.708 | 4.372 | 0.175 | 90 | 96.7 | 98.4 | — | — | |
| MarigoldZero-shot=true2025.02 | 0.099 | — | — | — | 91.6 | — | — | — | — | |
| PackNet-SemResolution=1280 × 384, Supervision=M+L2021.05 | 0.1 | 0.761 | 4.27 | 0.175 | 90.2 | 96.5 | 98.2 | — | — | |
| DPTZero-shot=true2025.02 | 0.1 | — | — | — | 90.1 | — | — | — | — | |
| Ours-CADepthsemantic information usage=Training, Resolution=192 x 6402023.12 | 0.103 | 0.73 | 4.427 | 0.179 | 89.5 | 96.6 | 98.4 | — | — | |
| CUT3RScale=Scale-Invariant2026.05 | 0.103 | — | 5.308 | — | 88.5 | — | — | — | — | |
| FSRE-Depthsemantic information usage=Training, Resolution=192 x 6402023.12 | 0.105 | 0.708 | 4.546 | 0.182 | 88.6 | 96.4 | 98.3 | — | — | |
| CADepthResolution=192 x 6402023.12 | 0.105 | 0.769 | 4.535 | 0.181 | 89.2 | 96.4 | 98.3 | — | — | |
| Repurposing DiffusionArchitecture=Diffusion2026.03 | 0.105 | — | — | — | 90.4 | — | — | — | — | |
| Self-Mono-SFtraining_mode=stereo sequences2020.04 | 0.106 | 0.888 | 4.853 | 0.175 | 87.9 | 96.5 | 98.7 | — | — | |
| Monodepth2-MSResolution=1024 × 320, Supervision=S+M2021.05 | 0.106 | 0.806 | 4.63 | 0.193 | 87.6 | 95.8 | 98 | — | — | |
| Petrovai et al.Resolution=192 x 6402023.12 | 0.106 | 0.751 | 4.485 | 0.18 | 88.5 | 96.4 | 98.4 | — | — | |
| CARVE2026.04 | 0.106 | — | — | — | — | — | — | 88.5 | 1.86 | |
| MASt3R+OursFine-tuning=true, Evaluation Protocol=Metric depth2025.11 | 0.1069 | — | — | — | 89.1 | — | — | — | — | |
| Monodepth2-SResolution=1024 x 320, Supervision=S2021.05 | 0.107 | 0.849 | 4.764 | 0.201 | 87.4 | 95.3 | 97.7 | — | — | |
| SGDepth-fullResolution=1280 × 384, Supervision=M+L2021.05 | 0.107 | 0.768 | 4.468 | 0.186 | 89.1 | 96.3 | 98.2 | — | — | |
| PackNet-SfMResolution=1280 × 384, Supervision=M2021.05 | 0.107 | 0.802 | 4.538 | 0.186 | 88.9 | 96.2 | 98.1 | — | — |