@inproceedings{77a4e49ddd5b4bd8b349680dffb813a4,
title = "Guiding Monocular Depth Estimation Using Depth-Attention Volume",
abstract = "Recovering the scene depth from a single image is an ill-posed problem that requires additional priors, often referred to as monocular depth cues, to disambiguate different 3D interpretations. In recent works, those priors have been learned in an end-to-end manner from large datasets by using deep neural networks. In this paper, we propose guiding depth estimation to favor planar structures that are ubiquitous especially in indoor environments. This is achieved by incorporating a non-local coplanarity constraint to the network with a novel attention mechanism called depth-attention volume (DAV). Experiments on two popular indoor datasets, namely NYU-Depth-v2 and ScanNet, show that our method achieves state-of-the-art depth estimation results while using only a fraction of the number of parameters needed by the competing methods. Code is available at: https://github.com/HuynhLam/DAV.",
keywords = "Attention mechanism, Depth estimation, Monocular depth",
author = "Lam Huynh and Phong Nguyen-Ha and Jiri Matas and Esa Rahtu and Janne Heikkil{\"a}",
note = "Publisher Copyright: {\textcopyright} 2020, Springer Nature Switzerland AG. Copyright: Copyright 2020 Elsevier B.V., All rights reserved. jufoid=62555; European Conference on Computer Vision ; Conference date: 23-08-2020 Through 28-08-2020",
year = "2020",
doi = "10.1007/978-3-030-58574-7\_35",
language = "English",
isbn = "9783030585730",
series = "Lecture Notes in Computer Science",
publisher = "Springer",
pages = "581--597",
editor = "Andrea Vedaldi and Horst Bischof and Thomas Brox and Jan-Michael Frahm",
booktitle = "Computer Vision {\textendash} ECCV 2020 - 16th European Conference, 2020, Proceedings",
}