This paper proposes an end-to-end learning framework for multiview stereopsis. We term the network SurfaceNet. It takes a set of images and their corresponding camera parameters as input and directly infers the 3D model. The key advantage of the framework is that both photo-consistency as well geometric relations of the surface structure can be directly learned for the purpose of multiview stereopsis in an end-to-end fashion. SurfaceNet is a fully 3D convolutional network which is achieved by encoding the camera parameters together with the images in a 3D voxel representation. We evaluate SurfaceNet on the large-scale DTU benchmark.
@article{arxiv.1708.01749,
title = {SurfaceNet: An End-to-end 3D Neural Network for Multiview Stereopsis},
author = {Mengqi Ji and Juergen Gall and Haitian Zheng and Yebin Liu and Lu Fang},
journal= {arXiv preprint arXiv:1708.01749},
year = {2020}
}