@inproceedings{4d1f50cd1da447a691e2b0960426f315,
title = "Joint semantic and geometric segmentation of videos with a stage model",
abstract = "We address the problem of geometric and semantic consistent video segmentation for outdoor scenes. With no assumption on camera movement, we jointly model the semantic-geometric class of spatio-temporal regions (supervoxels) and geometric scene layout in each frame. Our main contribution is to propose a stage scene model to efficiently capture the dependency between the semantic and geometric labels. We build a unified CRF model on supervoxel labels and stage parameters, and design an alternating inference algorithm to minimize the resulting energy function. We also extend smoothing based on hierarchical image segmentation to spatio-temporal setting and show it achieves better performance than a pairwise random field model. Our method is evaluated on the CamVid dataset and achieves state-of-the-art per-pixel as well as per-class accuracy in predicting both semantic and geometric labels.",
author = "Buyu Liu and Xuming He and Stephen Gould",
year = "2014",
doi = "10.1109/WACV.2014.6836029",
language = "English",
isbn = "9781479949854",
series = "2014 IEEE Winter Conference on Applications of Computer Vision, WACV 2014",
publisher = "IEEE Computer Society",
pages = "737--744",
booktitle = "2014 IEEE Winter Conference on Applications of Computer Vision, WACV 2014",
address = "United States",
note = "2014 IEEE Winter Conference on Applications of Computer Vision, WACV 2014 ; Conference date: 24-03-2014 Through 26-03-2014",
}