@inproceedings{8ecfae6433424ab3a58b3a8bfb2c6a8b,
title = "DISCO: Describing images using scene contexts and objects",
abstract = "In this paper, we propose a bottom-up approach to generating short descriptive sentences from images, to enhance scene understanding. We demonstrate automatic methods for mapping the visual content in an image to natural spoken or written language. We also introduce a human-in-the-loop evaluation strategy that quantitatively captures the meaningfulness of the generated sentences. We recorded a correctness rate of 60.34\% when human users were asked to judge the meaningfulness of the sentences generated from relatively challenging images. Also, our automatic methods compared well with the state-of-the-art techniques for the related computer vision tasks.",
author = "Ifeoma Nwogu and Yingbo Zhou and Christopher Brown",
year = "2011",
language = "English",
isbn = "9781577355090",
series = "Proceedings of the National Conference on Artificial Intelligence",
pages = "1487--1493",
booktitle = "AAAI-11 / IAAI-11 - Proceedings of the 25th AAAI Conference on Artificial Intelligence and the 23rd Innovative Applications of Artificial Intelligence Conference",
note = "25th AAAI Conference on Artificial Intelligence and the 23rd Innovative Applications of Artificial Intelligence Conference, AAAI-11 / IAAI-11 ; Conference date: 07-08-2011 Through 11-08-2011",
}