With the rise of community-generated web content, the need for automatic assessment of resource quality has grown. We demonstrate how developing a concrete characterization of quality for web-based resources can make machine learning approaches to automating quality assessment in the realm of educational digital libraries tractable. Using data from several previous studies of quality, we gathered a set of key dimensions and indicators of quality that were commonly identified by educators. We then performed a mixed-method study of digital library quality experts, showing that our characterization of quality captured the subjective processes used by the experts when assessing resource quality. Using key indicators of quality selected from a statistical analysis of our expert study data, we developed a set of annotation guidelines and annotated a corpus of 1000 digital resources for the presence or absence of the key quality indicators. Agreement among annotators was high, and initial machine learning models trained from this corpus were able to identify some indicators of quality with as much as an 18% improvement over the baseline.
Description
Automatically assessing resource quality for educational digital libraries
%0 Conference Paper
%1 Wetzler:2009:AAR:1526993.1526997
%A Wetzler, Philipp G.
%A Bethard, Steven
%A Butcher, Kirsten
%A Martin, James H.
%A Sumner, Tamara
%B Proceedings of the 3rd workshop on Information credibility on the web
%C New York, NY, USA
%D 2009
%I ACM
%K assessing automatically content educational generated quality resource user
%P 3--10
%R 10.1145/1526993.1526997
%T Automatically assessing resource quality for educational digital libraries
%U http://doi.acm.org/10.1145/1526993.1526997
%X With the rise of community-generated web content, the need for automatic assessment of resource quality has grown. We demonstrate how developing a concrete characterization of quality for web-based resources can make machine learning approaches to automating quality assessment in the realm of educational digital libraries tractable. Using data from several previous studies of quality, we gathered a set of key dimensions and indicators of quality that were commonly identified by educators. We then performed a mixed-method study of digital library quality experts, showing that our characterization of quality captured the subjective processes used by the experts when assessing resource quality. Using key indicators of quality selected from a statistical analysis of our expert study data, we developed a set of annotation guidelines and annotated a corpus of 1000 digital resources for the presence or absence of the key quality indicators. Agreement among annotators was high, and initial machine learning models trained from this corpus were able to identify some indicators of quality with as much as an 18% improvement over the baseline.
%@ 978-1-60558-488-1
@inproceedings{Wetzler:2009:AAR:1526993.1526997,
abstract = {With the rise of community-generated web content, the need for automatic assessment of resource quality has grown. We demonstrate how developing a concrete characterization of quality for web-based resources can make machine learning approaches to automating quality assessment in the realm of educational digital libraries tractable. Using data from several previous studies of quality, we gathered a set of key dimensions and indicators of quality that were commonly identified by educators. We then performed a mixed-method study of digital library quality experts, showing that our characterization of quality captured the subjective processes used by the experts when assessing resource quality. Using key indicators of quality selected from a statistical analysis of our expert study data, we developed a set of annotation guidelines and annotated a corpus of 1000 digital resources for the presence or absence of the key quality indicators. Agreement among annotators was high, and initial machine learning models trained from this corpus were able to identify some indicators of quality with as much as an 18% improvement over the baseline.},
acmid = {1526997},
added-at = {2011-12-08T13:26:16.000+0100},
address = {New York, NY, USA},
author = {Wetzler, Philipp G. and Bethard, Steven and Butcher, Kirsten and Martin, James H. and Sumner, Tamara},
biburl = {https://www.bibsonomy.org/bibtex/20d58ee1d570da9fe9693dfa79d710259/griesbau},
booktitle = {Proceedings of the 3rd workshop on Information credibility on the web},
description = {Automatically assessing resource quality for educational digital libraries},
doi = {10.1145/1526993.1526997},
interhash = {e1dcc74cdcf5c87fce9bda4a6044faae},
intrahash = {0d58ee1d570da9fe9693dfa79d710259},
isbn = {978-1-60558-488-1},
keywords = {assessing automatically content educational generated quality resource user},
location = {Madrid, Spain},
numpages = {8},
pages = {3--10},
publisher = {ACM},
series = {WICOW '09},
timestamp = {2011-12-08T13:26:16.000+0100},
title = {Automatically assessing resource quality for educational digital libraries},
url = {http://doi.acm.org/10.1145/1526993.1526997},
year = 2009
}