Automatically clustering web pages into semantic groups promises improved search and browsing on the web. In this paper, we demonstrate how user-generated tags from large-scale social bookmarking websites such as del.icio.us can be used as a complementary data source to page text and anchor text for improving automatic clustering of web pages. This paper explores the use of tags in 1) K-means clustering in an extended vector space model that includes tags as well as page text and 2) a novel generative clustering algorithm based on latent Dirichlet allocation that jointly models text and tags. We evaluate the models by comparing their output to an established web directory. We find that the naive inclusion of tagging data improves cluster quality versus page text alone, but a more principled inclusion can substantially improve the quality of all models with a statistically significant absolute F-score increase of 4\%. The generative model outperforms K-means with another 8\% F-score increase.
%0 Conference Paper
%1 citeulike:4079568
%A Ramage, Daniel
%A Heymann, Paul
%A Manning, Christopher D.
%A Molina, Hector G.
%B Proceedings of the Second ACM International Conference on Web Search and Data Mining
%C New York, NY, USA
%D 2009
%I ACM
%K clustering, dlpaws, tagging
%P 54--63
%R 10.1145/1498759.1498809
%T Clustering the tagged web
%U http://dx.doi.org/10.1145/1498759.1498809
%X Automatically clustering web pages into semantic groups promises improved search and browsing on the web. In this paper, we demonstrate how user-generated tags from large-scale social bookmarking websites such as del.icio.us can be used as a complementary data source to page text and anchor text for improving automatic clustering of web pages. This paper explores the use of tags in 1) K-means clustering in an extended vector space model that includes tags as well as page text and 2) a novel generative clustering algorithm based on latent Dirichlet allocation that jointly models text and tags. We evaluate the models by comparing their output to an established web directory. We find that the naive inclusion of tagging data improves cluster quality versus page text alone, but a more principled inclusion can substantially improve the quality of all models with a statistically significant absolute F-score increase of 4\%. The generative model outperforms K-means with another 8\% F-score increase.
%@ 978-1-60558-390-7
@inproceedings{citeulike:4079568,
abstract = {{Automatically clustering web pages into semantic groups promises improved search and browsing on the web. In this paper, we demonstrate how user-generated tags from large-scale social bookmarking websites such as del.icio.us can be used as a complementary data source to page text and anchor text for improving automatic clustering of web pages. This paper explores the use of tags in 1) K-means clustering in an extended vector space model that includes tags as well as page text and 2) a novel generative clustering algorithm based on latent Dirichlet allocation that jointly models text and tags. We evaluate the models by comparing their output to an established web directory. We find that the naive inclusion of tagging data improves cluster quality versus page text alone, but a more principled inclusion can substantially improve the quality of all models with a statistically significant absolute F-score increase of 4\%. The generative model outperforms K-means with another 8\% F-score increase.}},
added-at = {2017-11-15T17:02:25.000+0100},
address = {New York, NY, USA},
author = {Ramage, Daniel and Heymann, Paul and Manning, Christopher D. and Molina, Hector G.},
biburl = {https://www.bibsonomy.org/bibtex/2a3b1ee1739eae79ffdd79d96a9e43d9a/brusilovsky},
booktitle = {Proceedings of the Second ACM International Conference on Web Search and Data Mining},
citeulike-article-id = {4079568},
citeulike-linkout-0 = {http://portal.acm.org/citation.cfm?id=1498759.1498809},
citeulike-linkout-1 = {http://dx.doi.org/10.1145/1498759.1498809},
doi = {10.1145/1498759.1498809},
interhash = {3d999a9a9e0e4edde2c64c247e98efa1},
intrahash = {a3b1ee1739eae79ffdd79d96a9e43d9a},
isbn = {978-1-60558-390-7},
keywords = {clustering, dlpaws, tagging},
location = {Barcelona, Spain},
pages = {54--63},
posted-at = {2009-03-18 18:05:52},
priority = {4},
publisher = {ACM},
series = {WSDM '09},
timestamp = {2017-11-15T17:02:25.000+0100},
title = {{Clustering the tagged web}},
url = {http://dx.doi.org/10.1145/1498759.1498809},
year = 2009
}