@inproceedings{554e239c6f714e998826e4444b328b62,
title = "An efficient distributed hierarchical-clustering algorithm for large scale data",
abstract = "The data-classification process can possibly involve a huge amount of data in today's cloud computing environment. It could take a long time for processing, and could consume many resources for computation and storage. This study focuses on the problem of using the traditional hierarchical agglomerative clustering algorithm on a distributed environment since hierarchical agglomerative clustering has high applicability and efficiency. A parallel hierarchical agglomerative clustering algorithm is proposed in this study. The proposed algorithm divides the whole computation into several small tasks, distribute the tasks to message-passing processes, and merge the results to form a hierarchical cluster. A threshold is used to reduce the storage requirement during the computation. To evaluate the performance and limitation of our algorithm, this study has conducted several experiments using real astronomical data, the main asteroid belt catalog. The experimental results confirm that the proposed parallel algorithm is efficient.",
keywords = "Hierarchical clustering, Parallel computing",
author = "Tang, {Cheng Hsien} and Huang, {An Ching} and Tsai, {Meng Feng} and Wang, {Wei Jen}",
year = "2010",
doi = "10.1109/COMPSYM.2010.5685388",
language = "???core.languages.en_GB???",
isbn = "9781424476404",
series = "ICS 2010 - International Computer Symposium",
pages = "869--874",
booktitle = "ICS 2010 - International Computer Symposium",
note = "2010 International Computer Symposium, ICS 2010 ; Conference date: 16-12-2010 Through 18-12-2010",
}