@inproceedings{d6a7e03f45b24623ba236c200e7bfe88,
title = "GPU-accelerated incremental correlation clustering of large data with visual feedback",
abstract = "Clustering is an important preparation step in big data processing. It may even be used to detect redundant data points as well as outliers. Elimination of redundant data and duplicates can serve as a viable means for data reduction and it can also aid in sampling. Visual feedback is very valuable here to give users confidence in this process. Furthermore, big data preprocessing is seldom interactive, which stands at conflict with users who seek answers immediately. The best one can do is incremental preprocessing in which partial and hopefully quite accurate results become available relatively quickly and are then refined over time. We propose a correlation clustering framework which uses MDS for layout and GPU-acceleration to accomplish these goals. Our domain application is the correlation clustering of atmospheric mass spectrum data with 8 million data points of 450 dimensions each.",
keywords = "big data, clustering, correlation, GPU, visual analytics, visualization",
author = "Eric Papenhausen and Bing Wang and Sungsoo Ha and Alla Zelenyuk and Dan Imre and Klaus Mueller",
year = "2013",
doi = "10.1109/BigData.2013.6691716",
language = "English",
isbn = "9781479912926",
series = "Proceedings - 2013 IEEE International Conference on Big Data, Big Data 2013",
publisher = "IEEE Computer Society",
pages = "63--70",
booktitle = "Proceedings - 2013 IEEE International Conference on Big Data, Big Data 2013",
note = "2013 IEEE International Conference on Big Data, Big Data 2013 ; Conference date: 06-10-2013 Through 09-10-2013",
}