@inproceedings{829510a9f0ae4be49b0b015a8889a549,
title = "GS4: Generating synthetic samples for semi-supervised nearest neighbor classification",
abstract = "In this paper, we propose a method to improve nearest neighbor classification accuracy under a semi-supervised setting. We call our approach GS4 (i.e., Generating Synthetic Samples Semi-Supervised). Existing self-training approaches classify unlabeled samples by exploiting local information. These samples are then incorporated into the training set of labeled data. However, errors are propagated and misclassifications at an early stage severely degrade the classification accuracy. To address this problem, the proposed method exploits the unlabeled data by using weights proportional to the classification confidence to generate synthetic samples. Specifically, our scheme is inspired by the Synthetic Minority Over-Sampling Technique. That is, each unlabeled sample is used to generate as many labeled samples as the number of classes represented by its k-nearest neighbors. In particular, the distance of each synthetic sample from its k-nearest neighbors of the same class is proportional to the classification confidence. As a result, the robustness to misclassification errors is increased and better accuracy is achieved. Experimental results using publicly available datasets demonstrate that statistically significant improvements are obtained when the proposed approach is employed.",
keywords = "Classification, K-nearest neighbor, Semi-supervised learning, Synthetic samples",
author = "Panagiotis Moutafis and Kakadiaris, \{Ioannis A.\}",
note = "Publisher Copyright: {\textcopyright} Springer International Publishing Switzerland 2014.; International Workshops on Data Analytics for Targeted Healthcare, DANTH 2014, Biologically Inspired Data Mining Techniques, BDM 2014, Mobile Data Management, Mining, and Computing on Social Networks, MobiSocia 2014, Big Data Science and Engineering on E-Commerce, BigEC 2014, Cloud Service Discovery, CloudSD 2014, Mobile Sensing, Mining and Visualization for Human Behavior Inferences, MSMV-MBI 2014, Scalable Dats Analytics: Theory and Algorithms, SDA 2014, Data Mining and Decision Analytics for Public Health and Wellness, DMDA-Health 2014, Algorithms for Large-Scale Information Processing in Knowledge Discovery, ALSIP 2014, Data Mining in Social Networks, SocNet 2014, Data Mining in Biomedical informatics and Healthcare, DMBIH 2014, Pattern Mining and Application of Big Data, BigPMA 2014 and Pacific Asia Workshop on Intelligence and Security Informatics, PAISI 2014, in conjunction with 18th Pacific-Asia Conference on Knowledge Discovery and Data Mining, PAKDD 2014 ; Conference date: 13-05-2014 Through 16-05-2014",
year = "2014",
doi = "10.1007/978-3-319-13186-3\_36",
language = "English (US)",
series = "Lecture Notes in Computer Science",
publisher = "Springer-Verlag",
pages = "393--403",
editor = "Wen-Chih Peng and Haixun Wang and James Bailey and Tseng, \{Vincent S.\} and Ho, \{Tu Bao\} and Zhi-Hua Zhou and Chen, \{Arbee L.P.\}",
booktitle = "Trends and Applications in Knowledge Discovery and Data Mining - PAKDD 2014 International Workshops",
}