@inproceedings{5b5cd37b3a4c4c7388b89da5b352cd79,
title = "A discriminative method for protein remote homology detection based on N-nary profiles",
abstract = "Protein homology detection is a key problem in computational biology. In this paper, a novel building block for protein called N-nary profile which contains the evolutionary information of protein sequence frequency profiles has been presented. The protein sequence frequency profiles calculated from the multiple sequence alignments outputted by PSI-BLAST are converted into N-nary profiles. Such N-nary profiles are filtered by a feature selection algorithm called chi-square algorithm. The protein sequences are transformed into fixed-dimension feature vectors by the occurrence times of each N-nary profile and then the corresponding vectors are inputted to support vector machine (SVM). The latent semantic analysis (LSA) model, an efficient feature extraction algorithm, is adopted to further improve the performance of this method. When tested on the SCOP 1.53 data set, the prediction performance of N-nary profile method outperforms all compared methods of protein remote homology detection. The ROC50 score is 0.736, which is higher than the current best method for nearly 4 percent.",
keywords = "Chi-square algorithm, Latent semantic analysis, N-nary profiles, Remote homology",
author = "Bin Liu and Lei Lin and Xiaolong Wang and Qiwen Dong and Xuan Wang",
year = "2008",
doi = "10.1007/978-3-540-70600-7_6",
language = "English",
isbn = "9783540705987",
series = "Communications in Computer and Information Science",
publisher = "Springer Verlag",
pages = "74--86",
booktitle = "Bioinformatics Research and Development - Second International Conference, BIRD 2008, Proceedings",
address = "Germany",
note = "2nd International Conference on Bioinformatics Research and Development, BIRD 2008 ; Conference date: 07-07-2008 Through 09-07-2008",
}