@inproceedings{a6c96a4ab1014a09a68bdf4b3ba8174c,
title = "Using protein domains to improve the accuracy of Ab Initio gene finding",
abstract = "Background: Protein domains are the common functional elements used by nature to generate tremendous diversity among proteins, and they are used repeatedly in different combinations across all major domains of life. In this paper we address the problem of using similarity to known protein domains in helping with the identification of genes in a DNA sequence. We have adapted the generalized hidden Markov model (GHMM) architecture of the ab intio gene finder GlimmerHMM such that a higher probability is assigned to exons that contain homologues to protein domains. To our knowledge, this domain homology based approach has not been used previously in the context of ab initio gene prediction. Results: GlimmerHMM was augmented with a protein domain module that recognizes gene structures that are similar to Pfam models. The augmented system, GlimmerHMM+, shows 2% improvement in sensitivity and a 1% increase in specificity in predicting exact gene structures compared to GlimmerHMM without this option. These results were obtained on two very different model organisms: Arabidopsis thaliana (mustard wee) and Danio rerio (zebrafish), and together these preliminary results demonstrate the value of using protein domain homology in gene prediction. The results obtained are encouraging, and we believe that a more comprehensive approach including a model that reflects the statistical characteristics of specific sets of protein domain families would result in a greater increase of the accuracy of gene prediction. GlimmerHMM and GlimmerHMM+ are freely available as open source software at http://cbcb.umd.edu/software.",
keywords = "GHMM, Pfam, Profile HMM, Protein domain, ab intio gene finding",
author = "Mihaela Pertea and Salzberg, {Steven L.}",
year = "2007",
doi = "10.1007/978-3-540-74126-8_20",
language = "English (US)",
isbn = "9783540741251",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer Verlag",
pages = "208--215",
booktitle = "Algorithms in Bioinformatics - 7th International Workshop, WABI 2007, Proceedings",
note = "7th International Workshop on Algorithms in Bioinformatics, WABI 2007 ; Conference date: 08-09-2007 Through 09-09-2007",
}