
ACADEMIC PROFILES
SOCIAL
REPOSITORIES
CONTACTS
+39 049 827 6260
+39 049 827 6269
Journal Articles
2013
Piovesan D; Martelli P L; Fariselli P; Profiti G; Zauli A; Rossi I; Casadio R
How to inherit statistically validated annotation within BAR+ protein clusters Journal Article
In: BMC Bioinformatics, vol. 14, no. SUPPL.3, 2013, (Cited by: 7; Open Access).
Abstract | Altmetric | Dimensions | PlumX | Links:
@article{SCOPUS_ID:84879314610,
title = {How to inherit statistically validated annotation within BAR+ protein clusters},
author = {Damiano Piovesan and Pier Luigi Martelli and Piero Fariselli and Giuseppe Profiti and Andrea Zauli and Ivan Rossi and Rita Casadio},
url = {https://www.scopus.com/record/display.uri?eid=2-s2.0-84879314610&origin=inward},
doi = {10.1186/1471-2105-14-S3-S4},
year = {2013},
date = {2013-01-01},
journal = {BMC Bioinformatics},
volume = {14},
number = {SUPPL.3},
abstract = {Background: In the genomic era a key issue is protein annotation, namely how to endow protein sequences, upon translation from the corresponding genes, with structural and functional features. Routinely this operation is electronically done by deriving and integrating information from previous knowledge. The reference database for protein sequences is UniProtKB divided into two sections, UniProtKB/TrEMBL which is automatically annotated and not reviewed and UniProtKB/Swiss-Prot which is manually annotated and reviewed. The annotation process is essentially based on sequence similarity search. The question therefore arises as to which extent annotation based on transfer by inheritance is valuable and specifically if it is possible to statistically validate inherited features when little homology exists among the target sequence and its template(s).Results: In this paper we address the problem of annotating protein sequences in a statistically validated manner considering as a reference annotation resource UniProtKB. The test case is the set of 48,298 proteins recently released by the Critical Assessment of Function Annotations (CAFA) organization. We show that we can transfer after validation, Gene Ontology (GO) terms of the three main categories and Pfam domains to about 68% and 72% of the sequences, respectively. This is possible after alignment of the CAFA sequences towards BAR+, our annotation resource that allows discriminating among statistically validated and not statistically validated annotation. By comparing with a direct UniProtKB annotation, we find that besides validating annotation of some 78% of the CAFA set, we assign new and statistically validated annotation to 14.8% of the sequences and find new structural templates for about 25% of the chains, half of which share less than 30% sequence identity to the corresponding template/s.Conclusion: Inheritance of annotation by transfer generally requires a careful selection of the identity value among the target and the template in order to transfer structural and/or functional features. Here we prove that even distantly remote homologs can be safely endowed with structural templates and GO and/or Pfam terms provided that annotation is done within clusters collecting cluster-related protein sequences and where a statistical validation of the shared structural and functional features is possible. © 2013 Piovesan et al.; licensee BioMed Central Ltd.},
note = {Cited by: 7; Open Access},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Piovesan D; Profiti G; Martelli P L; Fariselli P; Fontanesi L; Casadio R
SUS-BAR: A database of pig proteins with statistically validated structural and functional annotation Journal Article
In: Database, vol. 2013, 2013, (Cited by: 5; Open Access).
Abstract | Altmetric | Dimensions | PlumX | Links:
@article{SCOPUS_ID:84892759581,
title = {SUS-BAR: A database of pig proteins with statistically validated structural and functional annotation},
author = {Damiano Piovesan and Giuseppe Profiti and Pier Luigi Martelli and Piero Fariselli and Luca Fontanesi and Rita Casadio},
url = {https://www.scopus.com/record/display.uri?eid=2-s2.0-84892759581&origin=inward},
doi = {10.1093/database/bat065},
year = {2013},
date = {2013-01-01},
journal = {Database},
volume = {2013},
abstract = {Given the relevance of the pig proteome in different studies, including human complex maladies, a statistical validation of the annotation is required for a better understanding of the role of specific genes and proteins in the complex networks underlying biological processes in the animal. Presently, approximately 80% of the pig proteome is still poorly annotated, and the existence of protein sequences is routinely inferred automatically by sequence alignment towards preexisting sequences. In this article, we introduce SUS-BAR, a database that derives information mainly from UniProt Knowledgebase and that includes 26 206 pig protein sequences. In SUS-BAR, 16 675 of the pig protein sequences are endowed with statistically validated functional and structural annotation. Our statistical validation is determined by adopting a cluster-centric annotation procedure that allows transfer of different types of annotation, including structure and function. Each sequence in the database can be associated with a set of statistically validated Gene Ontologies (GOs) of the three main subontologies (Molecular Function, Biological Process and Cellular Component), with Pfam functional domains, and when possible, with a cluster Hidden Markov Model that allows modelling the 3D structure of the protein. A database search allows some statistics demonstrating the enrichment in both GO and Pfam annotations of the pig proteins as compared with UniProt Knowledgebase annotation. Searching in SUS-BAR allows retrieval of the pig protein annotation for further analysis. The search is also possible on the basis of specific GO terms and this allows retrieval of all the pig sequences participating into a given biological process, after annotation with our system. Alternatively, the search is possible on the basis of structural information, allowing retrieval of all the pig sequences with the same structural characteristics. © The Author(s) 2013. Published by Oxford University Press.},
note = {Cited by: 5; Open Access},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Piovesan D; Profiti G; Martelli P L; Fariselli P; Casadio R
Extended and robust protein sequence annotation over conservative nonhierarchical clusters: The case study of the abc transporters Journal Article
In: ACM Journal on Emerging Technologies in Computing Systems, vol. 9, no. 4, 2013, (Cited by: 2).
Abstract | Altmetric | Dimensions | PlumX | Links:
@article{SCOPUS_ID:84888597395,
title = {Extended and robust protein sequence annotation over conservative nonhierarchical clusters: The case study of the abc transporters},
author = {Damiano Piovesan and Giuseppe Profiti and Pier Luigi Martelli and Piero Fariselli and Rita Casadio},
url = {https://www.scopus.com/record/display.uri?eid=2-s2.0-84888597395&origin=inward},
doi = {10.1145/2504729},
year = {2013},
date = {2013-01-01},
journal = {ACM Journal on Emerging Technologies in Computing Systems},
volume = {9},
number = {4},
publisher = {Association for Computing Machinery},
abstract = {Genome annotation is one of the most important issues in the genomic era. The exponential growth rate of newly sequenced genomes and proteomes urges the development of fast and reliable annotation methods, suited to exploit all the information available in curated databases of protein sequences and structures. To this aim we developed BAR+, the Bologna Annotation Resource.1 The basic notion is that sequences with high identity value to a counterpart can inherit the same function/s and structure, if available. As a case study we describe how the ATP-binding domain of the ABC transporters can be found and modeled in over 30,000 new sequences not annotated before. We also mapped into BAR+ all the ABC transporters listed in the Transporter Classification DataBase2 and found that within our environment annotation could be extended to another 256,866 sequences. ©c 2013 ACM 1550-4832/2013/11-ART27 15.00.},
note = {Cited by: 2},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2012
Piovesan D; Profiti G; Martelli P L; Casadio R
The human “magnesome”: Detecting magnesium binding sites on human proteins Journal Article
In: BMC Bioinformatics, vol. 13, no. SUPPL 1, 2012, (Cited by: 29; Open Access).
Abstract | Altmetric | Dimensions | PlumX | Links:
@article{SCOPUS_ID:84872285924,
title = {The human "magnesome": Detecting magnesium binding sites on human proteins},
author = {Damiano Piovesan and Giuseppe Profiti and Pier Luigi Martelli and Rita Casadio},
url = {https://www.scopus.com/record/display.uri?eid=2-s2.0-84872285924&origin=inward},
doi = {10.1186/1471-2105-13-S14-S10},
year = {2012},
date = {2012-01-01},
journal = {BMC Bioinformatics},
volume = {13},
number = {SUPPL 1},
abstract = {Background: Magnesium research is increasing in molecular medicine due to the relevance of this ion in several important biological processes and associated molecular pathogeneses. It is still difficult to predict from the protein covalent structure whether a human chain is or not involved in magnesium binding. This is mainly due to little information on the structural characteristics of magnesium binding sites in proteins and protein complexes. Magnesium binding features, differently from those of other divalent cations such as calcium and zinc, are elusive. Here we address a question that is relevant in protein annotation: how many human proteins can bind Mg2+? Our analysis is performed taking advantage of the recently implemented Bologna Annotation Resource (BAR-PLUS), a non hierarchical clustering method that relies on the pair wise sequence comparison of about 14 millions proteins from over 300.000 species and their grouping into clusters where annotation can safely be inherited after statistical validation. Results: After cluster assignment of the latest version of the human proteome, the total number of human proteins for which we can assign putative Mg binding sites is 3,751. Among these proteins, 2,688 inherit annotation directly from human templates and 1,063 inherit annotation from templates of other organisms. Protein structures are highly conserved inside a given cluster. Transfer of structural properties is possible after alignment of a given sequence with the protein structures that characterise a given cluster as obtained with a Hidden Markov Model (HMM) based procedure. Interestingly a set of 370 human sequences inherit Mg2+ binding sites from templates sharing less than 30% sequence identity with the template. Conclusion: We describe and deliver the "human magnesome", a set of proteins of the human proteome that inherit putative binding of magnesium ions. With our BAR-hMG, 251 clusters including 1,341 magnesium binding protein structures corresponding to 387 sequences are sufficient to annotate some 13,689 residues in 3,751 human sequences as "magnesium binding". Protein structures act therefore as three dimensional seeds for structural and functional annotation of human sequences. The data base collects specifically all the human proteins that can be annotated according to our procedure as "magnesium binding", the corresponding structures and BAR+ clusters from where they derive the annotation (http://bar.biocomp.unibo.it/mg). © 2012 Piovesan et al.; licensee BioMed Central Ltd.},
note = {Cited by: 29; Open Access},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2011
Piovesan D; Martelli P L; Fariselli P; Zauli A; Rossi I; Casadio R
BAR-PLUS: The Bologna Annotation Resource Plus for functional and structural annotation of protein sequences Journal Article
In: Nucleic Acids Research, vol. 39, no. SUPPL. 2, 2011, (Cited by: 24; Open Access).
Abstract | Altmetric | Dimensions | PlumX | Links:
@article{SCOPUS_ID:79959918507,
title = {BAR-PLUS: The Bologna Annotation Resource Plus for functional and structural annotation of protein sequences},
author = {Damiano Piovesan and Pier Luigi Martelli and Piero Fariselli and Andrea Zauli and Ivan Rossi and Rita Casadio},
url = {https://www.scopus.com/record/display.uri?eid=2-s2.0-79959918507&origin=inward},
doi = {10.1093/nar/gkr292},
year = {2011},
date = {2011-01-01},
journal = {Nucleic Acids Research},
volume = {39},
number = {SUPPL. 2},
abstract = {We introduce BAR-PLUS (BAR+), a web server for functional and structural annotation of protein sequences. BAR+ is based on a large-scale genome cross comparison and a non-hierarchical clustering procedure characterized by a metric that ensures a reliable transfer of features within clusters. In this version, the method takes advantage of a large-scale pairwise sequence comparison of 13495736 protein chains also including 988 complete proteomes. Available sequence annotation is derived from UniProtKB, GO, Pfam and PDB. When PDB templates are present within a cluster (with or without their SCOP classification), profile Hidden Markov Models (HMMs) are computed on the basis of sequence to structure alignment and are cluster-associated (Cluster-HMM). Therefrom, a library of 10858 HMMs is made available for aligning even distantly related sequences for structural modelling. The server also provides pairwise query sequence-structural target alignments computed from the correspondent Cluster-HMM. BAR+ in its present version allows three main categories of annotation: PDB [with or without SCOP (*)] and GO and/or Pfam; PDB (*) without GO and/or Pfam; GO and/or Pfam without PDB (*) and no annotation. Each category can further comprise clusters where GO and Pfam functional annotations are or are not statistically significant. BAR+ is available at http://bar.biocomp.unibo.it/bar2.0. © 2011 The Author(s).},
note = {Cited by: 24; Open Access},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
