Group Publications
2016
Baglio, S. Rubina; Eijndhoven, Monique A. J.; Koppers-Lalic, Danijela; Berenguer, Jordi; Lougheed, Sinéad M.; Gibbs, Susan; Léveillé, Nicolas; Rinkel, Rico N. P. M.; Hopmans, Erik S.; Swaminathan, Sankar; Verkuijlen, Sandra A. W. M.; Scheffer, George L.; Kuppeveld, Frank J. M.; Gruijl, Tanja D.; Bultink, Irene E. M.; Jordanova, Ekaterina S.; Hackenberg, Michael; Piersma, Sander R.; Knol, Jaco C.; Voskuyl, Alexandre E.; Wurdinger, Thomas; Jiménez, Connie R.; Middeldorp, Jaap M.; Pegtel, D. Michiel
Sensing of latent EBV infection through exosomal transfer of 5'pppRNA Journal Article
In: Proceedings of the National Academy of Sciences of the United States of America, vol. 113, no. 5, pp. E587–596, 2016, ISSN: 1091-6490.
@article{baglio_sensing_2016,
title = {Sensing of latent EBV infection through exosomal transfer of 5'pppRNA},
author = {S. Rubina Baglio and Monique A. J. Eijndhoven and Danijela Koppers-Lalic and Jordi Berenguer and Sinéad M. Lougheed and Susan Gibbs and Nicolas Léveillé and Rico N. P. M. Rinkel and Erik S. Hopmans and Sankar Swaminathan and Sandra A. W. M. Verkuijlen and George L. Scheffer and Frank J. M. Kuppeveld and Tanja D. Gruijl and Irene E. M. Bultink and Ekaterina S. Jordanova and Michael Hackenberg and Sander R. Piersma and Jaco C. Knol and Alexandre E. Voskuyl and Thomas Wurdinger and Connie R. Jiménez and Jaap M. Middeldorp and D. Michiel Pegtel},
doi = {10.1073/pnas.1518130113},
issn = {1091-6490},
year = {2016},
date = {2016-02-01},
journal = {Proceedings of the National Academy of Sciences of the United States of America},
volume = {113},
number = {5},
pages = {E587–596},
abstract = {Complex interactions between DNA herpesviruses and host factors determine the establishment of a life-long asymptomatic latent infection. The lymphotropic Epstein-Barr virus (EBV) seems to avoid recognition by innate sensors despite massive transcription of immunostimulatory small RNAs (EBV-EBERs). Here we demonstrate that in latently infected B cells, EBER1 transcripts interact with the lupus antigen (La) ribonucleoprotein, avoiding cytoplasmic RNA sensors. However, in coculture experiments we observed that latent-infected cells trigger antiviral immunity in dendritic cells (DCs) through selective release and transfer of RNA via exosomes. In ex vivo tonsillar cultures, we observed that EBER1-loaded exosomes are preferentially captured and internalized by human plasmacytoid DCs (pDCs) that express the TIM1 phosphatidylserine receptor, a known viral- and exosomal target. Using an EBER-deficient EBV strain, enzymatic removal of 5'ppp, in vitro transcripts, and coculture experiments, we established that 5'pppEBER1 transfer via exosomes drives antiviral immunity in nonpermissive DCs. Lupus erythematosus patients suffer from elevated EBV load and activated antiviral immunity, in particular in skin lesions that are infiltrated with pDCs. We detected high levels of EBER1 RNA in such skin lesions, as well as EBV-microRNAs, but no intact EBV-DNA, linking non-cell-autonomous EBER1 presence with skin inflammation in predisposed individuals. Collectively, our studies indicate that virus-modified exosomes have a physiological role in the host-pathogen stand-off and may promote inflammatory disease.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2015
Rueda, Antonio; Barturen, Guillermo; Lebrón, Ricardo; Gómez-Martín, Cristina; Alganza, Ángel; Oliver, José L.; Hackenberg, Michael
sRNAtoolbox: an integrated collection of small RNA research tools Journal Article
In: Nucleic Acids Research, vol. 43, no. W1, pp. W467–473, 2015, ISSN: 1362-4962.
@article{rueda_srnatoolbox_2015,
title = {sRNAtoolbox: an integrated collection of small RNA research tools},
author = {Antonio Rueda and Guillermo Barturen and Ricardo Lebrón and Cristina Gómez-Martín and Ángel Alganza and José L. Oliver and Michael Hackenberg},
doi = {10.1093/nar/gkv555},
issn = {1362-4962},
year = {2015},
date = {2015-07-01},
journal = {Nucleic Acids Research},
volume = {43},
number = {W1},
pages = {W467–473},
abstract = {Small RNA research is a rapidly growing field. Apart from microRNAs, which are important regulators of gene expression, other types of functional small RNA molecules have been reported in animals and plants. MicroRNAs are important in host-microbe interactions and parasite microRNAs might modulate the innate immunity of the host. Furthermore, small RNAs can be detected in bodily fluids making them attractive non-invasive biomarker candidates. Given the general broad interest in small RNAs, and in particular microRNAs, a large number of bioinformatics aided analysis types are needed by the scientific community. To facilitate integrated sRNA research, we developed sRNAtoolbox, a set of independent but interconnected tools for expression profiling from high-throughput sequencing data, consensus differential expression, target gene prediction, visual exploration in a genome context as a function of read length, gene list analysis and blast search of unmapped reads. All tools can be used independently or for the exploration and downstream analysis of sRNAbench results. Workflows like the prediction of consensus target genes of parasite microRNAs in the host followed by the detection of enriched pathways can be easily established. The web-interface interconnecting all these tools is available at http://bioinfo5.ugr.es/srnatoolbox.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Li, Zhiguang; Qin, Taichun; Wang, Kejian; Hackenberg, Michael; Yan, Jian; Gao, Yuan; Yu, Li-Rong; Shi, Leming; Su, Zhenqiang; Chen, Tao
In: BMC genomics, vol. 16, no. 1, pp. 365, 2015, ISSN: 1471-2164.
@article{li_integrated_2015,
title = {Integrated microRNA, mRNA, and protein expression profiling reveals microRNA regulatory networks in rat kidney treated with a carcinogenic dose of aristolochic acid},
author = {Zhiguang Li and Taichun Qin and Kejian Wang and Michael Hackenberg and Jian Yan and Yuan Gao and Li-Rong Yu and Leming Shi and Zhenqiang Su and Tao Chen},
doi = {10.1186/s12864-015-1516-2},
issn = {1471-2164},
year = {2015},
date = {2015-05-01},
journal = {BMC genomics},
volume = {16},
number = {1},
pages = {365},
abstract = {BACKGROUND: Aristolochic Acid (AA), a natural component of Aristolochia plants that is found in a variety of herbal remedies and health supplements, is classified as a Group 1 carcinogen by the International Agency for Research on Cancer. Given that microRNAs (miRNAs) are involved in cancer initiation and progression and their role remains unknown in AA-induced carcinogenesis, we examined genome-wide AA-induced dysregulation of miRNAs as well as the regulation of miRNAs on their target gene expression in rat kidney.
RESULTS: We treated rats with 10 mg/kg AA and vehicle control for 12 weeks and eight kidney samples (4 for the treatment and 4 for the control) were used for examining miRNA and mRNA expression by deep sequencing, and protein expression by proteomics. AA treatment resulted in significant differential expression of miRNAs, mRNAs and proteins as measured by both principal component analysis (PCA) and hierarchical clustering analysis (HCA). Specially, 63 miRNAs (adjusted p value < 0.05 and fold change > 1.5), 6,794 mRNAs (adjusted p value < 0.05 and fold change > 2.0), and 800 proteins (fold change > 2.0) were significantly altered by AA treatment. The expression of 6 selected miRNAs was validated by quantitative real-time PCR analysis. Ingenuity Pathways Analysis (IPA) showed that cancer is the top network and disease associated with those dysregulated miRNAs. To further investigate the influence of miRNAs on kidney mRNA and protein expression, we combined proteomic and transcriptomic data in conjunction with miRNA target selection as confirmed and reported in miRTarBase. In addition to translational repression and transcriptional destabilization, we also found that miRNAs and their target genes were expressed in the same direction at levels of transcription (169) or translation (227). Furthermore, we identified that up-regulation of 13 oncogenic miRNAs was associated with translational activation of 45 out of 54 cancer-related targets.
CONCLUSIONS: Our findings suggest that dysregulated miRNA expression plays an important role in AA-induced carcinogenesis in rat kidney, and that the integrated approach of multiple profiling provides a new insight into a post-transcriptional regulation of miRNAs on their target repression and activation in a genome-wide scale.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We treated rats with 10 mg/kg AA and vehicle control for 12 weeks and eight kidney samples (4 for the treatment and 4 for the control) were used for examining miRNA and mRNA expression by deep sequencing, and protein expression by proteomics. AA treatment resulted in significant differential expression of miRNAs, mRNAs and proteins as measured by both principal component analysis (PCA) and hierarchical clustering analysis (HCA). Specially, 63 miRNAs (adjusted p value < 0.05 and fold change > 1.5), 6,794 mRNAs (adjusted p value < 0.05 and fold change > 2.0), and 800 proteins (fold change > 2.0) were significantly altered by AA treatment. The expression of 6 selected miRNAs was validated by quantitative real-time PCR analysis. Ingenuity Pathways Analysis (IPA) showed that cancer is the top network and disease associated with those dysregulated miRNAs. To further investigate the influence of miRNAs on kidney mRNA and protein expression, we combined proteomic and transcriptomic data in conjunction with miRNA target selection as confirmed and reported in miRTarBase. In addition to translational repression and transcriptional destabilization, we also found that miRNAs and their target genes were expressed in the same direction at levels of transcription (169) or translation (227). Furthermore, we identified that up-regulation of 13 oncogenic miRNAs was associated with translational activation of 45 out of 54 cancer-related targets.
CONCLUSIONS: Our findings suggest that dysregulated miRNA expression plays an important role in AA-induced carcinogenesis in rat kidney, and that the integrated approach of multiple profiling provides a new insight into a post-transcriptional regulation of miRNAs on their target repression and activation in a genome-wide scale.
Hackenberg, Michael; Gustafson, Perry; Langridge, Peter; Shi, Bu-Jun
Differential expression of microRNAs and other small RNAs in barley between water and drought conditions Journal Article
In: Plant Biotechnology Journal, vol. 13, no. 1, pp. 2–13, 2015, ISSN: 1467-7652.
@article{hackenberg_differential_2015,
title = {Differential expression of microRNAs and other small RNAs in barley between water and drought conditions},
author = {Michael Hackenberg and Perry Gustafson and Peter Langridge and Bu-Jun Shi},
doi = {10.1111/pbi.12220},
issn = {1467-7652},
year = {2015},
date = {2015-01-01},
journal = {Plant Biotechnology Journal},
volume = {13},
number = {1},
pages = {2–13},
abstract = {Drought is a major constraint to crop production, and microRNAs (miRNAs) play an important role in plant drought tolerance. Analysis of miRNAs and other classes of small RNAs (sRNAs) in barley grown under water and drought conditions reveals that drought selectively regulates expression of miRNAs and other classes of sRNAs. Low-expressed miRNAs and all repeat-associated siRNAs (rasiRNAs) tended towards down-regulation, while tRNA-derived sRNAs (tsRNAs) had the tendency to be up-regulated, under drought. Antisense sRNAs (putative siRNAs) did not have such a tendency under drought. In drought-tolerant transgenic barley overexpressing DREB transcription factor, most of the low-expressed miRNAs were also down-regulated. In contrast, tsRNAs, rasiRNAs and other classes of sRNAs were not consistently expressed between the drought-treated and transgenic plants. The differential expression of miRNAs and siRNAs was further confirmed by Northern hybridization and quantitative real-time PCR (qRT-PCR). Targets of the drought-regulated miRNAs and siRNAs were predicted, identified by degradome libraries and confirmed by qRT-PCR. Their functions are diverse, but most are involved in transcriptional regulation. Our data provide insight into the expression profiles of miRNAs and other sRNAs, and their relationship under drought, thereby helping understand how miRNAs and sRNAs respond to drought stress in cereal crops.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2014
Dios, Francisco; Barturen, Guillermo; Lebrón, Ricardo; Rueda, Antonio; Hackenberg, Michael; Oliver, José L.
DNA clustering and genome complexity Journal Article
In: Computational Biology and Chemistry, vol. 53 Pt A, pp. 71–78, 2014, ISSN: 1476-928X.
@article{dios_dna_2014,
title = {DNA clustering and genome complexity},
author = {Francisco Dios and Guillermo Barturen and Ricardo Lebrón and Antonio Rueda and Michael Hackenberg and José L. Oliver},
doi = {10.1016/j.compbiolchem.2014.08.011},
issn = {1476-928X},
year = {2014},
date = {2014-12-01},
journal = {Computational Biology and Chemistry},
volume = {53 Pt A},
pages = {71–78},
abstract = {Early global measures of genome complexity (power spectra, the analysis of fluctuations in DNA walks or compositional segmentation) uncovered a high degree of complexity in eukaryotic genome sequences. The main evolutionary mechanisms leading to increases in genome complexity (i.e. gene duplication and transposon proliferation) can all potentially produce increases in DNA clustering. To quantify such clustering and provide a genome-wide description of the formed clusters, we developed GenomeCluster, an algorithm able to detect clusters of whatever genome element identified by chromosome coordinates. We obtained a detailed description of clusters for ten categories of human genome elements, including functional (genes, exons, introns), regulatory (CpG islands, TFBSs, enhancers), variant (SNPs) and repeat (Alus, LINE1) elements, as well as DNase hypersensitivity sites. For each category, we located their clusters in the human genome, then quantifying cluster length and composition, and estimated the clustering level as the proportion of clustered genome elements. In average, we found a 27% of elements in clusters, although a considerable variation occurs among different categories. Genes form the lowest number of clusters, but these are the longest ones, both in bp and the average number of components, while the shortest clusters are formed by SNPs. Functional and regulatory elements (genes, CpG islands, TFBSs, enhancers) show the highest clustering level, as compared to DNase sites, repeats (Alus, LINE1) or SNPs. Many of the genome elements we analyzed are known to be composed of clusters of low-level entities. In addition, we found here that the clusters generated by GenomeCluster can be in turn clustered into high-level super-clusters. The observation of 'clusters-within-clusters' parallels the 'domains within domains' phenomenon previously detected through global statistical methods in eukaryotic sequences, and reveals a complex human genome landscape dominated by hierarchical clustering.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Schwarz, Alexandra; Tenzer, Stefan; Hackenberg, Michael; Erhart, Jan; Gerhold-Ay, Aslihan; Mazur, Johanna; Kuharev, Jörg; Ribeiro, José M. C.; Kotsyfakis, Michail
In: Molecular & cellular proteomics: MCP, vol. 13, no. 10, pp. 2725–2735, 2014, ISSN: 1535-9484.
@article{schwarz_systems_2014,
title = {A systems level analysis reveals transcriptomic and proteomic complexity in Ixodes ricinus midgut and salivary glands during early attachment and feeding},
author = {Alexandra Schwarz and Stefan Tenzer and Michael Hackenberg and Jan Erhart and Aslihan Gerhold-Ay and Johanna Mazur and Jörg Kuharev and José M. C. Ribeiro and Michail Kotsyfakis},
doi = {10.1074/mcp.M114.039289},
issn = {1535-9484},
year = {2014},
date = {2014-10-01},
journal = {Molecular & cellular proteomics: MCP},
volume = {13},
number = {10},
pages = {2725–2735},
abstract = {Although pathogens are usually transmitted within the first 24-48 h of attachment of the castor bean tick Ixodes ricinus, little is known about the tick's biological responses at these earliest phases of attachment. Tick midgut and salivary glands are the main tissues involved in tick blood feeding and pathogen transmission but the limited genomic information for I. ricinus delays the application of high-throughput methods to study their physiology. We took advantage of the latest advances in the fields of Next Generation RNA-Sequencing and Label-free Quantitative Proteomics to deliver an unprecedented, quantitative description of the gene expression dynamics in the midgut and salivary glands of this disease vector upon attachment to the vertebrate host. A total of 373 of 1510 identified proteins had higher expression in the salivary glands, but only 110 had correspondingly high transcript levels in the same tissue. Furthermore, there was midgut-specific expression of 217 genes at both the transcriptome and proteome level. Tissue-dependent transcript, but not protein, accumulation was revealed for 552 of 885 genes. Moreover, we discovered the enrichment of tick salivary glands in proteins involved in gene transcription and translation, which agrees with the secretory role of this tissue; this finding also agrees with our finding of lower tick t-RNA representation in the salivary glands when compared with the midgut. The midgut, in turn, is enriched in metabolic components and proteins that support its mechanical integrity in order to accommodate and metabolize the ingested blood. Beyond understanding the physiological events that support hematophagy by arthropod ectoparasites, we discovered more than 1500 proteins located at the interface between ticks, the vertebrate host, and the tick-borne pathogens. Thus, our work significantly improves the knowledge of the genetics underlying the transmission lifecycle of this tick species, which is an essential step for developing alternative methods to better control tick-borne diseases.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Koppers-Lalic, Danijela; Hackenberg, Michael; Bijnsdorp, Irene V.; Eijndhoven, Monique A. J.; Sadek, Payman; Sie, Daud; Zini, Nicoletta; Middeldorp, Jaap M.; Ylstra, Bauke; Menezes, Renee X.; Würdinger, Thomas; Meijer, Gerrit A.; Pegtel, D. Michiel
Nontemplated nucleotide additions distinguish the small RNA composition in cells from exosomes Journal Article
In: Cell Reports, vol. 8, no. 6, pp. 1649–1658, 2014, ISSN: 2211-1247.
@article{koppers-lalic_nontemplated_2014,
title = {Nontemplated nucleotide additions distinguish the small RNA composition in cells from exosomes},
author = {Danijela Koppers-Lalic and Michael Hackenberg and Irene V. Bijnsdorp and Monique A. J. Eijndhoven and Payman Sadek and Daud Sie and Nicoletta Zini and Jaap M. Middeldorp and Bauke Ylstra and Renee X. Menezes and Thomas Würdinger and Gerrit A. Meijer and D. Michiel Pegtel},
doi = {10.1016/j.celrep.2014.08.027},
issn = {2211-1247},
year = {2014},
date = {2014-09-01},
journal = {Cell Reports},
volume = {8},
number = {6},
pages = {1649–1658},
abstract = {Functional biomolecules, including small noncoding RNAs (ncRNAs), are released and transmitted between mammalian cells via extracellular vesicles (EVs), including endosome-derived exosomes. The small RNA composition in cells differs from exosomes, but underlying mechanisms have not been established. We generated small RNA profiles by RNA sequencing (RNA-seq) from a panel of human B cells and their secreted exosomes. A comprehensive bioinformatics and statistical analysis revealed nonrandomly distributed subsets of microRNA (miRNA) species between B cells and exosomes. Unexpectedly, 3' end adenylated miRNAs are relatively enriched in cells, whereas 3' end uridylated isoforms appear overrepresented in exosomes, as validated in naturally occurring EVs isolated from human urine samples. Collectively, our findings suggest that posttranscriptional modifications, notably 3' end adenylation and uridylation, exert opposing effects that may contribute, at least in part, to direct ncRNA sorting into EVs.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Bernal, Dolores; Trelis, Maria; Montaner, Sergio; Cantalapiedra, Fernando; Galiano, Alicia; Hackenberg, Michael; Marcilla, Antonio
Surface analysis of Dicrocoelium dendriticum. The molecular characterization of exosomes reveals the presence of miRNAs Journal Article
In: Journal of Proteomics, vol. 105, pp. 232–241, 2014, ISSN: 1876-7737.
@article{bernal_surface_2014,
title = {Surface analysis of Dicrocoelium dendriticum. The molecular characterization of exosomes reveals the presence of miRNAs},
author = {Dolores Bernal and Maria Trelis and Sergio Montaner and Fernando Cantalapiedra and Alicia Galiano and Michael Hackenberg and Antonio Marcilla},
doi = {10.1016/j.jprot.2014.02.012},
issn = {1876-7737},
year = {2014},
date = {2014-06-01},
journal = {Journal of Proteomics},
volume = {105},
pages = {232–241},
abstract = {With the aim of characterizing the molecules involved in the interaction of Dicrocoelium dendriticum adults and the host, we have performed proteomic analyses of the external surface of the parasite using the currently available datasets including the transcriptome of the related species Echinostoma caproni. We have identified 182 parasite proteins on the outermost surface of D. dendriticum. The presence of exosome-like vesicles in the ESP of D. dendriticum and their components has also been characterized. Using proteomic approaches, we have characterized 84 proteins in these vesicles. Interestingly, we have detected miRNA in D. dendriticum exosomes, thus representing the first report of miRNA in helminth exosomes.
BIOLOGICAL SIGNIFICANCE: In order to identify potential targets for intervention against parasitic helminths, we have analyzed the surface of the parasitic helminth Dicrocoelium dendriticum. Along with the proteomic analyses of the outermost layer of the parasite, our work describes the molecular characterization of the exosomes of D. dendriticum. Our proteomic data confirm the improvement of protein identification from "non-model organisms" like helminths, when using different search engines against a combination of available databases. In addition, this work represents the first report of miRNAs in parasitic helminth exosomes. These vesicles can pack specific proteins and RNAs providing stability and resistance to RNAse digestion in body fluids, and provide a way to regulate host-parasite interplay. The present data should provide a solid foundation for the development of novel methods to control this non-model organism and related parasites. This article is part of a Special Issue entitled: Proteomics of non-model organisms.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
BIOLOGICAL SIGNIFICANCE: In order to identify potential targets for intervention against parasitic helminths, we have analyzed the surface of the parasitic helminth Dicrocoelium dendriticum. Along with the proteomic analyses of the outermost layer of the parasite, our work describes the molecular characterization of the exosomes of D. dendriticum. Our proteomic data confirm the improvement of protein identification from "non-model organisms" like helminths, when using different search engines against a combination of available databases. In addition, this work represents the first report of miRNAs in parasitic helminth exosomes. These vesicles can pack specific proteins and RNAs providing stability and resistance to RNAse digestion in body fluids, and provide a way to regulate host-parasite interplay. The present data should provide a solid foundation for the development of novel methods to control this non-model organism and related parasites. This article is part of a Special Issue entitled: Proteomics of non-model organisms.
Bali, Kiran Kumar; Hackenberg, Michael; Lubin, Avigail; Kuner, Rohini; Devor, Marshall
Sources of individual variability: miRNAs that predispose to neuropathic pain identified using genome-wide sequencing Journal Article
In: Molecular Pain, vol. 10, pp. 22, 2014, ISSN: 1744-8069.
@article{bali_sources_2014,
title = {Sources of individual variability: miRNAs that predispose to neuropathic pain identified using genome-wide sequencing},
author = {Kiran Kumar Bali and Michael Hackenberg and Avigail Lubin and Rohini Kuner and Marshall Devor},
doi = {10.1186/1744-8069-10-22},
issn = {1744-8069},
year = {2014},
date = {2014-03-01},
journal = {Molecular Pain},
volume = {10},
pages = {22},
abstract = {BACKGROUND: We carried out a genome-wide study, using microRNA sequencing (miRNA-seq), aimed at identifying miRNAs in primary sensory neurons that are associated with neuropathic pain. Such scans usually yield long lists of transcripts regulated by nerve injury, but not necessarily related to pain. To overcome this we tried a novel search strategy: identification of transcripts regulated differentially by nerve injury in rat lines very similar except for a contrasting pain phenotype. Dorsal root ganglia (DRGs) L4 and 5 in the two lines were excised 3 days after spinal nerve ligation surgery (SNL) and small RNAs were extracted and sequenced.
RESULTS: We identified 284 mature miRNA species expressed in rat DRGs, including several not previously reported, and 3340 unique small RNA sequences. Baseline expression of miRNA was nearly identical in the two rat lines, consistent with their shared genetic background. In both lines many miRNAs were nominally up- or down-regulated following SNL, but the change was similar across lines. Only 3 miRNAs that were expressed abundantly (rno-miR-30d-5p, rno-miR-125b-5p) or at moderate levels (rno-miR-379-5p) were differentially regulated. This makes them prime candidates as novel PNS determinants of neuropathic pain. The first two are known miRNA regulators of the expression of Tnf, Bdnf and Stat3, gene products intimately associated with neuropathic pain phenotype. A few non-miRNA, small noncoding RNAs (sncRNAs) were also differentially regulated.
CONCLUSIONS: Despite its genome-wide coverage, our search strategy yielded a remarkably short list of neuropathic pain-related miRNAs. As 2 of the 3 are validated regulators of important pro-nociceptive compounds, it is likely that they contribute to the orchestration of gene expression changes that determine individual variability in pain phenotype. Further research is required to determine whether some of the other known or predicted gene targets of these miRNAs, or of the differentially regulated non-miRNA sncRNAs, also contribute.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We identified 284 mature miRNA species expressed in rat DRGs, including several not previously reported, and 3340 unique small RNA sequences. Baseline expression of miRNA was nearly identical in the two rat lines, consistent with their shared genetic background. In both lines many miRNAs were nominally up- or down-regulated following SNL, but the change was similar across lines. Only 3 miRNAs that were expressed abundantly (rno-miR-30d-5p, rno-miR-125b-5p) or at moderate levels (rno-miR-379-5p) were differentially regulated. This makes them prime candidates as novel PNS determinants of neuropathic pain. The first two are known miRNA regulators of the expression of Tnf, Bdnf and Stat3, gene products intimately associated with neuropathic pain phenotype. A few non-miRNA, small noncoding RNAs (sncRNAs) were also differentially regulated.
CONCLUSIONS: Despite its genome-wide coverage, our search strategy yielded a remarkably short list of neuropathic pain-related miRNAs. As 2 of the 3 are validated regulators of important pro-nociceptive compounds, it is likely that they contribute to the orchestration of gene expression changes that determine individual variability in pain phenotype. Further research is required to determine whether some of the other known or predicted gene targets of these miRNAs, or of the differentially regulated non-miRNA sncRNAs, also contribute.
Jurak, Igor; Hackenberg, Michael; Kim, Ju Youn; Pesola, Jean M.; Everett, Roger D.; Preston, Chris M.; Wilson, Angus C.; Coen, Donald M.
Expression of herpes simplex virus 1 microRNAs in cell culture models of quiescent and latent infection Journal Article
In: Journal of Virology, vol. 88, no. 4, pp. 2337–2339, 2014, ISSN: 1098-5514.
@article{jurak_expression_2014,
title = {Expression of herpes simplex virus 1 microRNAs in cell culture models of quiescent and latent infection},
author = {Igor Jurak and Michael Hackenberg and Ju Youn Kim and Jean M. Pesola and Roger D. Everett and Chris M. Preston and Angus C. Wilson and Donald M. Coen},
doi = {10.1128/JVI.03486-13},
issn = {1098-5514},
year = {2014},
date = {2014-02-01},
journal = {Journal of Virology},
volume = {88},
number = {4},
pages = {2337–2339},
abstract = {To facilitate studies of herpes simplex virus 1 latency, cell culture models of quiescent or latent infection have been developed. Using deep sequencing, we analyzed the expression of viral microRNAs (miRNAs) in two models employing human fibroblasts and one using rat neurons. In all cases, the expression patterns differed from that in productively infected cells, with the rat neuron pattern most closely resembling that found in latently infected human or mouse ganglia in vivo.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Geisen, Stefanie; Barturen, Guillermo; Alganza, Ángel M.; Hackenberg, Michael; Oliver, José L.
NGSmethDB: an updated genome resource for high quality, single-cytosine resolution methylomes Journal Article
In: Nucleic Acids Research, vol. 42, no. Database issue, pp. D53–59, 2014, ISSN: 1362-4962.
@article{geisen_ngsmethdb_2014,
title = {NGSmethDB: an updated genome resource for high quality, single-cytosine resolution methylomes},
author = {Stefanie Geisen and Guillermo Barturen and Ángel M. Alganza and Michael Hackenberg and José L. Oliver},
doi = {10.1093/nar/gkt1202},
issn = {1362-4962},
year = {2014},
date = {2014-01-01},
journal = {Nucleic Acids Research},
volume = {42},
number = {Database issue},
pages = {D53–59},
abstract = {The updated release of 'NGSmethDB' (http://bioinfo2.ugr.es/NGSmethDB) is a repository for single-base whole-genome methylome maps for the best-assembled eukaryotic genomes. Short-read data sets from NGS bisulfite-sequencing projects of cell lines, fresh and pathological tissues are first pre-processed and aligned to the corresponding reference genome, and then the cytosine methylation levels are profiled. One major improvement is the application of a unique bioinformatics protocol to all data sets, thereby assuring the comparability of all values with each other. We implemented stringent quality controls to minimize important error sources, such as sequencing errors, bisulfite failures, clonal reads or single nucleotide variants (SNVs). This leads to reliable and high-quality methylomes, all obtained under uniform settings. Another significant improvement is the detection in parallel of SNVs, which might be crucial for many downstream analyses (e.g. SNVs and differential-methylation relationships). A next-generation methylation browser allows fast and smooth scrolling and zooming, thus speeding data download/upload, at the same time requiring fewer server resources. Several data mining tools allow the comparison/retrieval of methylation levels in different tissues or genome regions. NGSmethDB methylomes are also available as native tracks through a UCSC hub, which allows comparison with a wide range of third-party annotations, in particular phenotype or disease annotations.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2013
Hackenberg, Michael; Shi, Bu-Jun; Gustafson, Perry; Langridge, Peter
In: BMC plant biology, vol. 13, pp. 214, 2013, ISSN: 1471-2229.
@article{hackenberg_characterization_2013,
title = {Characterization of phosphorus-regulated miR399 and miR827 and their isomirs in barley under phosphorus-sufficient and phosphorus-deficient conditions},
author = {Michael Hackenberg and Bu-Jun Shi and Perry Gustafson and Peter Langridge},
doi = {10.1186/1471-2229-13-214},
issn = {1471-2229},
year = {2013},
date = {2013-12-01},
journal = {BMC plant biology},
volume = {13},
pages = {214},
abstract = {BACKGROUND: miR399 and miR827 are both involved in conserved phosphorus (P) deficiency signalling pathways. miR399 targets the PHO2 gene encoding E2 enzyme that negatively regulates phosphate uptake and root-to-shoot allocation, while miR827 targets SPX-domain-containing genes that negatively regulate other P-responsive genes. However, the response of miR399 and miR827 to P conditions in barley has not been investigated.
RESULTS: In this study, we investigated the expression profiles of miR399 and miR827 in barley (Hordeum vulagre L.) under P-deficient and P-sufficient conditions. We identified 10 members of the miR399 family and one miR827 gene in barley, all of which were significantly up-regulated under deficient P. In addition, we found many isomirs of the miR399 family and miR827, most of which were also significantly up-regulated under deficient P. Several isomirs of miR399 members were found to be able to cleave their predicted targets in vivo. Surprisingly, a few small RNAs (sRNAs) derived from the single-stranded loops of the hairpin structures of MIR399b and MIR399e-1 were also found to be able to cleave their predicted targets in vivo. Many antisense sRNAs of miR399 and a few for miR827 were also detected, but they did not seem to be regulated by P. Intriguingly, the lowest expressed member, hvu-miR399k, had four-fold more antisense sRNAs than sense sRNAs, and furthermore under P sufficiency, the antisense sRNAs are more frequent than the sense sRNAs. We identified a potential regulatory network among miR399, its target HvPHO2 and target mimics HvIPS1 and HvIPS2 in barley under P-deficient and P-sufficient conditions.
CONCLUSIONS: Our data provide an important insight into the mechanistic regulation and function of miR399, miR827 and their isomirs in barley under different P conditions.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: In this study, we investigated the expression profiles of miR399 and miR827 in barley (Hordeum vulagre L.) under P-deficient and P-sufficient conditions. We identified 10 members of the miR399 family and one miR827 gene in barley, all of which were significantly up-regulated under deficient P. In addition, we found many isomirs of the miR399 family and miR827, most of which were also significantly up-regulated under deficient P. Several isomirs of miR399 members were found to be able to cleave their predicted targets in vivo. Surprisingly, a few small RNAs (sRNAs) derived from the single-stranded loops of the hairpin structures of MIR399b and MIR399e-1 were also found to be able to cleave their predicted targets in vivo. Many antisense sRNAs of miR399 and a few for miR827 were also detected, but they did not seem to be regulated by P. Intriguingly, the lowest expressed member, hvu-miR399k, had four-fold more antisense sRNAs than sense sRNAs, and furthermore under P sufficiency, the antisense sRNAs are more frequent than the sense sRNAs. We identified a potential regulatory network among miR399, its target HvPHO2 and target mimics HvIPS1 and HvIPS2 in barley under P-deficient and P-sufficient conditions.
CONCLUSIONS: Our data provide an important insight into the mechanistic regulation and function of miR399, miR827 and their isomirs in barley under different P conditions.
Hackenberg, Michael; Huang, Po-Jung; Huang, Chun-Yuan; Shi, Bu-Jun; Gustafson, Perry; Langridge, Peter
In: DNA research: an international journal for rapid publication of reports on genes and genomes, vol. 20, no. 2, pp. 109–125, 2013, ISSN: 1756-1663.
@article{hackenberg_comprehensive_2013,
title = {A comprehensive expression profile of microRNAs and other classes of non-coding small RNAs in barley under phosphorous-deficient and -sufficient conditions},
author = {Michael Hackenberg and Po-Jung Huang and Chun-Yuan Huang and Bu-Jun Shi and Perry Gustafson and Peter Langridge},
doi = {10.1093/dnares/dss037},
issn = {1756-1663},
year = {2013},
date = {2013-04-01},
journal = {DNA research: an international journal for rapid publication of reports on genes and genomes},
volume = {20},
number = {2},
pages = {109–125},
abstract = {Phosphorus (P) is essential for plant growth. MicroRNAs (miRNAs) play a key role in phosphate homeostasis. However, little is known about P effect on miRNA expression in barley (Hordeum vulgare L.). In this study, we used Illumina's next-generation sequencing technology to sequence small RNAs (sRNAs) in barley grown under P-deficient and P-sufficient conditions. We identified 221 conserved miRNAs and 12 novel miRNAs, of which 55 were only present in P-deficient treatment while 32 only existed in P-sufficient treatment. Total 47 miRNAs were significantly differentially expressed between the two P treatments (textbarlog2textbar > 1). We also identified many other classes of sRNAs, including sense and antisense sRNAs, repeat-associated sRNAs, transfer RNA (tRNA)-derived sRNAs and chloroplast-derived sRNAs, and some of which were also significantly differentially expressed between the two P treatments. Of all the sRNAs identified, antisense sRNAs were the most abundant sRNA class in both P treatments. Surprisingly, about one-fourth of sRNAs were derived from the chloroplast genome, and a chloroplast-encoded tRNA-derived sRNA was the most abundant sRNA of all the sRNAs sequenced. Our data provide valuable clues for understanding the properties of sRNAs and new insights into the potential roles of miRNAs and other classes of sRNAs in the control of phosphate homeostasis.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Barturen, Guillermo; Rueda, Antonio; Oliver, José L.; Hackenberg, Michael
MethylExtract: High-Quality methylation maps and SNV calling from whole genome bisulfite sequencing data Journal Article
In: F1000Research, vol. 2, pp. 217, 2013, ISSN: 2046-1402.
@article{barturen_methylextract_2013,
title = {MethylExtract: High-Quality methylation maps and SNV calling from whole genome bisulfite sequencing data},
author = {Guillermo Barturen and Antonio Rueda and José L. Oliver and Michael Hackenberg},
doi = {10.12688/f1000research.2-217.v2},
issn = {2046-1402},
year = {2013},
date = {2013-01-01},
journal = {F1000Research},
volume = {2},
pages = {217},
abstract = {Whole genome methylation profiling at a single cytosine resolution is now feasible due to the advent of high-throughput sequencing techniques together with bisulfite treatment of the DNA. To obtain the methylation value of each individual cytosine, the bisulfite-treated sequence reads are first aligned to a reference genome, and then the profiling of the methylation levels is done from the alignments. A huge effort has been made to quickly and correctly align the reads and many different algorithms and programs to do this have been created. However, the second step is just as crucial and non-trivial, but much less attention has been paid to the final inference of the methylation states. Important error sources do exist, such as sequencing errors, bisulfite failure, clonal reads, and single nucleotide variants. We developed MethylExtract, a user friendly tool to: i) generate high quality, whole genome methylation maps and ii) detect sequence variation within the same sample preparation. The program is implemented into a single script and takes into account all major error sources. MethylExtract detects variation (SNVs - Single Nucleotide Variants) in a similar way to VarScan, a very sensitive method extensively used in SNV and genotype calling based on non-bisulfite-treated reads. The usefulness of MethylExtract is shown by means of extensive benchmarking based on artificial bisulfite-treated reads and a comparison to a recently published method, called Bis-SNP. MethylExtract is able to detect SNVs within High-Throughput Sequencing experiments of bisulfite treated DNA at the same time as it generates high quality methylation maps. This simultaneous detection of DNA methylation and sequence variation is crucial for many downstream analyses, for example when deciphering the impact of SNVs on differential methylation. An exclusive feature of MethylExtract, in comparison with existing software, is the possibility to assess the bisulfite failure in a statistical way. The source code, tutorial and artificial bisulfite datasets are available at http://bioinfo2.ugr.es/MethylExtract/ and http://sourceforge.net/projects/methylextract/, and also permanently accessible from 10.5281/zenodo.7144.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Barturen, Guillermo; Geisen, Stefanie; Dios, Francisco; Hamberg, E. J. Maarten; Hackenberg, Michael; Oliver, José L.
CpGislandEVO: a database and genome browser for comparative evolutionary genomics of CpG islands Journal Article
In: BioMed Research International, vol. 2013, pp. 709042, 2013, ISSN: 2314-6141.
@article{barturen_cpgislandevo_2013,
title = {CpGislandEVO: a database and genome browser for comparative evolutionary genomics of CpG islands},
author = {Guillermo Barturen and Stefanie Geisen and Francisco Dios and E. J. Maarten Hamberg and Michael Hackenberg and José L. Oliver},
doi = {10.1155/2013/709042},
issn = {2314-6141},
year = {2013},
date = {2013-01-01},
journal = {BioMed Research International},
volume = {2013},
pages = {709042},
abstract = {Hypomethylated, CpG-rich DNA segments (CpG islands, CGIs) are epigenome markers involved in key biological processes. Aberrant methylation is implicated in the appearance of several disorders as cancer, immunodeficiency, or centromere instability. Furthermore, methylation differences at promoter regions between human and chimpanzee strongly associate with genes involved in neurological/psychological disorders and cancers. Therefore, the evolutionary comparative analyses of CGIs can provide insights on the functional role of these epigenome markers in both health and disease. Given the lack of specific tools, we developed CpGislandEVO. Briefly, we first compile a database of statistically significant CGIs for the best assembled mammalian genome sequences available to date. Second, by means of a coupled browser front-end, we focus on the CGIs overlapping orthologous genes extracted from OrthoDB, thus ensuring the comparison between CGIs located on truly homologous genome segments. This allows comparing the main compositional features between homologous CGIs. Finally, to facilitate nucleotide comparisons, we lifted genome coordinates between assemblies from different species, which enables the analysis of sequence divergence by direct count of nucleotide substitutions and indels occurring between homologous CGIs. The resulting CpGislandEVO database, linking together CGIs and single-cytosine DNA methylation data from several mammalian species, is freely available at our website.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2012
Ferreiro, María José; Rodríguez-Ezpeleta, Naiara; Pérez, Coralia; Hackenberg, Michael; Aransay, Ana María; Barrio, Rosa; Cantera, Rafael
Whole transcriptome analysis of a reversible neurodegenerative process in Drosophila reveals potential neuroprotective genes Journal Article
In: BMC genomics, vol. 13, pp. 483, 2012, ISSN: 1471-2164.
@article{ferreiro_whole_2012,
title = {Whole transcriptome analysis of a reversible neurodegenerative process in Drosophila reveals potential neuroprotective genes},
author = {María José Ferreiro and Naiara Rodríguez-Ezpeleta and Coralia Pérez and Michael Hackenberg and Ana María Aransay and Rosa Barrio and Rafael Cantera},
doi = {10.1186/1471-2164-13-483},
issn = {1471-2164},
year = {2012},
date = {2012-09-01},
journal = {BMC genomics},
volume = {13},
pages = {483},
abstract = {BACKGROUND: Neurodegenerative diseases are progressive and irreversible and they can be initiated by mutations in specific genes. Spalt-like genes (Sall) encode transcription factors expressed in the central nervous system. In humans, SALL mutations are associated with hereditary syndromes characterized by mental retardation, sensorineural deafness and motoneuron problems, among others. Drosophila sall mutants exhibit severe neurodegeneration of the central nervous system at embryonic stage 16, which surprisingly reverts later in development at embryonic stage 17, suggesting a potential to recover from neurodegeneration. We hypothesize that this recovery is mediated by a reorganization of the transcriptome counteracting SALL lost. To identify genes associated to neurodegeneration and neuroprotection, we used mRNA-Seq to compare the transcriptome of Drosophila sall mutant and wild type embryos from neurodegeneration and reversal stages.
RESULTS: Neurodegeneration stage is associated with transcriptional changes in 220 genes, of which only 5% were already described as relevant for neurodegeneration. Genes related to the groups of Redox, Lifespan/Aging and Mitochondrial diseases are significantly represented at this stage. By contrast, neurodegeneration reversal stage is associated with significant changes in 480 genes, including 424 not previously associated with neuroprotection. Immune response and Salt stress are the most represented groups at this stage.
CONCLUSIONS: We identify new genes associated to neurodegeneration and neuroprotection by using an mRNA-Seq approach. The strong homology between Drosophila and human genes raises the possibility to unveil novel genes involved in neurodegeneration and neuroprotection also in humans.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: Neurodegeneration stage is associated with transcriptional changes in 220 genes, of which only 5% were already described as relevant for neurodegeneration. Genes related to the groups of Redox, Lifespan/Aging and Mitochondrial diseases are significantly represented at this stage. By contrast, neurodegeneration reversal stage is associated with significant changes in 480 genes, including 424 not previously associated with neuroprotection. Immune response and Salt stress are the most represented groups at this stage.
CONCLUSIONS: We identify new genes associated to neurodegeneration and neuroprotection by using an mRNA-Seq approach. The strong homology between Drosophila and human genes raises the possibility to unveil novel genes involved in neurodegeneration and neuroprotection also in humans.
Hackenberg, Michael; Rueda, Antonio; Carpena, Pedro; Bernaola-Galván, Pedro; Barturen, Guillermo; Oliver, José L.
Clustering of DNA words and biological function: a proof of principle Journal Article
In: Journal of Theoretical Biology, vol. 297, pp. 127–136, 2012, ISSN: 1095-8541.
@article{hackenberg_clustering_2012,
title = {Clustering of DNA words and biological function: a proof of principle},
author = {Michael Hackenberg and Antonio Rueda and Pedro Carpena and Pedro Bernaola-Galván and Guillermo Barturen and José L. Oliver},
doi = {10.1016/j.jtbi.2011.12.024},
issn = {1095-8541},
year = {2012},
date = {2012-03-01},
journal = {Journal of Theoretical Biology},
volume = {297},
pages = {127–136},
abstract = {Relevant words in literary texts (key words) are known to be clustered, while common words are randomly distributed. Given the clustered distribution of many functional genome elements, we hypothesize that the biological text per excellence, the DNA sequence, might behave in the same way: k-length words (k-mers) with a clear function may be spatially clustered along the one-dimensional chromosome sequence, while less-important, non-functional words may be randomly distributed. To explore this linguistic analogy, we calculate a clustering coefficient for each k-mer (k=2-9bp) in human and mouse chromosome sequences, then checking if clustered words are enriched in the functional part of the genome. First, we found a positive general trend relating clustering level and word enrichment within exons and Transcription Factor Binding Sites (TFBSs), while a much weaker relation exists for repeats, and no relation at all exists for introns. Second, we found that 38.45% of the 200 top-clustered 8-mers, but only 7.70% of the non-clustered words, are represented in known motif databases. Third, enrichment/depletion experiments show that highly clustered words are significantly enriched in exons and TFBSs, while they are depleted in introns and repetitive DNA. Considering exons and TFBSs together, 1417 (or 72.26%) in human and 1385 (or 72.97%) in mouse of the top-clustered 8-mers showed a statistically significant association to either exons or TFBSs, thus strongly supporting the link between word clustering and biological function. Lastly, we identified a subset of clustered, diagnostic words that are enriched in exons but depleted in introns, and therefore might help to discriminate between these two gene regions. The clustering of DNA words thus appears as a novel principle to detect functionality in genome sequences. As evolutionary conservation is not a prerequisite, the proof of principle described here may open new ways to detect species-specific functional DNA sequences and the improvement of gene and promoter predictions, thus contributing to the quest for function in the genome.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Meng, Fanxue; Hackenberg, Michael; Li, Zhiguang; Yan, Jian; Chen, Tao
Discovery of novel microRNAs in rat kidney using next generation sequencing and microarray validation Journal Article
In: PloS One, vol. 7, no. 3, pp. e34394, 2012, ISSN: 1932-6203.
@article{meng_discovery_2012,
title = {Discovery of novel microRNAs in rat kidney using next generation sequencing and microarray validation},
author = {Fanxue Meng and Michael Hackenberg and Zhiguang Li and Jian Yan and Tao Chen},
doi = {10.1371/journal.pone.0034394},
issn = {1932-6203},
year = {2012},
date = {2012-01-01},
journal = {PloS One},
volume = {7},
number = {3},
pages = {e34394},
abstract = {MicroRNAs (miRNAs) are small non-coding RNAs that regulate a variety of biological processes. The latest version of the miRBase database (Release 18) includes 1,157 mouse and 680 rat mature miRNAs. Only one new rat mature miRNA was added to the rat miRNA database from version 16 to version 18 of miRBase, suggesting that many rat miRNAs remain to be discovered. Given the importance of rat as a model organism, discovery of the completed set of rat miRNAs is necessary for understanding rat miRNA regulation. In this study, next generation sequencing (NGS), microarray analysis and bioinformatics technologies were applied to discover novel miRNAs in rat kidneys. MiRanalyzer was utilized to analyze the sequences of the small RNAs generated from NGS analysis of rat kidney samples. Hundreds of novel miRNA candidates were examined according to the mappings of their reads to the rat genome, presence of sequences that can form a miRNA hairpin structure around the mapped locations, Dicer cleavage patterns, and the levels of their expression determined by both NGS and microarray analyses. Nine novel rat hairpin precursor miRNAs (pre-miRNA) were discovered with high confidence. Five of the novel pre-miRNAs are also reported in other species while four of them are rat specific. In summary, 9 novel pre-miRNAs (14 novel mature miRNAs) were identified via combination of NGS, microarray and bioinformatics high-throughput technologies.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Hackenberg, Michael; Shi, Bu-Jun; Gustafson, Perry; Langridge, Peter
A transgenic transcription factor (TaDREB3) in barley affects the expression of microRNAs and other small non-coding RNAs Journal Article
In: PloS One, vol. 7, no. 8, pp. e42030, 2012, ISSN: 1932-6203.
@article{hackenberg_transgenic_2012,
title = {A transgenic transcription factor (TaDREB3) in barley affects the expression of microRNAs and other small non-coding RNAs},
author = {Michael Hackenberg and Bu-Jun Shi and Perry Gustafson and Peter Langridge},
doi = {10.1371/journal.pone.0042030},
issn = {1932-6203},
year = {2012},
date = {2012-01-01},
journal = {PloS One},
volume = {7},
number = {8},
pages = {e42030},
abstract = {Transcription factors (TFs), microRNAs (miRNAs), small interfering RNAs (siRNAs) and other functional non-coding small RNAs (sRNAs) are important gene regulators. Comparison of sRNA expression profiles between transgenic barley over-expressing a drought tolerant TF (TaDREB3) and non-transgenic control barley revealed many group-specific sRNAs. In addition, 42% of the shared sRNAs were differentially expressed between the two groups (textbarlog(2)textbar >1). Furthermore, TaDREB3-derived sRNAs were only detected in transgenic barley despite the existence of homologous genes in non-transgenic barley. These results demonstrate that the TF strongly affects the expression of sRNAs and siRNAs could in turn affect the TF stability. The TF also affects size distribution and abundance of sRNAs including miRNAs. About half of the sRNAs in each group were derived from chloroplast. A sRNA derived from tRNA-His(GUG) encoded by the chloroplast genome is the most abundant sRNA, accounting for 42.2% of the total sRNAs in transgenic barley and 28.9% in non-transgenic barley. This sRNA, which targets a gene (TC245676) involved in biological processes, was only present in barley leaves but not roots. 124 and 136 miRNAs were detected in transgenic and non-transgenic barley, respectively. miR156 was the most abundant miRNA and up-regulated in transgenic barley, while miR168 was the most abundant miRNA and up-regulated in non-transgenic barley. Eight out of 20 predicted novel miRNAs were differentially expressed between the two groups. All the predicted novel miRNA targets were validated using a degradome library. Our data provide an insight into the effect of TF on the expression of sRNAs in barley.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2011
Hackenberg, Michael; Rodríguez-Ezpeleta, Naiara; Aransay, Ana M.
miRanalyzer: an update on the detection and analysis of microRNAs in high-throughput sequencing experiments Journal Article
In: Nucleic Acids Research, vol. 39, no. Web Server issue, pp. W132–138, 2011, ISSN: 1362-4962.
@article{hackenberg_miranalyzer_2011,
title = {miRanalyzer: an update on the detection and analysis of microRNAs in high-throughput sequencing experiments},
author = {Michael Hackenberg and Naiara Rodríguez-Ezpeleta and Ana M. Aransay},
doi = {10.1093/nar/gkr247},
issn = {1362-4962},
year = {2011},
date = {2011-07-01},
journal = {Nucleic Acids Research},
volume = {39},
number = {Web Server issue},
pages = {W132–138},
abstract = {We present a new version of miRanalyzer, a web server and stand-alone tool for the detection of known and prediction of new microRNAs in high-throughput sequencing experiments. The new version has been notably improved regarding speed, scope and available features. Alignments are now based on the ultrafast short-read aligner Bowtie (granting also colour space support, allowing mismatches and improving speed) and 31 genomes, including 6 plant genomes, can now be analysed (previous version contained only 7). Differences between plant and animal microRNAs have been taken into account for the prediction models and differential expression of both, known and predicted microRNAs, between two conditions can be calculated. Additionally, consensus sequences of predicted mature and precursor microRNAs can be obtained from multiple samples, which increases the reliability of the predicted microRNAs. Finally, a stand-alone version of the miRanalyzer that is based on a local and easily customized database is also available; this allows the user to have more control on certain parameters as well as to use specific data such as unpublished assemblies or other libraries that are not available in the web server. miRanalyzer is available at http://bioinfo2.ugr.es/miRanalyzer/miRanalyzer.php.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Hackenberg, Michael; Barturen, Guillermo; Oliver, José L.
NGSmethDB: a database for next-generation sequencing single-cytosine-resolution DNA methylation data Journal Article
In: Nucleic Acids Research, vol. 39, no. Database issue, pp. D75–79, 2011, ISSN: 1362-4962.
@article{hackenberg_ngsmethdb_2011,
title = {NGSmethDB: a database for next-generation sequencing single-cytosine-resolution DNA methylation data},
author = {Michael Hackenberg and Guillermo Barturen and José L. Oliver},
doi = {10.1093/nar/gkq942},
issn = {1362-4962},
year = {2011},
date = {2011-01-01},
journal = {Nucleic Acids Research},
volume = {39},
number = {Database issue},
pages = {D75–79},
abstract = {Next-generation sequencing (NGS) together with bisulphite conversion allows the generation of whole genome methylation maps at single-cytosine resolution. This allows studying the absence of methylation in a particular genome region over a range of tissues, the differential tissue methylation or the changes occurring along pathological conditions. However, no database exists fully addressing such requirements. We propose here NGSmethDB (http://bioinfo2.ugr.es/NGSmethDB/gbrowse/) for the storage and retrieval of methylation data derived from NGS. Two cytosine methylation contexts (CpG and CAG/CTG) are considered. Through a browser interface coupled to a MySQL backend and several data mining tools, the user can search for methylation states in a set of tissues, retrieve methylation values for a set of tissues in a given chromosomal region, or display the methylation of promoters among different tissues. NGSmethDB is currently populated with human, mouse and Arabidopsis data, but other methylomes will be incorporated through an automatic pipeline as soon as new data become available. Dump downloads for three coverage levels (1, 5 or 10 reads) are available. NGSmethDB will be useful for experimental researchers, as well as for bioinformaticians, who might use the data as input for further research.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Hackenberg, Michael; Carpena, Pedro; Bernaola-Galván, Pedro; Barturen, Guillermo; Alganza, Angel M.; Oliver, José L.
WordCluster: detecting clusters of DNA words and genomic elements Journal Article
In: Algorithms for molecular biology: AMB, vol. 6, pp. 2, 2011, ISSN: 1748-7188.
@article{hackenberg_wordcluster_2011,
title = {WordCluster: detecting clusters of DNA words and genomic elements},
author = {Michael Hackenberg and Pedro Carpena and Pedro Bernaola-Galván and Guillermo Barturen and Angel M. Alganza and José L. Oliver},
doi = {10.1186/1748-7188-6-2},
issn = {1748-7188},
year = {2011},
date = {2011-01-01},
journal = {Algorithms for molecular biology: AMB},
volume = {6},
pages = {2},
abstract = {BACKGROUND: Many k-mers (or DNA words) and genomic elements are known to be spatially clustered in the genome. Well established examples are the genes, TFBSs, CpG dinucleotides, microRNA genes and ultra-conserved non-coding regions. Currently, no algorithm exists to find these clusters in a statistically comprehensible way. The detection of clustering often relies on densities and sliding-window approaches or arbitrarily chosen distance thresholds.
RESULTS: We introduce here an algorithm to detect clusters of DNA words (k-mers), or any other genomic element, based on the distance between consecutive copies and an assigned statistical significance. We implemented the method into a web server connected to a MySQL backend, which also determines the co-localization with gene annotations. We demonstrate the usefulness of this approach by detecting the clusters of CAG/CTG (cytosine contexts that can be methylated in undifferentiated cells), showing that the degree of methylation vary drastically between inside and outside of the clusters. As another example, we used WordCluster to search for statistically significant clusters of olfactory receptor (OR) genes in the human genome.
CONCLUSIONS: WordCluster seems to predict biological meaningful clusters of DNA words (k-mers) and genomic entities. The implementation of the method into a web server is available at http://bioinfo2.ugr.es/wordCluster/wordCluster.php including additional features like the detection of co-localization with gene regions or the annotation enrichment tool for functional analysis of overlapped genes.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We introduce here an algorithm to detect clusters of DNA words (k-mers), or any other genomic element, based on the distance between consecutive copies and an assigned statistical significance. We implemented the method into a web server connected to a MySQL backend, which also determines the co-localization with gene annotations. We demonstrate the usefulness of this approach by detecting the clusters of CAG/CTG (cytosine contexts that can be methylated in undifferentiated cells), showing that the degree of methylation vary drastically between inside and outside of the clusters. As another example, we used WordCluster to search for statistically significant clusters of olfactory receptor (OR) genes in the human genome.
CONCLUSIONS: WordCluster seems to predict biological meaningful clusters of DNA words (k-mers) and genomic entities. The implementation of the method into a web server is available at http://bioinfo2.ugr.es/wordCluster/wordCluster.php including additional features like the detection of co-localization with gene regions or the annotation enrichment tool for functional analysis of overlapped genes.
2010
Hackenberg, Michael; Barturen, Guillermo; Carpena, Pedro; Luque-Escamilla, Pedro L.; Previti, Christopher; Oliver, José L.
Prediction of CpG-island function: CpG clustering vs. sliding-window methods Journal Article
In: BMC genomics, vol. 11, pp. 327, 2010, ISSN: 1471-2164.
@article{hackenberg_prediction_2010,
title = {Prediction of CpG-island function: CpG clustering vs. sliding-window methods},
author = {Michael Hackenberg and Guillermo Barturen and Pedro Carpena and Pedro L. Luque-Escamilla and Christopher Previti and José L. Oliver},
doi = {10.1186/1471-2164-11-327},
issn = {1471-2164},
year = {2010},
date = {2010-05-01},
journal = {BMC genomics},
volume = {11},
pages = {327},
abstract = {BACKGROUND: Unmethylated stretches of CpG dinucleotides (CpG islands) are an outstanding property of mammal genomes. Conventionally, these regions are detected by sliding window approaches using %G + C, CpG observed/expected ratio and length thresholds as main parameters. Recently, clustering methods directly detect clusters of CpG dinucleotides as a statistical property of the genome sequence.
RESULTS: We compare sliding-window to clustering (i.e. CpGcluster) predictions by applying new ways to detect putative functionality of CpG islands. Analyzing the co-localization with several genomic regions as a function of window size vs. statistical significance (p-value), CpGcluster shows a higher overlap with promoter regions and highly conserved elements, at the same time showing less overlap with Alu retrotransposons. The major difference in the prediction was found for short islands (CpG islets), often exclusively predicted by CpGcluster. Many of these islets seem to be functional, as they are unmethylated, highly conserved and/or located within the promoter region. Finally, we show that window-based islands can spuriously overlap several, differentially regulated promoters as well as different methylation domains, which might indicate a wrong merge of several CpG islands into a single, very long island. The shorter CpGcluster islands seem to be much more specific when concerning the overlap with alternative transcription start sites or the detection of homogenous methylation domains.
CONCLUSIONS: The main difference between sliding-window approaches and clustering methods is the length of the predicted islands. Short islands, often differentially methylated, are almost exclusively predicted by CpGcluster. This suggests that CpGcluster may be the algorithm of choice to explore the function of these short, but putatively functional CpG islands.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We compare sliding-window to clustering (i.e. CpGcluster) predictions by applying new ways to detect putative functionality of CpG islands. Analyzing the co-localization with several genomic regions as a function of window size vs. statistical significance (p-value), CpGcluster shows a higher overlap with promoter regions and highly conserved elements, at the same time showing less overlap with Alu retrotransposons. The major difference in the prediction was found for short islands (CpG islets), often exclusively predicted by CpGcluster. Many of these islets seem to be functional, as they are unmethylated, highly conserved and/or located within the promoter region. Finally, we show that window-based islands can spuriously overlap several, differentially regulated promoters as well as different methylation domains, which might indicate a wrong merge of several CpG islands into a single, very long island. The shorter CpGcluster islands seem to be much more specific when concerning the overlap with alternative transcription start sites or the detection of homogenous methylation domains.
CONCLUSIONS: The main difference between sliding-window approaches and clustering methods is the length of the predicted islands. Short islands, often differentially methylated, are almost exclusively predicted by CpGcluster. This suggests that CpGcluster may be the algorithm of choice to explore the function of these short, but putatively functional CpG islands.
Sturm, Martin; Hackenberg, Michael; Langenberger, David; Frishman, Dmitrij
TargetSpy: a supervised machine learning approach for microRNA target prediction Journal Article
In: BMC bioinformatics, vol. 11, pp. 292, 2010, ISSN: 1471-2105.
@article{sturm_targetspy_2010,
title = {TargetSpy: a supervised machine learning approach for microRNA target prediction},
author = {Martin Sturm and Michael Hackenberg and David Langenberger and Dmitrij Frishman},
doi = {10.1186/1471-2105-11-292},
issn = {1471-2105},
year = {2010},
date = {2010-05-01},
journal = {BMC bioinformatics},
volume = {11},
pages = {292},
abstract = {BACKGROUND: Virtually all currently available microRNA target site prediction algorithms require the presence of a (conserved) seed match to the 5' end of the microRNA. Recently however, it has been shown that this requirement might be too stringent, leading to a substantial number of missed target sites.
RESULTS: We developed TargetSpy, a novel computational approach for predicting target sites regardless of the presence of a seed match. It is based on machine learning and automatic feature selection using a wide spectrum of compositional, structural, and base pairing features covering current biological knowledge. Our model does not rely on evolutionary conservation, which allows the detection of species-specific interactions and makes TargetSpy suitable for analyzing unconserved genomic sequences.In order to allow for an unbiased comparison of TargetSpy to other methods, we classified all algorithms into three groups: I) no seed match requirement, II) seed match requirement, and III) conserved seed match requirement. TargetSpy predictions for classes II and III are generated by appropriate postfiltering. On a human dataset revealing fold-change in protein production for five selected microRNAs our method shows superior performance in all classes. In Drosophila melanogaster not only our class II and III predictions are on par with other algorithms, but notably the class I (no-seed) predictions are just marginally less accurate. We estimate that TargetSpy predicts between 26 and 112 functional target sites without a seed match per microRNA that are missed by all other currently available algorithms.
CONCLUSION: Only a few algorithms can predict target sites without demanding a seed match and TargetSpy demonstrates a substantial improvement in prediction accuracy in that class. Furthermore, when conservation and the presence of a seed match are required, the performance is comparable with state-of-the-art algorithms. TargetSpy was trained on mouse and performs well in human and drosophila, suggesting that it may be applicable to a broad range of species. Moreover, we have demonstrated that the application of machine learning techniques in combination with upcoming deep sequencing data results in a powerful microRNA target site prediction tool http://www.targetspy.org.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We developed TargetSpy, a novel computational approach for predicting target sites regardless of the presence of a seed match. It is based on machine learning and automatic feature selection using a wide spectrum of compositional, structural, and base pairing features covering current biological knowledge. Our model does not rely on evolutionary conservation, which allows the detection of species-specific interactions and makes TargetSpy suitable for analyzing unconserved genomic sequences.In order to allow for an unbiased comparison of TargetSpy to other methods, we classified all algorithms into three groups: I) no seed match requirement, II) seed match requirement, and III) conserved seed match requirement. TargetSpy predictions for classes II and III are generated by appropriate postfiltering. On a human dataset revealing fold-change in protein production for five selected microRNAs our method shows superior performance in all classes. In Drosophila melanogaster not only our class II and III predictions are on par with other algorithms, but notably the class I (no-seed) predictions are just marginally less accurate. We estimate that TargetSpy predicts between 26 and 112 functional target sites without a seed match per microRNA that are missed by all other currently available algorithms.
CONCLUSION: Only a few algorithms can predict target sites without demanding a seed match and TargetSpy demonstrates a substantial improvement in prediction accuracy in that class. Furthermore, when conservation and the presence of a seed match are required, the performance is comparable with state-of-the-art algorithms. TargetSpy was trained on mouse and performs well in human and drosophila, suggesting that it may be applicable to a broad range of species. Moreover, we have demonstrated that the application of machine learning techniques in combination with upcoming deep sequencing data results in a powerful microRNA target site prediction tool http://www.targetspy.org.
Hackenberg, Michael; Matthiesen, Rune
Algorithms and methods for correlating experimental results with annotation databases Journal Article
In: Methods in Molecular Biology, vol. 593, pp. 315–340, 2010, ISSN: 1940-6029.
@article{hackenberg_algorithms_2010,
title = {Algorithms and methods for correlating experimental results with annotation databases},
author = {Michael Hackenberg and Rune Matthiesen},
doi = {10.1007/978-1-60327-194-3_15},
issn = {1940-6029},
year = {2010},
date = {2010-01-01},
journal = {Methods in Molecular Biology},
volume = {593},
pages = {315–340},
address = {Clifton, N.J.},
abstract = {An important procedure in biomedical research is the detection of genes that are differentially expressed under pathologic conditions. These genes, or at least a subset of them, are key biomarkers and are thought to be important to describe and understand the analyzed biological system (the pathology) at a molecular level. To obtain this understanding, it is indispensable to link those genes to biological knowledge stored in databases. Ontological analysis is nowadays a standard procedure to analyze large gene lists. By detecting enriched and depleted gene properties and functions, important insights on the biological system can be obtained. In this chapter, we will give a brief survey of the general layout of the methods used in an ontological analysis and of the most important tools that have been developed.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2009
Hackenberg, Michael; Sturm, Martin; Langenberger, David; Falcón-Pérez, Juan Manuel; Aransay, Ana M.
miRanalyzer: a microRNA detection and analysis tool for next-generation sequencing experiments Journal Article
In: Nucleic Acids Research, vol. 37, no. Web Server issue, pp. W68–76, 2009, ISSN: 1362-4962.
@article{hackenberg_miranalyzer_2009,
title = {miRanalyzer: a microRNA detection and analysis tool for next-generation sequencing experiments},
author = {Michael Hackenberg and Martin Sturm and David Langenberger and Juan Manuel Falcón-Pérez and Ana M. Aransay},
doi = {10.1093/nar/gkp347},
issn = {1362-4962},
year = {2009},
date = {2009-07-01},
journal = {Nucleic Acids Research},
volume = {37},
number = {Web Server issue},
pages = {W68–76},
abstract = {Next-generation sequencing allows now the sequencing of small RNA molecules and the estimation of their expression levels. Consequently, there will be a high demand of bioinformatics tools to cope with the several gigabytes of sequence data generated in each single deep-sequencing experiment. Given this scene, we developed miRanalyzer, a web server tool for the analysis of deep-sequencing experiments for small RNAs. The web server tool requires a simple input file containing a list of unique reads and its copy numbers (expression levels). Using these data, miRanalyzer (i) detects all known microRNA sequences annotated in miRBase, (ii) finds all perfect matches against other libraries of transcribed sequences and (iii) predicts new microRNAs. The prediction of new microRNAs is an especially important point as there are many species with very few known microRNAs. Therefore, we implemented a highly accurate machine learning algorithm for the prediction of new microRNAs that reaches AUC values of 97.9% and recall values of up to 75% on unseen data. The web tool summarizes all the described steps in a single output page, which provides a comprehensive overview of the analysis, adding links to more detailed output pages for each analysis module. miRanalyzer is available at http://web.bioinformatics.cicbiogune.es/microRNA/.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Hackenberg, Michael; Lasso, Gorka; Matthiesen, Rune
ContDist: a tool for the analysis of quantitative gene and promoter properties Journal Article
In: BMC bioinformatics, vol. 10, pp. 7, 2009, ISSN: 1471-2105.
@article{hackenberg_contdist_2009,
title = {ContDist: a tool for the analysis of quantitative gene and promoter properties},
author = {Michael Hackenberg and Gorka Lasso and Rune Matthiesen},
doi = {10.1186/1471-2105-10-7},
issn = {1471-2105},
year = {2009},
date = {2009-01-01},
journal = {BMC bioinformatics},
volume = {10},
pages = {7},
abstract = {BACKGROUND: The understanding of how promoter regions regulate gene expression is complicated and far from being fully understood. It is known that histones' regulation of DNA compactness, DNA methylation, transcription factor binding sites and CpG islands play a role in the transcriptional regulation of a gene. Many high-throughput techniques exist nowadays which permit the detection of epigenetic marks and regulatory elements in the promoter regions of thousands of genes. However, so far the subsequent analysis of such experiments (e.g. the resulting gene lists) have been hampered by the fact that currently no tool exists for a detailed analysis of the promoter regions.
RESULTS: We present ContDist, a tool to statistically analyze quantitative gene and promoter properties. The software includes approximately 200 quantitative features of gene and promoter regions for 7 commonly studied species. In contrast to "traditionally" ontological analysis which only works on qualitative data, all the features in the underlying annotation database are quantitative gene and promoter properties.Utilizing the strong focus on the promoter region of this tool, we show its usefulness in two case studies; the first on differentially methylated promoters and the second on the fundamental differences between housekeeping and tissue specific genes. The two case studies allow both the confirmation of recent findings as well as revealing previously unreported biological relations.
CONCLUSION: ContDist is a new tool with two important properties: 1) it has a strong focus on the promoter region which is usually disregarded by virtually all ontology tools and 2) it uses quantitative (continuously distributed) features of the genes and its promoter regions which are not available in any other tool. ContDist is available from http://web.bioinformatics.cicbiogune.es/CD/ContDistribution.php.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We present ContDist, a tool to statistically analyze quantitative gene and promoter properties. The software includes approximately 200 quantitative features of gene and promoter regions for 7 commonly studied species. In contrast to "traditionally" ontological analysis which only works on qualitative data, all the features in the underlying annotation database are quantitative gene and promoter properties.Utilizing the strong focus on the promoter region of this tool, we show its usefulness in two case studies; the first on differentially methylated promoters and the second on the fundamental differences between housekeeping and tissue specific genes. The two case studies allow both the confirmation of recent findings as well as revealing previously unreported biological relations.
CONCLUSION: ContDist is a new tool with two important properties: 1) it has a strong focus on the promoter region which is usually disregarded by virtually all ontology tools and 2) it uses quantitative (continuously distributed) features of the genes and its promoter regions which are not available in any other tool. ContDist is available from http://web.bioinformatics.cicbiogune.es/CD/ContDistribution.php.
2008
Hackenberg, Michael; Matthiesen, Rune
Annotation-Modules: a tool for finding significant combinations of multisource annotations for gene lists Journal Article
In: Bioinformatics, vol. 24, no. 11, pp. 1386–1393, 2008, ISSN: 1367-4811.
@article{hackenberg_annotation-modules_2008,
title = {Annotation-Modules: a tool for finding significant combinations of multisource annotations for gene lists},
author = {Michael Hackenberg and Rune Matthiesen},
doi = {10.1093/bioinformatics/btn178},
issn = {1367-4811},
year = {2008},
date = {2008-06-01},
journal = {Bioinformatics},
volume = {24},
number = {11},
pages = {1386–1393},
address = {Oxford, England},
abstract = {MOTIVATION: The ontological analysis of the gene lists obtained from DNA microarray experiments constitutes an important step in understanding the underlying biology of the analyzed system. Over the last years, many other high-throughput techniques emerged, covering now basically all 'omics' fields. However, for some of these techniques the generally used functional ontologies might not be sufficient to describe the biological system represented by the derived gene lists. For a more complete and correct interpretation of these experiments, it is important to extend substantially the number of annotations, adapting the ontological analysis to the new emerging techniques.
RESULTS: We developed Annotation-Modules, which offers an improvement over the current tools in two critical aspects. First, the underlying annotation database implements features from many different fields like gene regulation and expression, sequence properties, evolution and conservation, genomic localization and functional categories-resulting in about 60 different annotation features. Second, it examines not only single annotations but also all the combinations, which is important to gain insight into the interplay of different mechanisms in the analyzed biological system.
AVAILABILITY: http://web.bioinformatics.cicbiogune.es/AM/AnnotationModules.php},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: We developed Annotation-Modules, which offers an improvement over the current tools in two critical aspects. First, the underlying annotation database implements features from many different fields like gene regulation and expression, sequence properties, evolution and conservation, genomic localization and functional categories-resulting in about 60 different annotation features. Second, it examines not only single annotations but also all the combinations, which is important to gain insight into the interplay of different mechanisms in the analyzed biological system.
AVAILABILITY: http://web.bioinformatics.cicbiogune.es/AM/AnnotationModules.php
Oliver, José L.; Bernaola-Galván, Pedro; Hackenberg, Michael; Carpena, Pedro
Phylogenetic distribution of large-scale genome patchiness Journal Article
In: BMC evolutionary biology, vol. 8, pp. 107, 2008, ISSN: 1471-2148.
@article{oliver_phylogenetic_2008,
title = {Phylogenetic distribution of large-scale genome patchiness},
author = {José L. Oliver and Pedro Bernaola-Galván and Michael Hackenberg and Pedro Carpena},
doi = {10.1186/1471-2148-8-107},
issn = {1471-2148},
year = {2008},
date = {2008-04-01},
journal = {BMC evolutionary biology},
volume = {8},
pages = {107},
abstract = {BACKGROUND: The phylogenetic distribution of large-scale genome structure (i.e. mosaic compositional patchiness) has been explored mainly by analytical ultracentrifugation of bulk DNA. However, with the availability of large, good-quality chromosome sequences, and the recently developed computational methods to directly analyze patchiness on the genome sequence, an evolutionary comparative analysis can be carried out at the sequence level.
RESULTS: The local variations in the scaling exponent of the Detrended Fluctuation Analysis are used here to analyze large-scale genome structure and directly uncover the characteristic scales present in genome sequences. Furthermore, through shuffling experiments of selected genome regions, computationally-identified, isochore-like regions were identified as the biological source for the uncovered large-scale genome structure. The phylogenetic distribution of short- and large-scale patchiness was determined in the best-sequenced genome assemblies from eleven eukaryotic genomes: mammals (Homo sapiens, Pan troglodytes, Mus musculus, Rattus norvegicus, and Canis familiaris), birds (Gallus gallus), fishes (Danio rerio), invertebrates (Drosophila melanogaster and Caenorhabditis elegans), plants (Arabidopsis thaliana) and yeasts (Saccharomyces cerevisiae). We found large-scale patchiness of genome structure, associated with in silico determined, isochore-like regions, throughout this wide phylogenetic range.
CONCLUSION: Large-scale genome structure is detected by directly analyzing DNA sequences in a wide range of eukaryotic chromosome sequences, from human to yeast. In all these genomes, large-scale patchiness can be associated with the isochore-like regions, as directly detected in silico at the sequence level.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: The local variations in the scaling exponent of the Detrended Fluctuation Analysis are used here to analyze large-scale genome structure and directly uncover the characteristic scales present in genome sequences. Furthermore, through shuffling experiments of selected genome regions, computationally-identified, isochore-like regions were identified as the biological source for the uncovered large-scale genome structure. The phylogenetic distribution of short- and large-scale patchiness was determined in the best-sequenced genome assemblies from eleven eukaryotic genomes: mammals (Homo sapiens, Pan troglodytes, Mus musculus, Rattus norvegicus, and Canis familiaris), birds (Gallus gallus), fishes (Danio rerio), invertebrates (Drosophila melanogaster and Caenorhabditis elegans), plants (Arabidopsis thaliana) and yeasts (Saccharomyces cerevisiae). We found large-scale patchiness of genome structure, associated with in silico determined, isochore-like regions, throughout this wide phylogenetic range.
CONCLUSION: Large-scale genome structure is detected by directly analyzing DNA sequences in a wide range of eukaryotic chromosome sequences, from human to yeast. In all these genomes, large-scale patchiness can be associated with the isochore-like regions, as directly detected in silico at the sequence level.
2006
Hackenberg, Michael; Previti, Christopher; Luque-Escamilla, Pedro Luis; Carpena, Pedro; Martínez-Aroza, José; Oliver, José L.
CpGcluster: a distance-based algorithm for CpG-island detection Journal Article
In: BMC bioinformatics, vol. 7, pp. 446, 2006, ISSN: 1471-2105.
@article{hackenberg_cpgcluster_2006,
title = {CpGcluster: a distance-based algorithm for CpG-island detection},
author = {Michael Hackenberg and Christopher Previti and Pedro Luis Luque-Escamilla and Pedro Carpena and José Martínez-Aroza and José L. Oliver},
doi = {10.1186/1471-2105-7-446},
issn = {1471-2105},
year = {2006},
date = {2006-10-01},
journal = {BMC bioinformatics},
volume = {7},
pages = {446},
abstract = {BACKGROUND: Despite their involvement in the regulation of gene expression and their importance as genomic markers for promoter prediction, no objective standard exists for defining CpG islands (CGIs), since all current approaches rely on a large parameter space formed by the thresholds of length, CpG fraction and G+C content.
RESULTS: Given the higher frequency of CpG dinucleotides at CGIs, as compared to bulk DNA, the distance distributions between neighboring CpGs should differ for bulk and island CpGs. A new algorithm (CpGcluster) is presented, based on the physical distance between neighboring CpGs on the chromosome and able to predict directly clusters of CpGs, while not depending on the subjective criteria mentioned above. By assigning a p-value to each of these clusters, the most statistically significant ones can be predicted as CGIs. CpGcluster was benchmarked against five other CGI finders by using a test sequence set assembled from an experimental CGI library. CpGcluster reached the highest overall accuracy values, while showing the lowest rate of false-positive predictions. Since a minimum-length threshold is not required, CpGcluster can find short but fully functional CGIs usually missed by other algorithms. The CGIs predicted by CpGcluster present the lowest degree of overlap with Alu retrotransposons and, simultaneously, the highest overlap with vertebrate Phylogenetic Conserved Elements (PhastCons). CpGcluster's CGIs overlapping with the Transcription Start Site (TSS) show the highest statistical significance, as compared to the islands in other genome locations, thus qualifying CpGcluster as a valuable tool in discriminating functional CGIs from the remaining islands in the bulk genome.
CONCLUSION: CpGcluster uses only integer arithmetic, thus being a fast and computationally efficient algorithm able to predict statistically significant clusters of CpG dinucleotides. Another outstanding feature is that all predicted CGIs start and end with a CpG dinucleotide, which should be appropriate for a genomic feature whose functionality is based precisely on CpG dinucleotides. The only search parameter in CpGcluster is the distance between two consecutive CpGs, in contrast to previous algorithms. Therefore, none of the main statistical properties of CpG islands (neither G+C content, CpG fraction nor length threshold) are needed as search parameters, which may lead to the high specificity and low overlap with spurious Alu elements observed for CpGcluster predictions.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
RESULTS: Given the higher frequency of CpG dinucleotides at CGIs, as compared to bulk DNA, the distance distributions between neighboring CpGs should differ for bulk and island CpGs. A new algorithm (CpGcluster) is presented, based on the physical distance between neighboring CpGs on the chromosome and able to predict directly clusters of CpGs, while not depending on the subjective criteria mentioned above. By assigning a p-value to each of these clusters, the most statistically significant ones can be predicted as CGIs. CpGcluster was benchmarked against five other CGI finders by using a test sequence set assembled from an experimental CGI library. CpGcluster reached the highest overall accuracy values, while showing the lowest rate of false-positive predictions. Since a minimum-length threshold is not required, CpGcluster can find short but fully functional CGIs usually missed by other algorithms. The CGIs predicted by CpGcluster present the lowest degree of overlap with Alu retrotransposons and, simultaneously, the highest overlap with vertebrate Phylogenetic Conserved Elements (PhastCons). CpGcluster's CGIs overlapping with the Transcription Start Site (TSS) show the highest statistical significance, as compared to the islands in other genome locations, thus qualifying CpGcluster as a valuable tool in discriminating functional CGIs from the remaining islands in the bulk genome.
CONCLUSION: CpGcluster uses only integer arithmetic, thus being a fast and computationally efficient algorithm able to predict statistically significant clusters of CpG dinucleotides. Another outstanding feature is that all predicted CGIs start and end with a CpG dinucleotide, which should be appropriate for a genomic feature whose functionality is based precisely on CpG dinucleotides. The only search parameter in CpGcluster is the distance between two consecutive CpGs, in contrast to previous algorithms. Therefore, none of the main statistical properties of CpG islands (neither G+C content, CpG fraction nor length threshold) are needed as search parameters, which may lead to the high specificity and low overlap with spurious Alu elements observed for CpGcluster predictions.
2005
Hackenberg, Michael; Bernaola-Galván, Pedro; Carpena, Pedro; Oliver, José L.
The biased distribution of Alus in human isochores might be driven by recombination Journal Article
In: Journal of Molecular Evolution, vol. 60, no. 3, pp. 365–377, 2005, ISSN: 0022-2844.
@article{hackenberg_biased_2005,
title = {The biased distribution of Alus in human isochores might be driven by recombination},
author = {Michael Hackenberg and Pedro Bernaola-Galván and Pedro Carpena and José L. Oliver},
doi = {10.1007/s00239-004-0197-2},
issn = {0022-2844},
year = {2005},
date = {2005-03-01},
journal = {Journal of Molecular Evolution},
volume = {60},
number = {3},
pages = {365–377},
abstract = {Alu retrotransposons do not show a homogeneous distribution over the human genome but have a higher density in GC-rich (H) than in AT-rich (L) isochores. However, since they preferentially insert into the L isochores, the question arises: What is the evolutionary mechanism that shifts the Alu density maximum from L to H isochores? To disclose the role played by each of the potential mechanisms involved in such biased distribution, we carried out a genome-wide analysis of the density of the Alus as a function of their evolutionary age, isochore membership, and intron vs. intergene location. Since Alus depend on the retrotransposase encoded by the LINE1 elements, we also studied the distribution of LINE1 to provide a complete evolutionary scenario. We consecutively check, and discard, the contributions of the Alu/LINE1 competition for retrotransposase, compositional matching pressure, and Alu overrepresentation in introns. In analyzing the role played by unequal recombination, we scan the genome for Alu trimers, a direct product of Alu-Alu recombination. Through computer simulations, we show that such trimers are much more frequent than expected, the observed/expected ratio being higher in L than in H isochores. This result, together with the known higher selective disadvantage of recombination products in H isochores, points to Alu-Alu recombination as the main agent provoking the density shift of Alus toward the GC-rich parts of the genome. Two independent pieces of evidence-the lower evolutionary divergence shown by recently inserted Alu subfamilies and the higher frequency of old stand-alone Alus in L isochores-support such a conclusion. Other evolutionary factors, such as population bottlenecks during primate speciation, may have accelerated the fast accumulation of Alus in GC-rich isochores.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2004
Oliver, José L.; Carpena, Pedro; Hackenberg, Michael; Bernaola-Galván, Pedro
IsoFinder: computational prediction of isochores in genome sequences Journal Article
In: Nucleic Acids Research, vol. 32, no. Web Server issue, pp. W287–292, 2004, ISSN: 1362-4962.
@article{oliver_isofinder_2004,
title = {IsoFinder: computational prediction of isochores in genome sequences},
author = {José L. Oliver and Pedro Carpena and Michael Hackenberg and Pedro Bernaola-Galván},
doi = {10.1093/nar/gkh399},
issn = {1362-4962},
year = {2004},
date = {2004-07-01},
journal = {Nucleic Acids Research},
volume = {32},
number = {Web Server issue},
pages = {W287–292},
abstract = {Isochores are long genome segments homogeneous in G+C. Here, we describe an algorithm (IsoFinder) running on the web (http://bioinfo2.ugr.es/IsoF/isofinder.html) able to predict isochores at the sequence level. We move a sliding pointer from left to right along the DNA sequence. At each position of the pointer, we compute the mean G+C values to the left and to the right of the pointer. We then determine the position of the pointer for which the difference between left and right mean values (as measured by the t-statistic) reaches its maximum. Next, we determine the statistical significance of this potential cutting point, after filtering out short-scale heterogeneities below 3 kb by applying a coarse-graining technique. Finally, the program checks whether this significance exceeds a probability threshold. If so, the sequence is cut at this point into two subsequences; otherwise, the sequence remains undivided. The procedure continues recursively for each of the two resulting subsequences created by each cut. This leads to the decomposition of a chromosome sequence into long homogeneous genome regions (LHGRs) with well-defined mean G+C contents, each significantly different from the G+C contents of the adjacent LHGRs. Most LHGRs can be identified with Bernardi's isochores, given their correlation with biological features such as gene density, SINE and LINE (short, long interspersed repetitive elements) densities, recombination rate or single nucleotide polymorphism variability. The resulting isochore maps are available at our web site (http://bioinfo2.ugr.es/isochores/), and also at the UCSC Genome Browser (http://genome.cse.ucsc.edu/).},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
2002
Oliver, José L.; Carpena, Pedro; Román-Roldán, Ramón; Mata-Balaguer, Trinidad; Mejías-Romero, Andrés; Hackenberg, Michael; Bernaola-Galván, Pedro
Isochore chromosome maps of the human genome Journal Article
In: Gene, vol. 300, no. 1-2, pp. 117–127, 2002, ISSN: 0378-1119.
@article{oliver_isochore_2002,
title = {Isochore chromosome maps of the human genome},
author = {José L. Oliver and Pedro Carpena and Ramón Román-Roldán and Trinidad Mata-Balaguer and Andrés Mejías-Romero and Michael Hackenberg and Pedro Bernaola-Galván},
doi = {10.1016/s0378-1119(02)01034-x},
issn = {0378-1119},
year = {2002},
date = {2002-10-01},
journal = {Gene},
volume = {300},
number = {1-2},
pages = {117–127},
abstract = {The human genome is a mosaic of isochores, which are long DNA segments (z.Gt;300 kbp) relatively homogeneous in G+C. Human isochores were first identified by density-gradient ultracentrifugation of bulk DNA, and differ in important features, e.g. genes are found predominantly in the GC-richest isochores. Here, we use a reliable segmentation method to partition the longest contigs in the human genome draft sequence into long homogeneous genome regions (LHGRs), thereby revealing the isochore structure of the human genome. The advantages of the isochore maps presented here are: (1) sequence heterogeneities at different scales are shown in the same plot; (2) pair-wise compositional differences between adjacent regions are all statistically significant; (3) isochore boundaries are accurately defined to single base pair resolution; and (4) both gradual and abrupt isochore boundaries are simultaneously revealed. Taking advantage of the wide sample of genome sequence analyzed, we investigate the correspondence between LHGRs and true human isochores revealed through DNA centrifugation. LHGRs show many of the typical isochore features, mainly size distribution, G+C range, and proportions of the isochore classes. The relative density of genes, Alu and long interspersed nuclear element repeats and the different types of single nucleotide polymorphisms on LHGRs also coincide with expectations in true isochores. Potential applications of isochore maps range from the improvement of gene-finding algorithms to the prediction of linkage disequilibrium levels in association studies between marker genes and complex traits. The coordinates for the LHGRs identified in all the contigs longer than 2 Mb in the human genome sequence are available at the online resource on isochore mapping: http://bioinfo2.ugr.es/isochores.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
