Publication BibTeX

These records are generated from the same BibTeX source used for the publications page and CV.

Fiber-TEnCATS reveals haplotype-specific chromatin accessibility and DNA methylation at human L1HS loci

@article{Pavlovic2026FiberTEnCATS,
  author = {Pavlovic, Katarina and McDonald, Torrin L. and Diehl, Adam G. and Switzenberg, Jessica A. and Boyle, Alan P.},
  title = {{Fiber-TEnCATS reveals haplotype-specific chromatin accessibility and DNA methylation at human L1HS loci}},
  year = {2026},
  doi = {10.64898/2026.06.26.734832},
  abstract = {Human-specific long interspersed nuclear element-1 (L1HS) is an active and autonomous retrotransposon in the human genome. Changes in its transcription and transposition are known to affect cellular processes involved in development and aging, and diseases such as neurological disorders and cancer. To better understand natural variability in epigenetic patterns that affect L1HS regulation, we developed a targeted long-read method to simultaneously profile individual haplotypes for DNA methylation and chromatin accessibility across L1HS loci in a healthy human cell line trio. We show that the intronic L1HS in the ZNF638 gene consistently displays high chromatin accessibility and DNA hypomethylation with bidirectional transcription. Our approach also reveals additional intronic and intergenic L1HS copies with allele-specific chromatin accessibility and methylation, and instances of reduced promoter DNA methylation that does not correspond with increased chromatin accessibility. We also identify potential cases of non-Mendelian inheritance of DNA methylation patterns over a subset of L1HS promoters. Our method{\textquoteright}s high coverage over L1HS loci enables detection and profiling of loci that are missed even by long-read-based assemblies and enables more accurate inheritance tracing of L1HS insertions. Overall, our results offer new insights into the locus-specific regulation of both reference and non-reference L1HS within the human genome.Competing Interest StatementThe authors have declared no competing interest.National Institutes of Health, R01 GM144484},
  url = {https://www.biorxiv.org/content/early/2026/06/28/2026.06.26.734832},
  pdf = {https://www.biorxiv.org/content/early/2026/06/28/2026.06.26.734832.full.pdf},
  journal = {bioRxiv}
}

GPatch enables chromosome-scale, gap-free pseudoassemblies from fragmented draft genomes

@article{Diehl2026GPatch,
  author = {Diehl, Adam G and Boyle, Alan P},
  title = {GPatch enables chromosome-scale, gap-free pseudoassemblies from fragmented draft genomes},
  year = {2026},
  doi = {10.1101/2025.05.22.655567},
  abstract = {Recent advancements in sequencing technologies have yielded numerous long-read draft genomes, promising to enhance understanding of genomic variation. However, draft genomes are typically highly fragmented, posing challenges for functional genomics. We introduce GPatch, a tool that constructs chromosome-scale pseudoassemblies from fragmented drafts using alignments to a reference assembly. GPatch produces complete, accurate, gap-free assemblies preserving over 95\% of nucleotides from human and non-human draft genomes. We show that GPatch pseudoassemblies can be used to construct Hi-C matrices, whereas fragmented draft assemblies cannot. Until complete genome assembly becomes routine, GPatch presents a necessary tool for maximizing the utility of draft genomes.Competing Interest StatementThe authors have declared no competing interest.National Institutes of Health, https://ror.org/01cwqze88, R01GM144484},
  url = {https://www.biorxiv.org/content/early/2026/06/09/2025.05.22.655567},
  pdf = {https://www.biorxiv.org/content/early/2026/06/09/2025.05.22.655567.full.pdf},
  journal = {bioRxiv}
}

SEMPLR: an R package for transcription factor binding prediction

@article{Kenney2026SEMPLR,
  author = {Kenney, Grace E and Sherpa, Rintsen N and Burgess, Jeremy D and Boyle, Alan P and Phanstiel, Douglas H},
  title = {{SEMPLR: an R package for transcription factor binding prediction}},
  journal = {Bioinformatics},
  volume = {42},
  number = {6},
  pages = {btag383},
  year = {2026},
  month = {06},
  abstract = {{SEMPLR is an R package that predicts transcription factor binding and variant effects using SNP Effect Matrices (SEMs), providing efficient, genome-wide scoring, enrichment testing, and visualization tools for comprehensive analysis of regulatory sequences.Available on GitHub at https://github.com/grkenney/SEMPLR and on Bioconductor at https://bioconductor.org/packages/release/bioc/html/SEMPLR.html.}},
  doi = {10.1093/bioinformatics/btag383},
  url = {https://doi.org/10.1093/bioinformatics/btag383},
  pdf = {https://academic.oup.com/bioinformatics/article-pdf/42/6/btag383/68520077/btag383.pdf},
  note = {{PMID:} 42286344}
}

3D chromatin compartment of round spermatids encodes the spatiotemporal program of histone-to-protamine exchange in spermiogenesis

@article{Rabbani2026SpermatidChromatin,
  author = {Rabbani, Mashiat and Apell, Zachary and Parnell, Timothy J. and Moritz, Lindsay and Kim, Sion and Srinivasan, Sowmya and Agrawal, Ritvija and Vargo, Alexander and Orchard, Peter and Xie, Wenxin and Freddolino, Lydia and Boyle, Alan P and Li, Jun Z. and Lesch, Bluma J. and Cairns, Bradley and Kim, Minji and Wilson, Thomas E. and Hammoud, Saher Sue},
  title = {{3D chromatin compartment of round spermatids encodes the spatiotemporal program of histone-to-protamine exchange in spermiogenesis}},
  year = {2026},
  doi = {10.64898/2026.03.10.710708},
  abstract = {Sperm formation requires a radical chromatin reorganization, where nucleosomes are replaced by transition proteins (TNPs) and subsequently by protamines (PRM1 and PRM2). Although essential for fertility, the regulatory logic governing this exchange is unknown, but it{\textquoteright}s presumed to be stochastic and unregulated. Using endogenously tagged PRM mouse models and stage-resolved, genome-wide profiling, we revise the order of histone-to-protamine exchange and show that the chromatin remodeling process is highly programmed. Imaging and biochemical experiments reveal a direct histone-to-PRM1 exchange, while TNPs appear after PRM1 but precede PRM2 incorporation. This temporal uncoupling of PRM1 and PRM2 incorporation coincides with dynamic, region-specific chromatin remodeling that is not governed by histone acetylation but is instructed by the three-dimensional nuclear architecture of round spermatids. Therefore, by integrating ATAC-seq, CUT\&Tag, and Hi-C, we define a compartment-encoded {\textquotedblleft}blueprint{\textquotedblright} that prescribes the assembly of the mature sperm epigenome and establishes a complex molecular hierarchy for the histone-to-protamine exchange.Competing Interest StatementThe authors have declared no competing interest.National Institute of Health, GM148028, 1DP2HD091949-01, R01HD104680 01, R01HD113274, GM148028, 5T32HD079342-10University of Michigan, https://ror.org/00jmfr291Open Philanthropy Grant, 2019-199327 (5384)},
  url = {https://www.biorxiv.org/content/early/2026/03/12/2026.03.10.710708},
  pdf = {https://www.biorxiv.org/content/early/2026/03/12/2026.03.10.710708.full.pdf},
  journal = {bioRxiv}
}

TFBSpedia: a comprehensive human and mouse transcription factor binding sites database

@article{Li2026TFBSpedia,
  author = {Li, Shiting and Chou, Elysia and Wang, Kai and Boyle, Alan P. and Sartor, Maureen A.},
  title = {{TFBSpedia: a comprehensive human and mouse transcription factor binding sites database}},
  year = {2026},
  doi = {10.64898/2026.03.04.709638},
  abstract = {Mapping the genomic locations and patterns of transcription factor binding sites (TFBS) is essential for understanding gene regulation and advancing treatments for diseases driven by DNA modifications, including epigenetic changes and sequence variants. Although several TFBS databases exist, no study has systematically benchmarked these databases across different sequencing technologies and computational algorithms. In this study, we addressed this gap by constructing a TFBS database that integrates all available ENCODE cell line ATAC-seq and Cistrome Data Browser ChIP-seq datasets, comprising 11.3 million human and 1.87 million mouse TFBS. We also integrated previously published TFBS resources (Factorbook, Unibind, RegulomeDB, and ENCODE_footprint) and found each contains a substantial fraction of unique TFBS predictions, highlighting significant discrepancies among existing resources. To assess the accuracy of the combined TFBS regions, we assembled ten independent genomic annotation datasets for evaluation and found that TFBS regions predicted by multiple databases are more likely to represent true and biologically meaningful binding sites. For each predicted TFBS region, we define two scores: the confidence score reflects prediction reliability, while the importance score represents biological functional relevance. Finally, we introduce TFBSpedia, a lightweight and efficient search engine that enables rapid retrieval of TFBS regions and comprehensive annotation information across the integrated databases.Competing Interest StatementThe authors have declared no competing interest.},
  url = {https://www.biorxiv.org/content/early/2026/03/06/2026.03.04.709638},
  pdf = {https://www.biorxiv.org/content/early/2026/03/06/2026.03.04.709638.full.pdf},
  journal = {bioRxiv}
}

Transcriptomic analysis to uncover the mechanism of radiosensitization of AR-positive triple-negative breast cancers with AR inhibition

@article{McBean2026RadiosensitizationARPositiveTNBC,
  author = {McBean, Breanna and Hauk, Benjamin and Michmerhuizen, Anna R and Chandler, Benjamin C and Pesch, Andrea M and Lerner, Lynn M and Gurdak, Douglas and Ward, Connor and Rana, Priyanka and Zeidane, Reine Abou and Mercer, Vesna and Tao, Mingfang and Hochmuth, Ethan and Jungles, Kassidy M and The, Stephanie and Liu, Meilan and Spratt, Daniel E and Boyle, Alan P and Pierce, Lori J and Speers, Corey W},
  title = {{Transcriptomic analysis to uncover the mechanism of radiosensitization of {AR-positive} triple-negative breast cancers with AR inhibition}},
  journal = {NPJ Breast Cancer},
  volume = {12},
  number = {1},
  year = {2026},
  month = {2},
  abstract = {The androgen receptor (AR) has been identified as a driver of tumor growth and radioresistance in triple-negative breast cancers (TNBC), though the mechanistic role of AR in response to radiation therapy (RT) remains unknown. Here, we demonstrate that inhibition with the second-generation anti-androgen, apalutamide, but not darolutamide, is sufficient to radiosensitize AR+ TNBC models (rER: 1.34-1.41; rER: 0.96-1.11, respectively). Cells with low AR expression were not radiosensitized by AR inhibition (rER: 0.96-1.03). Mechanistically, while stimulation with the AR-agonist R1881 is sufficient to induce nuclear translocation of AR in AR+ TNBC cells, AR inhibition with enzalutamide, apalutamide, or darolutamide blocked AR nuclear translocation. When cells are treated with R1881+RT, nuclear translocation of AR was induced at similar or greater levels compared to R1881 alone in AR+ TNBC cells. Combination treatment of RT with enzalutamide reduced nuclear localization of AR (32-39% reduction) compared to RT alone. Transcriptional evaluation with RNA-Seq after AR stimulation and RT demonstrated changes in the MAPK/ERK signaling pathway, among others. Overexpression of ERK reduces the radiosensitizing ability of second-generation anti-androgens, suggesting that AR-mediated radioresistance may be due, at least in part, to downstream MAPK/ERK signaling. These findings suggest that AR-mediated radioresistance is at least partially due to downstream MAPK/ERK signaling. Together this work builds on the mechanistic understanding of AR-mediated radioresistance in AR+ TNBC which may expose vulnerabilities in resistance to combination treatment with AR inhibition and RT.},
  doi = {10.1038/s41523-026-00916-1},
  url = {https://www.nature.com/articles/s41523-026-00916-1},
  note = {{PMID:} 41735341}
}

The IGVF catalog—from genetic variation to function

@article{Li2025IGVFCatalog,
  author = {Li, Daofeng and Liu, Shane and Assis, Pedro R and Li, Mingjie and Dong, Shengcheng and Whaling, Ian and Jolanki, Otto and Kagda, Meenakshi and Zhang, Wenjin and Macias-Velasco, Juan F and Liu, Tianjie and Cody, Sarah and Antonacci-Fulton, Lucinda and Huang, Yuanhao and Liu, Jie and Montgomery, Michael T and Zeiberg, Daniel and Jain, Shantanu and Pejaver, Vikas and Bergquist, Timothy and Chen, Yile and Radivojac, Predrag and Gersbach, Charles A and Sherpa, Rintsen N and Castro, Christopher P and Boyle, Alan P and Starita, Lea M and Fowler, Douglas M and Ahituv, Nadav and Dey, Kushal K and Majoros, William H and Reddy, Timothy E and Craven, Mark and Sinha, Riya and Sverchkov, Yuriy and Cai, Xiangmeng and Nzima, Mpathi Z and Calderwood, Michael A and Rozowsky, Joel and Gerstein, Mark and Ma, Jian and Yue, Feng and Cherry, J Michael and Love, Michael I and Engreitz, Jesse M and Hitz, Benjamin C and Wang, Ting},
  title = {{The IGVF catalog—from genetic variation to function}},
  journal = {Nucleic Acids Research},
  pages = {gkaf1341},
  year = {2025},
  month = {12},
  abstract = {Genomic variation between individuals is essential for understanding how differences in the genome sequence affect molecular and cellular processes. The Impact of Genomic Variation on Function (IGVF) Consortium aims to uncover the relationships among genomic variation, genome function, and phenotypes by combining experimental techniques, such as single-cell mapping and genomic perturbation assays, with computational approaches such as machine learning-based predictive modeling. The IGVF Data and Administrative Coordinating Centers collect, analyze, and disseminate data and results from across the consortium through an open-source platform called the IGVF Catalog. This resource includes, but is not limited to, data on the effects of coding variants on protein abundance and function, noncoding variants on enhancer activity (measured by MPRA or predicted computationally), and associations between variants and quantitative traits. All data are organized within a graph database comprising over 50 types of data collections with nearly 3 billion nodes and over 7.5 billion edges. The Catalog offers public API endpoints (https://api.catalogkg.igvf.org/) and a user-friendly interface for exploring, querying, and visualizing the data at https://catalog.igvf.org. We expect that this open-access platform will support the broader scientific community to advance our understanding of how genomic variation influences biology and disease.},
  issn = {1362-4962},
  doi = {10.1093/nar/gkaf1341},
  url = {https://doi.org/10.1093/nar/gkaf1341},
  note = {{PMID:} 41359121}
}

Comprehensive benchmarking of somatic mutation detection by the SMaHT Network

@article{SMaHTNetwork2025SomaticMutationBenchmarking,
  author = {{The Somatic Mosaicism across Human Tissues Network (SMaHT)}},
  title = {{Comprehensive benchmarking of somatic mutation detection by the SMaHT Network}},
  year = {2025},
  doi = {10.1101/2025.10.09.678885},
  publisher = {Cold Spring Harbor Laboratory},
  abstract = {Somatic mosaicism is increasingly recognized as a fundamental feature of human biology, yet the detection of somatic mutations remains challenging. The SMaHT Network conducted four large-scale benchmarking experiments to evaluate sequencing technologies, experimental approaches, and computational methods for detecting diverse somatic mutations. Cumulative sequencing coverage exceeded 1,000{\texttimes} with short reads and 100-400{\texttimes} with long reads for each of nine analyzed samples. We defined optimal strategies for integrating bulk short- and long-read sequencing for mutation detection and demonstrated that using donor-specific assemblies and human pangenome improved variant calling and extended mutation catalogs to challenging genomic regions. We benchmarked six duplex-seq technologies and showed that single-cell sequencing resolves cell type-specific mutational patterns and heterogeneity. Our results indicate that bulk, single-cell, and duplex analyses are complementary {\textendash} and leveraging all three provides comprehensive characterization of mosaicism within a tissue. Together, these findings provide a roadmap for accurate, genome-wide somatic mutation discovery and analysis.Competing Interest StatementCOINIH Common Fund, https://ror.org/001d55x84},
  url = {https://www.biorxiv.org/content/early/2025/10/10/2025.10.09.678885},
  pdf = {https://www.biorxiv.org/content/early/2025/10/10/2025.10.09.678885.full.pdf},
  journal = {bioRxiv}
}

Multi-platform framework for mapping somatic retrotransposition in‬ human tissues

@article{Wang2025SomaticRetrotransposition,
  author = {*Wang, Seunghyun and *Bae, Mingyun and *Wang, Jinhao and Zhao, Boxun and Nguyen, Khue and Mallett, Shayna and Switzenberg, Jessica A. and Losh, Steven J. and Sexton, Corinne E. and Miao, Benpeng and Dong, Shihua and Zeng, Xi and Wang, Ziying and McDonald, Torrin L. and Mumm, Camille and Gadde, Rohini K. and Tariq, Arnaz Maryam and Chen, Zhuofu and Feng, William C. and Burn, Aidan and Park, Junseok and Chu, Chong and Shen, Hui and Wang, Ting and Urban, Alexander E. and Zhu, Xiaowei and Li, Heng and Burns, Kathleen H. and Chun, Hye-Jung E. and Park, Peter J. and SMaHT MEI Working Group and {\dag}Boyle, Alan P and {\dag}Mills, Ryan E. and {\dag}Zhou, Weichen and {\dag}Lee, Eunjung Alice},
  title = {Multi-platform framework for mapping somatic retrotransposition in‬ human tissues},
  year = {2025},
  doi = {10.1101/2025.10.07.680917},
  abstract = {Mobile element insertions (MEI) shape the human genome in both germline and somatic tissues. While inherited MEIs are well characterized, mapping somatic MEIs (sMEI) in non-cancer tissues remains challenging due to their low allelic fraction and repetitive nature. We established an integrative framework for sMEI analysis leveraging modern sequencing technologies and analytical innovations. We first benchmarked sMEI detection and demonstrated advantages of long-read and MEI-targeted sequencing for ultra-low-frequency events using a mixture of well-established cell lines. We then showed that haplotype phasing and donor-specific assemblies refine sMEI detection, effectively distinguishing from germline and false signals in in-silico tumor-normal mixtures. We further developed a source-tracing strategy based on internal sequence variation, expanding the catalogue of active source elements beyond traditional transduction-based methods. Applying this framework to donor tissues, we identified 18 rare somatic L1 insertions, revealing structural and source diversity. Our work provides a foundational framework and biological insight into sMEIs.},
  url = {https://www.biorxiv.org/content/early/2025/10/07/2025.10.07.680917},
  pdf = {https://www.biorxiv.org/content/early/2025/10/07/2025.10.07.680917.full.pdf},
  journal = {bioRxiv}
}

Long-term immune reprogramming of classical monocytes with altered ontogeny mediates enhanced lung injury in sepsis survivors

@article{Denstaedt2025MonocyteReprogramming,
  author = {Denstaedt, Scott J. and McBean, Breanna and Boyle, Alan P and Arenberg, Brett C. and Mack, Matthias and Moore, Bethany B. and Newstead, Michael W. and Deng, Yanmei and Nesvizhskii, Alexey I. and Singer, Benjamin H. and Cano, Jennifer and Prescott, Hallie C. and Goodridge, Helen S. and Zemans, Rachel L.},
  title = {Long-term immune reprogramming of classical monocytes with altered ontogeny mediates enhanced lung injury in sepsis survivors},
  year = {2025},
  doi = {10.1101/2025.05.16.654442},
  abstract = {Patients who survive sepsis are predisposed to new hospitalizations for respiratory failure, but the underlying mechanisms are unknown. Using a murine model in which prior sepsis predisposes to enhanced lung injury, we previously discovered that classical monocytes persist in the lungs after long-term recovery from sepsis and exhibit enhanced cytokine expression after secondary challenge with intra-nasal lipopolysaccharide. Here, we hypothesized that immune reprogramming of post-sepsis monocytes and altered ontogeny predispose to enhanced lung injury. Monocyte depletion and/or adoptive transfer was performed three weeks and three months after sepsis. Monocytes from post-sepsis mice were necessary and sufficient for enhanced LPS-induced lung injury and promoted neutrophil degranulation. Prior sepsis enhanced JAK-STAT signaling and AP-1 binding in monocytes and shifted monocytes toward the neutrophil-like monocyte lineage. In human sepsis and/or pneumonia survivors, monocytes were predictive of 90-day mortality and exhibit transcriptional and proteomic neutrophil-like signatures. We conclude that sepsis reprograms monocytes into a pro-inflammatory phenotype and skews bone marrow progenitors and monocytes toward the neutrophil-like lineage, predisposing to neutrophil degranulation and lung injury.Competing Interest StatementConflicts of Interest: A.I.N. is the founder of Fragmatics and serves on the scientific advisory boards of Protai Bio, Infinitopes, and Mobilion Systems.National Institutes of Health, https://ror.org/01cwqze88, K08HL153799, U01HG011952, T32HG000040, R35HL144481, R01AG074968, R35HL160770National Institutes of Health, https://ror.org/01cwqze88, R01AI134987, R01HL147920, R01HL131608, R35HL160770, R01HS026725United States Department of Veterans Affairs, VA HSR 20-313},
  url = {https://www.biorxiv.org/content/early/2025/08/01/2025.05.16.654442},
  pdf = {https://www.biorxiv.org/content/early/2025/08/01/2025.05.16.654442.full.pdf},
  journal = {bioRxiv}
}

The Somatic Mosaicism across Human Tissues Network

@article{Coorens2025SMaHTNetwork,
  title = {{The Somatic Mosaicism across Human Tissues Network}},
  author = {Coorens, Tim H. H. and Oh, Ji Won and Choi, Yujin Angelina and Lim, Nam Seop and Zhao, Boxun and Voshall, Adam and Abyzov, Alexej and Antonacci-Fulton, Lucinda and Aparicio, Samuel and Ardlie, Kristin G. and Bell, Thomas J. and Bennett, James T. and Bernstein, Bradley E. and Blanchard, Thomas G. and Boyle, Alan P. and Buenrostro, Jason D. and Burns, Kathleen H. and Chen, Fei and Chen, Rui and Choudhury, Sangita and Doddapaneni, Harsha V. and Eichler, Evan E. and Evrony, Gilad D. and Faith, Melissa A. and Fazzio, Thomas G. and Fulton, Robert S. and Garber, Manuel and Gehlenborg, Nils and Germer, Soren and Getz, Gad and Gibbs, Richard A. and Hernandez, Raquel G. and Jin, Fulai and Korbel, Jan O. and Landau, Dan A. and Lawson, Heather A. and Lennon, Niall J. and Li, Heng and Li, Yan and Loh, Po-Ru and Marth, Gabor and McConnell, Michael J. and Mills, Ryan E. and Montgomery, Stephen B. and Natarajan, Pradeep and Park, Peter J. and Satija, Rahul and Sedlazeck, Fritz J. and Shao, Diane D. and Shen, Hui and Stergachis, Andrew B. and Underhill, Hunter R. and Urban, Alexander E. and VonDran, Melissa W. and Walsh, Christopher A. and Wang, Ting and Wu, Tao P. and Zong, Chenghang and Lee, Eunjung Alice and Vaccarino, Flora M. and {Somatic Mosaicism across Human Tissues Network}},
  journal = {Nature},
  year = {2025},
  month = {Jul},
  day = {01},
  volume = {643},
  number = {8070},
  pages = {47-59},
  abstract = {From fertilization onwards, the cells of the human body acquire variations in their DNA sequence, known as somatic mutations. These postzygotic mutations arise from intrinsic errors in DNA replication and repair, as well as from exposure to mutagens. Somatic mutations have been implicated in some diseases, but a fundamental understanding of the frequency, type and patterns of mutations across healthy human tissues has been limited. This is primarily due to the small proportion of cells harbouring specific somatic variants within an individual, making them more challenging to detect than inherited variants. Here we describe the Somatic Mosaicism across Human Tissues Network, which aims to create a reference catalogue of somatic mutations and their clonal patterns across 19 different tissue sites from 150 non-diseased donors and develop new technologies and computational tools to detect somatic mutations and assess their phenotypic consequences, including clonal expansions. This strategy enables a comprehensive examination of the mutational landscape across the human body, and provides a comparison baseline for somatic mutation in diseases. This will lead to a deep understanding of somatic mutations and clonal expansions across the lifespan, as well as their roles in health, in ageing and, by comparison, in diseases.},
  doi = {10.1038/s41586-025-09096-7},
  url = {https://doi.org/10.1038/s41586-025-09096-7},
  pdf = {http://boylelab.org/pubs/Nature_2025_SMaHT.pdf},
  note = {{PMID:} 40604182}
}

Enhanced detection and genotyping of disease-associated tandem repeats using HMMSTR and targeted long-read sequencing

@article{VanDeynze2025HMMSTR,
  author = {*Van Deynze, Kinsey and *Mumm, Camille and Maltby, Connor J and Switzenberg, Jessica A and Todd, Peter K and Boyle, Alan P},
  title = {{Enhanced detection and genotyping of disease-associated tandem repeats using HMMSTR and targeted long-read sequencing}},
  journal = {Nucleic Acids Research},
  volume = {53},
  number = {2},
  pages = {gkae1202},
  year = {2025},
  month = {Jan},
  abstract = {Tandem repeat sequences comprise approximately 8\% of the human genome and are linked to more than 50 neurodegenerative disorders. Accurate characterization of disease-associated repeat loci remains resource intensive and often lacks high resolution genotype calls. We introduce a multiplexed, targeted nanopore sequencing panel and HMMSTR, a sequence-based tandem repeat copy number caller which outperforms current signal- and sequence-based callers relative to two assemblies and we show it performs with high accuracy in heterozygous regions and at low read coverage. The flexible panel allows us to capture disease associated regions at an average coverage of \>150x. Using these tools, we successfully characterize known or suspected repeat expansions in patient derived samples. In these samples, we also identify unexpected expanded alleles at tandem repeat loci not previously associated with the underlying diagnosis. This genotyping approach for tandem repeat expansions is scalable, simple, flexible and accurate, offering significant potential for diagnostic applications and investigation of expansion co-occurrence in neurodegenerative disorders.},
  doi = {10.1093/nar/gkae1202},
  url = {https://doi.org/10.1093/nar/gkae1202},
  pdf = {http://boylelab.org/pubs/NAR_2024_VanDeynze.pdf},
  note = {{PMID:} 39676678}
}

Draft de novo genome construction of Scytonema sp. PRP1: identified from single-cell sequencing library preparation

@article{Parana2025ScytonemaGenome,
  author = {Preston Parana and Camille Mumm and Michael J. McConnell and Alan P. Boyle},
  title = {{Draft de novo genome construction of Scytonema sp. PRP1: identified from single-cell sequencing library preparation}},
  journal = {Microbiology Resource Announcements},
  pages = {e00029-25},
  year = {2025},
  doi = {10.1128/mra.00029-25},
  url = {https://journals.asm.org/doi/abs/10.1128/mra.00029-25},
  pdf = {http://boylelab.org/pubs/MRA_2025_Parana.pdf},
  abstract = {We present the genome sequence of Scytonema sp. PRP1, a cyanobacterium identified during single-cell sequencing library preparation of de-identified human brain samples associated with the Brain Somatic Mosaicism Network. Our 8,266,022 bp genome comprises 140 contigs, containing 6,765 protein-coding sequences, 37 tRNA genes, and 11 rRNA genes. },
  note = {{PMID:} 40744872}
}

Cryptic intronic transcriptional initiation generates efficient endogenous mRNA templates for C9orf72-associated RAN translation

@article{Miller2025C9orf72RANTranslation,
  author = {Shannon L. Miller  and Katelyn M. Green  and Bradley Crone  and Jessica A. Switzenberg  and Elizabeth M. H. Tank  and Amy Krans  and Karen Jansen-West  and Clare M. Wieland  and Eric W. Ji  and Leonard Petrucelli  and Sami J. Barmada  and Alan P. Boyle  and Peter K. Todd },
  title = {{Cryptic intronic transcriptional initiation generates efficient endogenous mRNA templates for C9orf72-associated RAN translation}},
  journal = {Proceedings of the National Academy of Sciences},
  volume = {122},
  number = {32},
  pages = {e2507334122},
  year = {2025},
  doi = {10.1073/pnas.2507334122},
  url = {https://www.pnas.org/doi/abs/10.1073/pnas.2507334122},
  pdf = {http://boylelab.org/pubs/PNAS_2025_Miller.pdf},
  abstract = {An intronic GGGGCC repeat expansion in C9orf72 supports an unusual translational initiation process known as repeat-associated non-AUG (RAN) translation to produce toxic dipeptide repeat (DPR) proteins that contribute to neurodegeneration in ALS and FTD. How an intronic repeat RNA engages with ribosomes to support such translation is unclear. Here, we identify a series of previously unannotated mRNA transcripts that initiate within the repeat-containing intron to create linear m7 G-capped templates for RAN translation from GGGGCC repeats. These cryptic mRNAs are present in patient iNeurons, engage with ribosomes, and robustly support RAN translation. This finding has important implications for both our understanding of the mechanism by which RAN translation occurs and on therapeutic development in this currently untreatable class of neurodegenerative disorders. Intronic GGGGCC hexanucleotide repeat expansions in C9orf72 are the most common genetic cause of amyotrophic lateral sclerosis (ALS) and frontotemporal dementia (FTD). Despite its intronic location, this repeat avidly supports synthesis of pathogenic dipeptide repeat (DPR) proteins via repeat-associated non-AUG (RAN) translation. However, the template RNA species that undergoes RAN translation endogenously remains unclear. Using long-read based 5′ RNA ligase-mediated rapid amplification of cDNA ends (5′ Repeat-RLM-RACE), we identified C9orf72 transcripts initiating within intron 1 in a C9BAC mouse model, patient-derived iNeurons, and iNeuron-derived polysomes. These cryptic m7G-capped mRNAs are at least partially polyadenylated and are more abundant than transcripts derived from intron retention or circular intron lariats. In RAN translation reporter assays, intronic template transcripts–even those with short (32 nucleotide) leaders–exhibited robust expression compared to exon–intron and repeat-containing lariat reporters. To assess endogenous repeat-containing lariat RNA contributions to RAN translation, we enhanced endogenous lariat stability by knocking down the lariat debranching enzyme Dbr1. However, this modulation did not impact DPR production in patient-derived iNeurons. These findings identify cryptic, linear, m7G-capped intron-initiating C9orf72 mRNAs as an endogenous template for RAN translation and DPR production, with implications for disease pathogenesis and therapeutic development.},
  note = {{PMID:} 40758885}
}

MELK as a Mediator of Stemness and Metastasis in Aggressive Subtypes of Breast Cancer

@article{McBean2025MELKStemnessMetastasis,
  author = {McBean, Breanna and Abou Zeidane, Reine and Lichtman-Mikol, Samuel and Hauk, Benjamin and Speers, Johnathan and Tidmore, Savannah and Flores, Citlally Lopez and Rana, Priyanka S. and Pisano, Courtney and Liu, Meilan and Santola, Alyssa and Montero, Alberto and Boyle, Alan P. and Speers, Corey W.},
  title = {{MELK as a Mediator of Stemness and Metastasis in Aggressive Subtypes of Breast Cancer}},
  journal = {International Journal of Molecular Sciences},
  volume = {26},
  year = {2025},
  number = {5},
  monnth = {Mar},
  pages = {2245},
  url = {https://www.mdpi.com/1422-0067/26/5/2245},
  pdf = {http://boylelab.org/pubs/IJMS_2025_McBean.pdf},
  abstract = {Triple-negative breast cancer (TNBC) is the breast cancer subtype with the poorest prognosis and lacks actionable molecular targets for treatment. Maternal embryonic leucine zipper kinase (MELK) is highly expressed in TNBC and has been implicated in poor clinical outcomes, though its mechanistic role in the aggressive biology of TNBC is poorly understood. Here, we demonstrate a role of MELK in TNBC progression and metastasis. Analysis of publicly available datasets revealed that high MELK expression correlates with worse overall survival, recurrence-free survival, and distant metastasis-free survival, and MELK is co-expressed with metastasis-related genes. Functional studies demonstrated that MELK inhibition, using genomic or pharmacologic inhibition, reduces mammosphere formation, migration, and invasion in high-MELK-expressing TNBC cell lines. Conversely, MELK overexpression in low-MELK-expressing cell lines significantly increased invasive capacity in vitro and metastatic potential in vivo, as evidenced by enhanced metastasis to the liver and lungs in a chorioallantoic membrane assay. These findings highlight MELK as a key regulator of TNBC aggressiveness and support its potential as a therapeutic target to mitigate metastasis and improve patient outcomes.},
  doi = {10.3390/ijms26052245},
  note = {{PMID:} 40076867}
}

Massively parallel reporter assay investigates shared genetic variants of eight psychiatric disorders

@article{Lee2025PsychiatricMPRA,
  author = {Sool Lee and Jessica C McAfee and Jiseok Lee and Alejandro Gomez and Austin T Ledford and Declan Clarke and Hyunggyu Min and Mark B Gerstein and Alan P Boyle and Patrick F Sullivan and Adriana S Beltran and Hyejung Won},
  title = {Massively parallel reporter assay investigates shared genetic variants of eight psychiatric disorders},
  journal = {Cell},
  volume = {S0092-8674},
  number = {24},
  pages = {01435-1},
  year = {2025},
  month = {Jan},
  abstract = {A meta-genome-wide association study across eight psychiatric disorders has highlighted the genetic architecture of pleiotropy in major psychiatric disorders. However, mechanisms underlying pleiotropic effects of the associated variants remain to be explored. We conducted a massively parallel reporter assay to decode the regulatory logic of variants with pleiotropic and disorder-specific effects. Pleiotropic variants differ from disorder-specific variants by exhibiting chromatin accessibility that extends across diverse cell types in the neuronal lineage and by altering motifs for transcription factors with higher connectivity in protein-protein interaction networks. We mapped pleiotropic and disorder-specific variants to putative target genes using functional genomics approaches and CRISPR perturbation. In vivo CRISPR perturbation of a pleiotropic and a disorder-specific gene suggests that pleiotropy may involve the regulation of genes expressed broadly across neuronal cell types and with higher network connectivity.},
  doi = {10.1016/j.cell.2024.12.022},
  url = {https://www.sciencedirect.com/science/article/pii/S0092867424014351},
  pdf = {http://boylelab.org/pubs/Cell_2025_Lee.pdf},
  note = {{PMID:} 39848247}
}

A personalized multi-platform assessment of somatic mosaicism in the human frontal cortex

@article{Zhou2024PersonalizedSomaticMosaicism,
  author = {*Zhou, Weichen and *Mumm, Camille and *Gan, Yanming and Switzenberg, Jessica A. and Wang, Jinhao and De Oliveira, Paulo and Kathuria, Kunal and Losh, Steven J. and McDonald, Torrin L. and Bessell, Brandt and Van Deynze, Kinsey and {\dag}McConnell, Michael J. and {\dag}Boyle, Alan P. and {\dag}Mills, Ryan E.},
  title = {A personalized multi-platform assessment of somatic mosaicism in the human frontal cortex},
  year = {2024},
  doi = {10.1101/2024.12.18.629274},
  abstract = {Somatic mutations in individual cells lead to genomic mosaicism, contributing to the intricate regulatory landscape of genetic disorders and cancers. To evaluate and refine the detection of somatic mosaicism across different technologies with personalized donor-specific assembly (DSA), we obtained tissue from the dorsolateral prefrontal cortex (DLPFC) of a post-mortem neurotypical 31-year-old individual.We sequenced bulk DLPFC tissue using Oxford Nanopore Technologies (\~{}60X), NovaSeq (\~{}30X), and linked-read sequencing (\~{}28X). Additionally, we applied Cas9 capture methodology coupled with long-read sequencing (TEnCATS), targeting active transposable elements. We also isolated and amplified DNA from flow-sorted single DLPFC neurons using MALBAC, sequencing 115 of these MALBAC libraries on Nanopore and 94 on NovaSeq.We constructed a haplotype-resolved assembly with a total length of 5.77 Gb and a phase block length of 2.67 Mb (N50) to facilitate cross-platform analysis of somatic genetic variations. We observed an increase in the phasing rate from 11.6\% to 38.0\% between short-read and long-read technologies. By generating a catalog of phased germline SNVs, CNVs, and TEs from the assembled genome, we applied standard approaches to recall these variants across sequencing technologies. We achieved aggregated recall rates from 97.3\% to 99.4\% based on long-read bulk tissue data, setting an upper bound for detection limits.Moreover, utilizing haplotype-based analysis from DSA, we achieved a remarkable reduction in false positive somatic calls in bulk tissue, ranging from 14.9\% to 72.4\%. We developed pipelines leveraging DSA information to enhance somatic large genetic variant calling in long-read single cells. By examining somatic variation using long-reads in 115 individual neurons, we identified 468 candidate somatic heterozygous large deletions (1.5Mb - 20Mb), 137 of which intersected with short-read single-cell data. Additionally, we identified 61 putative somatic TEs (60 Alus, one LINE-1) in the single-cell data.Collectively, our analysis spans personalized assembly to single-cell somatic variant calling, providing a comprehensive ab initio ad finem approach and resource in real human tissue.Competing Interest StatementThe authors have declared no competing interest.},
  url = {https://www.biorxiv.org/content/early/2024/12/21/2024.12.18.629274},
  eprint = {https://www.biorxiv.org/content/early/2024/12/21/2024.12.18.629274.full.pdf},
  journal = {bioRxiv}
}

AAGGG repeat expansions trigger RFC1-independent synaptic dysregulation in human CANVAS neurons

@article{Maltby2024CANVASNeurons,
  title = {{AAGGG repeat expansions trigger RFC1-independent synaptic dysregulation in human CANVAS neurons}},
  author = {Maltby, Connor J. and Krans, Amy and Grudzien, Samantha J. and Palacios, Yomira and Mui{\~n}os, Jessica and Su{\'a}rez, Andrea and Asher, Melissa and Willey, Sydney and Van Deynze, Kinsey and Mumm, Camille and Boyle, Alan P and Cortese, Andrea and Khurana, Vikram and Barmada, Sami J. and Dijkstra, Anke A. and Todd, Peter K.},
  year = {2024},
  month = {Sep},
  day = {6},
  volume = {10},
  number = {36},
  pages = {eadn2321},
  doi = {10.1126/sciadv.adn2321},
  abstract = {Cerebellar ataxia with neuropathy and vestibular areflexia syndrome (CANVAS) is a late onset, recessively inherited neurodegenerative disorder caused by biallelic, non-reference pentameric AAGGG(CCCTT) repeat expansions within the second intron of replication factor complex subunit 1 (RFC1). To investigate how these repeats cause disease, we generated CANVAS patient induced pluripotent stem cell (iPSC) derived neurons (iNeurons) and utilized calcium imaging and transcriptomic analysis to define repeat-elicited gain-of-function and loss-of-function contributions to neuronal toxicity. AAGGG repeat expansions do not alter neuronal RFC1 splicing, expression, or DNA repair pathway functions. In reporter assays, AAGGG repeats are translated into pentapeptide repeat proteins that selectively accumulate in CANVAS patient brains. However, neither these proteins nor repeat RNA foci were detected in iNeurons, and overexpression of these repeats in isolation did not induce neuronal toxicity. CANVAS iNeurons exhibit defects in neuronal development and diminished synaptic connectivity that is rescued by CRISPR deletion of a single expanded allele. These phenotypic deficits were not replicated by knockdown of RFC1 in control neurons and were not rescued by ectopic expression of RFC1. These findings support a repeat-dependent but RFC1-independent cause of neuronal dysfunction in CANVAS, with important implications for therapeutic development in this currently untreatable condition.Summary Human CANVAS neurons exhibit transcriptional and functional synaptic defects that are corrected by heterozygous repeat deletion but are independent of the gene within which they reside{\textemdash}RFC1.Competing Interest StatementThe authors declare no direct conflicts of interest related to the content of this manuscript. No commercial forces had editorial or supervisory input on the content of the manuscript or its figures. P.K.T. holds a shared patent on ASOs with Ionis Pharmaceuticals. He has served as a consultant with Denali Therapeutics, and he has licensed technology and antibodies to Denali and Abcam. V.K. is a co-founder of and senior advisor to DaCapo Brainscience and Yumanity Therapeutics, companies focused on CNS diseases.},
  url = {https://www.science.org/doi/10.1126/sciadv.adn2321},
  pdf = {http://boylelab.org/pubs/Sci_Advances_2024_Maltby.pdf},
  journal = {Science Advances},
  note = {{PMID:} 39231235}
}

Deciphering the impact of genomic variation on function

@article{IGVFConsortium2024GenomicVariationFunction,
  title = {{Deciphering the impact of genomic variation on function}},
  author = {{IGVF Consortium}},
  year = {2024},
  month = {Sep},
  day = {4},
  volume = {633},
  issue = {8028},
  pages = {47--57},
  doi = {10.1038/s41586-024-07510-0},
  abstract = {Our genomes influence nearly every aspect of human biology—from molecular and cellular functions to phenotypes in health and disease. Studying the differences in DNA sequence between individuals (genomic variation) could reveal previously unknown mechanisms of human biology, uncover the basis of genetic predispositions to diseases, and guide the development of new diagnostic tools and therapeutic agents. Yet, understanding how genomic variation alters genome function to influence phenotype has proved challenging. To unlock these insights, we need a systematic and comprehensive catalogue of genome function and the molecular and cellular effects of genomic variants. Towards this goal, the Impact of Genomic Variation on Function (IGVF) Consortium will combine approaches in single-cell mapping, genomic perturbations and predictive modelling to investigate the relationships among genomic variation, genome function and phenotypes. IGVF will create maps across hundreds of cell types and states describing how coding variants alter protein activity, how noncoding variants change the regulation of gene expression, and how such effects connect through gene-regulatory and protein-interaction networks. These experimental data, computational predictions and accompanying standards and pipelines will be integrated into an open resource that will catalyse community efforts to explore how our genomes influence biology and disease across populations.},
  url = {https://www.nature.com/articles/s41586-024-07510-0},
  pdf = {http://boylelab.org/pubs/Nature_2024_IGVF.pdf},
  journal = {Nature},
  note = {{PMID:} 39232149}
}

An activity-regulated transcriptional program directly drives synaptogenesis

@article{Yee2024ActivityRegulatedSynaptogenesis,
  author = {Yee, C and Xiao, Y and Chen, H and Reddy, AR and Xu, B and Medwig-Kinney, TN and Zhang, W and Boyle, Alan P. and Herbst, WA and Xiang, YK and Matus, DQ and Shen, K},
  title = {{An activity-regulated transcriptional program directly drives synaptogenesis}},
  year = {2024},
  month = {Aug},
  volume = {27},
  number = {9},
  pages = {1695-1707},
  doi = {10.1038/s41593-024-01728-x},
  abstract = {Although the molecular composition and architecture of synapses have been widely explored, much less is known about what genetic programs directly activate synaptic gene expression and how they are modulated. Here, using Caenorhabditis elegans dopaminergic neurons, we reveal that EGL-43/MECOM and FOS-1/FOS control an activity-dependent synaptogenesis program. Loss of either factor severely reduces presynaptic protein expression. Both factors bind directly to promoters of synaptic genes and act together with CUT homeobox transcription factors to activate transcription. egl-43 and fos-1 mutually promote each other's expression, and increasing the binding affinity of FOS-1 to the egl-43 locus results in increased presynaptic protein expression and synaptic function. EGL-43 regulates the expression of multiple transcription factors, including activity-regulated factors and developmental factors that define multiple aspects of dopaminergic identity. Together, we describe a robust genetic program underlying activity-regulated synapse formation during development.},
  url = {https://www.nature.com/articles/s41593-024-01728-x},
  pdf = {http://boylelab.org/pubs/Nat_Neuroscience_2024_Yee.pdf},
  journal = {Nature Neuroscience},
  note = {{PMID:} 39103556}
}

Enhancing Portability of Trans-Ancestral Polygenic Risk Scores through Tissue-Specific Functional Genomic Data Integration

@article{Crone2024TransAncestralPRS,
  author = {Crone, Bradley and Boyle, Alan P},
  title = {Enhancing Portability of Trans-Ancestral Polygenic Risk Scores through Tissue-Specific Functional Genomic Data Integration},
  year = {2024},
  month = {aug},
  day = {7},
  volume = {20},
  number = {8},
  pages = {e1011356},
  doi = {10.1371/journal.pgen.1011356},
  abstract = {Portability of trans-ancestral polygenic risk scores is often confounded by differences in linkage disequilibrium and genetic architecture between ancestries. Recent literature has shown that prioritizing GWAS SNPs with functional genomic evidence over strong association signals can improve model portability. We leveraged three RegulomeDB-derived functional regulatory annotations-SURF, TURF, and TLand-to construct polygenic risk models across a set of quantitative and binary traits highlighting functional mutations tagged by trait-associated tissue annotations. Tissue-specific prioritization by TURF and TLand provide a significant improvement in model accuracy over standard polygenic risk score (PRS) models across all traits. We developed the Trans-ancestral Iterative Tissue Refinement (TITR) algorithm to construct PRS models that prioritize functional mutations across multiple trait-implicated tissues. TITR-constructed PRS models show increased predictive accuracy over single tissue prioritization. This indicates our TITR approach captures a more comprehensive view of regulatory systems across implicated tissues that contribute to variance in trait expression.},
  url = {https://journals.plos.org/plosgenetics/article?id=10.1371/journal.pgen.1011356},
  pdf = {http://boylelab.org/pubs/PLoS_Genetics_2024_Crone.pdf},
  journal = {PLoS Genetics},
  note = {{PMID:} 39110742}
}

HaplotagLR: an efficient and configurable utility for haplotagging long reads

@article{Holmes2024HaplotagLR,
  author = {Holmes, Monica J. and Mahjour, Babak and Castro, Christopher P. and Farnum, Gregory A. and Diehl, Adam G. and Boyle, Alan P.},
  title = {{HaplotagLR: an efficient and configurable utility for haplotagging long reads}},
  year = {2024},
  month = {03},
  volume = {19},
  pages = {1-15},
  number = {3},
  doi = {10.1371/journal.pone.0298688},
  abstract = {Understanding the functional effects of sequence variation is crucial in genomics. Individual human genomes contain millions of variants that contribute to phenotypic variability and disease risks at the population level. Because variants rarely act in isolation, we must consider potential interactions of neighboring variants to accurately predict functional effects. We can accomplish this using haplotagging, which matches sequencing reads to their parental haplotypes using alleles observed at known heterozygous variants. However, few published tools for haplotagging exist and these share several technical and usability-related shortcomings that limit applicability, in particular a lack of insight or control over error rates, and lack of key metrics on the underlying sources of haplotagging error. Here we present HaplotagLR: a user-friendly tool that haplotags long sequencing reads based on a multinomial model and existing phased variant lists. HaplotagLR is user-configurable and includes a basic error model to control the empirical FDR in its output. We show that HaplotagLR outperforms the leading haplotagging method in simulated datasets, especially at high levels of specificity, and displays 7% greater sensitivity in haplotagging real data. HaplotagLR advances both the immediate utility of haplotagging and paves the way for further improvements to this important method.},
  url = {https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0298688},
  pdf = {http://boylelab.org/pubs/PLoS_One_2024_Holmes.pdf},
  journal = {{PLoS ONE}},
  note = {{PMID:} 38478504}
}

CAGI, the Critical Assessment of Genome Interpretation, establishes progress and prospects for computational genetic variant interpretation methods.

@article{CAGIConsortium2024VariantInterpretation,
  title = {{CAGI, the Critical Assessment of Genome Interpretation, establishes progress and prospects for computational genetic variant interpretation methods.}},
  author = {{The Critical Assessment of Genome Interpretation Consortium}},
  year = {2024},
  month = {02},
  volume = {25},
  pages = {53},
  number = {1},
  doi = {10.1186/s13059-023-03113-6},
  abstract = {Background: The Critical Assessment of Genome Interpretation (CAGI) aims to advance the state-of-the-art for computational prediction of genetic variant impact, particularly where relevant to disease. The five complete editions of the CAGI community experiment comprised 50 challenges, in which participants made blind predictions of phenotypes from genetic data, and these were evaluated by independent assessors. Results: Performance was particularly strong for clinical pathogenic variants, including some difficult-to-diagnose cases, and extends to interpretation of cancer-related variants. Missense variant interpretation methods were able to estimate biochemical effects with increasing accuracy. Assessment of methods for regulatory variants and complex trait disease risk was less definitive and indicates performance potentially suitable for auxiliary use in the clinic. Conclusions: Results show that while current methods are imperfect, they have major utility for research and clinical applications. Emerging methods and increasingly large, robust datasets for training and assessment promise further progress ahead.},
  url = {https://genomebiology.biomedcentral.com/articles/10.1186/s13059-023-03113-6},
  pdf = {http://boylelab.org/pubs/Genome_Biology_2024_CAGI.pdf},
  journal = {Genome Biology},
  note = {{PMID:} 38389099}
}

Systematic investigation of allelic regulatory activity of schizophrenia-associated common variants

@article{McAfee2023SchizophreniaRegulatoryVariants,
  author = {Jessica C. McAfee and Sool Lee and Jiseok Lee and Jessica L. Bell and Oleh Krupa and Jessica Davis and Kimberly Insigne and Marielle L. Bond and Zhao, Nanxiang and Alan P Boyle and Douglas H. Phanstiel and Michael I. Love and Jason L. Stein and W. Brad Ruzicka and Jose Davila-Velderrain and Sriram Kosuri and Hyejung Won},
  title = {Systematic investigation of allelic regulatory activity of schizophrenia-associated common variants},
  year = {2023},
  month = {oct},
  day = {11},
  volume = {3},
  pages = {100404},
  doi = {10.1016/j.xgen.2023.100404},
  abstract = {Genome-wide association studies (GWAS) have successfully identified 145 genomic regions that contribute to schizophrenia risk, but linkage disequilibrium (LD) makes it challenging to discern causal variants. Computational finemapping prioritized thousands of credible variants, \~{}98\% of which lie within poorly characterized non-coding regions. To functionally validate their regulatory effects, we performed a massively parallel reporter assay (MPRA) on 5,173 finemapped schizophrenia GWAS variants in primary human neural progenitors (HNPs). We identified 439 variants with allelic regulatory effects (MPRA-positive variants), with 71\% of GWAS loci containing at least one MPRA-positive variant. Transcription factor binding had modest predictive power for predicting the allelic activity of MPRA-positive variants, while GWAS association, finemap posterior probability, enhancer overlap, and evolutionary conservation failed to predict MPRA-positive variants. Furthermore, 64\% of MPRA-positive variants did not exhibit eQTL signature, suggesting that MPRA could identify yet unexplored variants with regulatory potentials. MPRA-positive variants differed from eQTLs, as they were more frequently located in distal neuronal enhancers. Therefore, we leveraged neuronal 3D chromatin architecture to identify 272 genes that physically interact with MPRA-positive variants. These genes annotated by chromatin interactome displayed higher mutational constraints and regulatory complexity than genes annotated by eQTLs, recapitulating a recent finding that eQTL- and GWAS-detected variants map to genes with different properties. Finally, we propose a model in which allelic activity of multiple variants within a GWAS locus can be aggregated to predict gene expression by taking chromatin contact frequency and accessibility into account. In conclusion, we demonstrate that MPRA can effectively identify functional regulatory variants and delineate previously unknown regulatory principles of schizophrenia.Competing Interest StatementThe authors have declared no competing interest.Funding StatementThis research was supported by the PsychENCODE consortium (U01MH122509, H.W. and J.L.S.), the IGVF consortium (UM1HG012003, H.W. and M.I.L.), the National Institute of General Medical Sciences (5T32GM067553, S.L; 5T32GM135128, J.C.M. and M.B.), the NIH New Innovator Award from the National Institute of Mental Health (DP2MH122403, H.W.), and the NARSAD Young Investigator Award from the Brain and Behavior Research Foundation (H.W.).Author DeclarationsI confirm all relevant ethical guidelines have been followed, and any necessary IRB and/or ethics committee approvals have been obtained.YesThe details of the IRB/oversight body that provided approval or exemption for the research described are given below:Human neural progenitors were acquired from the human fetal tissue obtained from the UCLA Gene and Cell Therapy Core according to IRB guidelines following voluntary termination of pregnancy. This study was performed under the auspices of the UCLA Office of Human Research Protection, which determined that it was exempt because samples are anonymous pathological specimens. Full informed consent was obtained from all of the parent donors.I confirm that all necessary patient/participant consent has been obtained and the appropriate institutional forms have been archived, and that any patient/participant/sample identifiers included were not known to anyone (e.g., hospital staff, patients or participants themselves) outside the research group so cannot be used to identify individuals.YesI understand that all clinical trials and any other prospective interventional studies must be registered with an ICMJE-approved registry, such as ClinicalTrials.gov. I confirm that any such study reported in the manuscript has been registered and the trial registration ID is provided (note: if posting a prospective study registered retrospectively, please provide a statement in the trial ID field explaining why the study was not registered in advance).YesI have followed all appropriate research reporting guidelines and uploaded the relevant EQUATOR Network research reporting checklist(s) and other pertinent material as supplementary files, if applicable.YesSequencing data are available via the Gene Expression Omnibus under the accession number GSE211045.},
  url = {https://www.cell.com/cell-genomics/fulltext/S2666-979X(23)00218-5},
  pdf = {http://boylelab.org/pubs/Cell_Genomics_2023_McAfee.pdf},
  journal = {Cell Genomics},
  note = {{PMID:} 37868037}
}

Organ-specific prioritization and annotation of non-coding regulatory variants in the human genome

@article{Zhao2023OrganSpecificRegulatoryVariants,
  author = {Zhao, Nanxiang and Dong, Shengcheng and Boyle, Alan P},
  title = {Organ-specific prioritization and annotation of non-coding regulatory variants in the human genome},
  year = {2023},
  doi = {10.1101/2023.09.07.556700},
  abstract = {{Identifying non-coding regulatory variants in the human genome remains a challenging task in genomics. Recently we advanced our leading regulatory variant database, RegulomeDB, to its second version. Building upon this comprehensive database, we developed a novel machine-learning architecture with stacked generalization, TLand, which utilizes RegulomeDB-derived features to predict regulatory variants at cell or organ-specific levels. In our holdout benchmarking, TLand consistently outperformed state-of-the-art models, demonstrating its ability to generalize to new cell lines or organs. We trained three types of organ-specific TLand models to overcome the common model bias toward high data availability cell lines or organs. These models accurately prioritize relevant organs for 2 million GWAS SNPs associated with GWAS traits. Moreover, our analysis of top-scoring variants in specific organ models showed a high enrichment of relevant GWAS traits. We expect that TLand and RegulomeDB will further advance our ability to understand human regulatory variants genome-wide.Competing Interest StatementThe authors have declared no competing interest.}},
  url = {https://www.biorxiv.org/content/early/2023/09/08/2023.09.07.556700},
  pdf = {https://www.biorxiv.org/content/early/2023/09/08/2023.09.07.556700.full.pdf},
  journal = {bioRxiv}
}

Sperm chromatin structure and reproductive fitness are altered by substitution of a single amino acid in mouse protamine 1

@article{Moritz2023ProtamineChromatin,
  author = {Moritz, Lindsay and Schon, Samantha B. and Rabbani, Mashiat and Sheng, Yi and Agrawal, Ritvija and Glass-Klaiber, Juniper and Sultan, Caleb and Camarillo Jeannie M and Clements, Jourdan and Baldwin, Michael R and Diehl, Adam G and Boyle, Alan P. and O'Brien, Patrick J and Ragunathan, Kaushik and Hu, Yueh-Chiang and Kelleher, Neil L and Nandakumar, Jayakrishnan and Li, Jun Z. and Orwig, Kyle E. and Redding, Sy and Hammoud, Saher Sue},
  title = {Sperm chromatin structure and reproductive fitness are altered by substitution of a single amino acid in mouse protamine 1},
  year = {2023},
  month = {jul},
  day = {17},
  doi = {10.1038/s41594-023-01033-4},
  abstract = {Conventional dogma presumes that protamine-mediated DNA compaction in sperm is achieved by electrostatic interactions between DNA and the arginine-rich core of protamines. Phylogenetic analysis reveals several non-arginine residues conserved within, but not across species. The significance of these residues and their post-translational modifications are poorly understood. Here, we investigated the role of K49, a rodent-specific lysine residue in protamine 1 (P1) that is acetylated early in spermiogenesis and retained in sperm. In sperm, alanine substitution (P1(K49A)) decreases sperm motility and male fertility-defects that are not rescued by arginine substitution (P1(K49R)). In zygotes, P1(K49A) leads to premature male pronuclear decompaction, altered DNA replication, and embryonic arrest. In vitro, P1(K49A) decreases protamine-DNA binding and alters DNA compaction and decompaction kinetics. Hence, a single amino acid substitution outside the P1 arginine core is sufficient to profoundly alter protein function and developmental outcomes, suggesting that protamine non-arginine residues are essential for reproductive fitness.},
  url = {https://www.nature.com/articles/s41594-023-01033-4},
  pdf = {http://boylelab.org/pubs/Nature_Structural_2023_Moritz.pdf},
  journal = {Nature Structural \& Molecular Biology},
  note = {{PMID:} 37460896}
}

OnRamp: rapid nanopore plasmid validation

@article{Mumm2023OnRamp,
  author = {*Mumm, Camille and *Drexel, Melissa L and McDonald, Torrin L and Diehl, Adam G and Switzenberg, Jessica A and Boyle, Alan P},
  title = {{OnRamp: rapid nanopore plasmid validation}},
  year = {2023},
  month = {may},
  day = {8},
  volume = {33},
  number = {5},
  pages = {741--749},
  doi = {10.1101/gr.277369.122},
  abstract = {Recombinant plasmid vectors are versatile tools that have facilitated discoveries in molecular biology, genetics, proteomics, and many other fields. As the enzymatic and bacterial processes used to create recombinant DNA can introduce errors, sequence validation is an essential step in plasmid assembly. Sanger sequencing is the current standard for plasmid validation; however, this method is limited by an inability to sequence through complex secondary structure, and lacks scalability when applied to full-plasmid sequencing of multiple plasmids due to read-length limits. While high-throughput sequencing does provide full-plasmid sequencing at scale, it is impractical and costly when utilized outside of library-scale validation. Here we present OnRamp (Oxford nanopore-based Rapid Analysis of Multiplexed Plasmids), an alternative method for routine plasmid validation which combines the advantages of high-throughput sequencing's full-plasmid coverage and scalability with Sanger's affordability and accessibility by leveraging nanopore's long-read sequencing technology. We include customized wet-lab protocols for plasmid preparation along with a pipeline designed for analysis of read data obtained using these protocols. This analysis pipeline is deployed on the OnRamp web app, which generates alignments between actual and predicted plasmid sequences, quality scores, and read-level views. OnRamp is designed to be broadly accessible regardless of programming experience to facilitate more widespread adoption of long-read sequencing for routine plasmid validation. Here we describe the OnRamp protocols and pipeline and demonstrate our ability to obtain full sequences from pooled plasmids while detecting sequence variation even in regions of high secondary structure at less than half the cost of equivalent Sanger sequencing.},
  url = {https://genome.cshlp.org/content/early/2023/05/08/gr.277369.122},
  pdf = {http://boylelab.org/pubs/Genome_Research_2023_Mumm.pdf},
  journal = {Genome Research},
  note = {{PMID:} 37156622}
}

Challenges in screening for de novo noncoding variants contributing to genetically complex phenotypes

@article{Castro2023DeNovoNoncodingVariants,
  author = {Castro, Christopher P. and Diehl, Adam G. and Boyle, Alan P.},
  title = {Challenges in screening for de novo noncoding variants contributing to genetically complex phenotypes},
  year = {2023},
  month = {may},
  day = {20},
  volume = {4},
  number = {3},
  pages = {100210},
  doi = {10.1016/j.xhgg.2023.100210},
  abstract = {{Understanding the genetic basis for complex, heterogeneous disorders, such as autism spectrum disorder (ASD), is a persistent challenge in human medicine. Owing to their phenotypic complexity, the genetic mechanisms underlying these disorders may be highly variable across individual patients. Furthermore, much of their heritability is unexplained by known regulatory or coding variants. Indeed, there is evidence that much of the causal genetic variation stems from rare and de novo variants arising from ongoing mutation. These variants occur mostly in noncoding regions, likely affecting regulatory processes for genes linked to the phenotype of interest. However, because there is no uniform code for assessing regulatory function, it is difficult to separate these mutations into likely functional and nonfunctional subsets. This makes finding associations between complex diseases and potentially causal de novo single-nucleotide variants (dnSNVs) a difficult task. To date, most published studies have struggled to find any significant associations between dnSNVs from ASD patients and any class of known regulatory elements. We sought to identify the underlying reasons for this and present strategies for overcoming these challenges. We show that, contrary to previous claims, the main reason for failure to find robust statistical enrichments is not only the number of families sampled, but also the quality and relevance to ASD of the annotations used to prioritize dnSNVs, and the reliability of the set of dnSNVs itself. We present a list of recommendations for designing future studies of this sort that will help researchers avoid common pitfalls.}},
  url = {https://www.sciencedirect.com/science/article/pii/S2666247723000428},
  pdf = {http://boylelab.org/pubs/HGG_Advances_2023_Castro.pdf},
  journal = {Human Genetics and Genomics Advances},
  note = {{PMID:} 37305558}
}

Annotating and prioritizing human non-coding variants with RegulomeDB v.2

@article{Dong2023RegulomeDB2,
  author = {*Dong, Shengcheng and *Zhao, Nanxiang and Spragins, Emma and Kagda, Meenakshi S and Li, Mingjie and Assis, Pedro R and Jolanki, Otto and Luo, Yunhai and Cherry, J Michael and {\dag}Boyle, Alan P and {\dag}Hitz, Benjamin C},
  title = {{Annotating and prioritizing human non-coding variants with RegulomeDB v.2}},
  year = {2023},
  month = {apr},
  day = {25},
  volume = {55},
  number = {5},
  pages = {724--726},
  doi = {10.1038/s41588-023-01365-3},
  abstract = {Nearly 90\% of the disease risk-associated variants identified from genome-wide association studies (GWAS) are in non-coding regions of the genome. The annotations obtained from analyzing functional genomics assays can provide additional information to pinpoint causal variants, which are often not the lead variants identified from association studies. However, the lack of available annotation tools limits the use of such data. To address the challenge, we have previously built the RegulomeDB database for prioritizing and annotating variants in non-coding regions1, which has been a highly utilized resource for the research community (Supplementary Fig. 1). RegulomeDB annotates a variant by intersecting its position with genomic intervals identified from functional genomic assays and computational approaches. It also incorporates those hits of a variant into a heuristic ranking score, representing its potential to be functional in regulatory elements. Here we present a newer version of the RegulomeDB web server, RegulomeDB v2.1 (http://regulomedb.org). We improve and boost annotation power by incorporating thousands of newly processed data from functional genomic assays in GRCh38 assembly, and now include probabilistic scores from the SURF algorithm that was the top performing non-coding variant predictor in CAGI 52. We also provide interactive charts and genome browser views to allow users an easy way to perform exploratory analyses in different tissue contexts.Competing Interest StatementThe authors have declared no competing interest.},
  url = {https://www.nature.com/articles/s41588-023-01365-3},
  pdf = {http://boylelab.org/pubs/Nature_Genetics_2023_Dong.pdf},
  journal = {Nature Genetics},
  note = {{PMID:} 37173523}
}

Explain-seq: an end-to-end pipeline from training to interpretation of sequence-based deep learning models

@article{Zhao2023ExplainSeq,
  author = {Zhao, Nanxiang and Wang, S and Huang, Q and Dong, Shengcheng and Boyle, Alan P.},
  title = {{Explain-seq: an end-to-end pipeline from training to interpretation of sequence-based deep learning models}},
  year = {2023},
  doi = {10.1101/2023.01.23.525250},
  url = {https://www.biorxiv.org/content/early/2023/01/23/2023.01.23.525250},
  pdf = {https://www.biorxiv.org/content/early/2023/01/23/2023.01.23.525250.full.pdf},
  journal = {bioRxiv}
}

Quantitative assessment of association between noncoding variants and transcription factor binding

@article{Ouyang2022NoncodingVariantTFBinding,
  author = {Ouyang, Ningxin and Boyle, Alan P.},
  title = {Quantitative assessment of association between noncoding variants and transcription factor binding},
  year = {2022},
  doi = {10.1101/2022.11.22.517559},
  abstract = {{Association fine-mapping of molecular traits is an essential method for understanding the impact of genetic variation. Sequencing-based assays, including RNA-seq, DNase-seq and ChIP-seq, have been widely used to measure different cellular traits and enabled genome-wide mapping of quantitative trait loci (QTLs). The disruption of cis-regulatory sequence, often occurring through variation within transcription factor binding motifs, has been strongly associated with gene dysregulation and human disease. We recently developed a computational method, TRACE, for transcription factor binding footprint prediction. TRACE integrates chromatin accessibility and transcription factor binding motifs to produce quantitative scores that describe the binding affinity of a TF for a specific TFBS locus. Here we have extended this method to incorporate variant data for 57 Yoruban individuals. Using genome-wide chromatin-accessibility data and human TF binding motifs, we have generated precise, genome-wide predictions of individual-specific transcription factor binding footprints. Subsequent association mapping between these footprints and nearby regulatory variants yielded numerous footprint-variant pairs with significant evidence for correlation, which we call footprint-QTLs (fpQTLs). fpQTLs appear to affect TF binding in a distance-dependent manner and share significant overlap with known dsQTLs and eQTLs. fpQTLs provide a rich resource for the study of regulatory variants, both within and outside known TFBSs, leading to improved functional interpretation of noncoding variation.Competing Interest StatementThe authors have declared no competing interest.}},
  url = {https://www.biorxiv.org/content/early/2022/11/23/2022.11.22.517559},
  pdf = {https://www.biorxiv.org/content/early/2022/11/23/2022.11.22.517559.full.pdf},
  journal = {bioRxiv}
}

SEMplMe: A tool for integrating DNA methylation effects in transcription factor binding affinity predictions

@article{Nishizaki2022SEMplMe,
  author = {Nishizaki, Sierra S and Boyle, Alan P},
  title = {{SEMplMe: A tool for integrating DNA methylation effects in transcription factor binding affinity predictions}},
  year = {2022},
  month = {aug},
  day = {4},
  volume = {23},
  number = {1},
  pages = {317},
  doi = {10.1186/s12859-022-04865-x},
  url = {https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-022-04865-x},
  pdf = {http://boylelab.org/pubs/BMC_Bioinformatics_2022_Nishizaki.pdf},
  abstract = {Aberrant DNA methylation in transcription factor binding sites has been shown to lead to anomalous gene regulation that is strongly associated with human disease. However, the majority of methylation-sensitive positions within transcription factor binding sites remain unknown. Here we introduce SEMplMe, a computational tool to generate predictions of the effect of methylation on transcription factor binding strength in every position within a transcription factor’s motif.},
  journal = {BMC Bioinformatics},
  note = {{PMID:} 35927613}
}

Comprehensive enhancer-target gene assignments improve gene set level interpretation of genome-wide regulatory data

@article{Qin2022EnhancerTargetGenes,
  author = {Qin, Tingting and Lee, Christopher and Li, Shiting and Cavalcante, Raymond G and Orchard, Peter and Yao, Heming and Zhang, Hanrui and Wang, Shuze and Patil, Snehal and Boyle, Alan P and Sartor, Maureen A},
  title = {{Comprehensive enhancer-target gene assignments improve gene set level interpretation of genome-wide regulatory data}},
  year = {2022},
  month = {apr},
  day = {26},
  volume = {23},
  number = {1},
  pages = {105},
  doi = {10.1186/s13059-022-02668-0},
  abstract = {{Revealing the gene targets of distal regulatory elements is challenging yet critical for interpreting regulome data. Experiment-derived enhancer-gene links are restricted to a small set of enhancers and/or cell types, while the accuracy of genome-wide approaches remains elusive due to the lack of a systematic evaluation. We combined multiple spatial and in silico approaches for defining enhancer locations and linking them to their target genes aggregated across >500 cell types, generating 1860 human genome-wide distal enhancer-to-target gene definitions (EnTDefs). To evaluate performance, we used gene set enrichment (GSE) testing on 87 independent ENCODE ChIP-seq datasets of 34 transcription factors (TFs) and assessed concordance of results with known TF Gene Ontology annotations, and other benchmarks.}},
  url = {https://doi.org/10.1186/s13059-022-02668-0},
  pdf = {http://boylelab.org/pubs/Genome_Biology_2022_Qin.pdf},
  journal = {Genome Biology},
  note = {{PMID:} 35473573}
}

Prioritization of regulatory variants with tissue-specific function in the non-coding regions of human genome

@article{Dong2021TissueSpecificRegulatoryVariants,
  author = {Dong, Shengcheng and Boyle, Alan P},
  title = {Prioritization of regulatory variants with tissue-specific function in the non-coding regions of human genome},
  journal = {Nucleic Acids Research},
  year = {2021},
  month = {10},
  volume = {50},
  number = {1},
  pages = {e6-e6},
  abstract = {Understanding the functional consequences of genetic variation in the non-coding regions of the human genome remains a challenge. We introduce h ere a computational tool, TURF, to prioritize regulatory variants with tissue-specific function by leveraging evidence from functional genomics experiments, including over 3000 functional genomics datasets from the ENCODE project provided in the RegulomeDB database. TURF is able to generate prediction scores at both organism and tissue/organ-specific levels for any non-coding variant on the genome. We present that TURF has an overall top performance in prediction by using validated variants from MPRA experiments. We also demonstrate how TURF can pick out the regulatory variants with tissue-specific function over a candidate list from associate studies. Furthermore, we found that various GWAS traits showed the enrichment of regulatory variants predicted by TURF scores in the trait-relevant organs, which indicates that these variants can be a valuable source for future studies.},
  doi = {10.1093/nar/gkab924},
  url = {https://doi.org/10.1093/nar/gkab924},
  pdf = {http://boylelab.org/pubs/NAR_2021_Dong.pdf},
  journal = {Nucleic Acids Research},
  note = {{PMID:} 34648033}
}

SquiggleNet: real-time, direct classification of nanopore signals

@article{Bao2021SquiggleNet,
  author = {Bao, Yuwei and Wadden, Jack and Erb-Downward, John R. and Ranjan, Piyush and Zhou, Weichen and McDonald, Torrin L and Mills, Ryan E and Boyle, Alan P and Dickson, Robert P. and Blaauw, David and Welch, Joshua D.},
  title = {{SquiggleNet: real-time, direct classification of nanopore signals}},
  year = {2021},
  month = {oct},
  day = {27},
  volume = {22},
  number = {1},
  pages = {298},
  doi = {10.1186/s13059-021-02511-y},
  abstract = {{We present SquiggleNet, the first deep-learning model that can classify nanopore reads directly from their electrical signals. SquiggleNet operates faster than DNA passes through the pore, allowing real-time classification and read ejection. Using 1 s of sequencing data, the classifier achieves significantly higher accuracy than base calling followed by sequence alignment. Our approach is also faster and requires an order of magnitude less memory than alignment-based approaches. SquiggleNet distinguished human from bacterial DNA with over 90% accuracy, generalized to unseen bacterial species in a human respiratory meta genome sample, and accurately classified sequences containing human long interspersed repeat elements.}},
  url = {https://genomebiology.biomedcentral.com/articles/10.1186/s13059-021-02511-y},
  pdf = {http://boylelab.org/pubs/Genome_Biology_2021_Bao.pdf},
  journal = {Genome Biology},
  note = {{PMID:} 34706748}
}

Cas9 targeted enrichment of mobile elements using nanopore sequencing

@article{McDonald2021Cas9MobileElements,
  author = {*McDonald, Torrin L. and *Zhou, Weichen and Castro, Christopher P. and Mumm, Camille and Switzenberg, Jessica A. and {\dag}Mills, Ryan E. and {\dag}Boyle, Alan P.},
  title = {Cas9 targeted enrichment of mobile elements using nanopore sequencing},
  year = {2021},
  month = {jun},
  day = {11},
  volume = {12},
  number = {1},
  pages = {3586},
  doi = {10.1038/s41467-021-23918-y},
  abstract = {{Mobile element insertions (MEIs) are repetitive genomic sequences that contribute to genetic variation and can lead to genetic disorders. Targeted and whole-genome approaches using short-read sequencing have been developed to identify reference and non-reference MEIs; however, the read length hampers detection of these elements in complex genomic regions. Here, we pair Cas9-targeted nanopore sequencing with computational methodologies to capture active MEIs in human genomes. We demonstrate parallel enrichment for distinct classes of MEIs, averaging 44% of reads on-targeted signals and exhibiting a 13.4-54x enrichment over whole-genome approaches. We show an individual flow cell can recover most MEIs (97% L1Hs, 93% AluYb, 51% AluYa, 99% SVA_F, and 65% SVA_E). We identify seventeen non-reference MEIs in GM12878 overlooked by modern, long-read analysis pipelines, primarily in repetitive genomic regions. This work introduces the utility of nanopore sequencing for MEI enrichment and lays the foundation for rapid discovery of elusive, repetitive genetic elements.}},
  url = {https://doi.org/10.1038/s41467-021-23918-y},
  pdf = {http://boylelab.org/pubs/Nature_Communications_2021_McDonald.pdf},
  journal = {Nature Communications},
  note = {{PMID:} 34117247}
}

F-Seq2: improving the feature density based peak caller with dynamic statistics

@article{Zhao2021FSeq2,
  author = {Zhao, Nanxiang and Boyle, Alan P},
  title = {{F-Seq2: improving the feature density based peak caller with dynamic statistics}},
  journal = {NAR Genomics and Bioinformatics},
  volume = {3},
  number = {1},
  year = {2021},
  month = {02},
  abstract = {{Genomic and epigenomic features are captured at a genome-wide level by using high-throughput sequencing (HTS) technologies. Peak calling delineates features identified in HTS experiments, such as open chromatin regions and transcription factor binding sites, by comparing the observed read distributions to a random expectation. Since its introduction, F-Seq has been widely used and shown to be the most sensitive and accurate peak caller for DNase I hypersensitive site (DNase-seq) data. However, the first release (F-Seq1) has two key limitations: lack of support for user-input control datasets, and poor test statistic reporting. These constrain its ability to capture systematic and experimental biases inherent to the background distributions in peak prediction, and to subsequently rank predicted peaks by confidence. To address these limitations, we present F-Seq2, which combines kernel density estimation and a dynamic ‘continuous’ Poisson test to account for local biases and accurately rank candidate peaks. The output of F-Seq2 is suitable for irreproducible discovery rate analysis as test statistics are calculated for individual candidate summits, allowing direct comparison of predictions across replicates. These improvements significantly boost the performance of F-Seq2 for ATAC-seq and ChIP-seq datasets, outperforming competing peak callers used by the ENCODE Consortium in terms of precision and recall.}},
  doi = {10.1093/nargab/lqab012},
  url = {https://doi.org/10.1093/nargab/lqab012},
  pdf = {http://boylelab.org/pubs/NARGB_2021_Zhao.pdf},
  note = {{PMID:} 33655209}
}

The inducible lac operator-repressor system is functional in zebrafish cells

@article{Nishizaki2021LacOperatorZebrafish,
  author = {*Nishizaki, Sierra S and *McDonald, Torrin L and Farnum, Gregory A and Holmes, Monica J and Drexel, Melissa L and Switzenberg, Jessica A and Boyle, Alan P},
  title = {The inducible lac operator-repressor system is functional in zebrafish cells},
  year = {2021},
  doi = {10.3389/fgene.2021.683394},
  url = {https://www.frontiersin.org/articles/10.3389/fgene.2021.683394/abstract},
  pdf = {http://boylelab.org/pubs/Frontiers_in_Genetics_2021_Nishizaki.pdf},
  abstract = {Background: Zebrafish are a foundational model organism for studying the spatio-temporal activity of genes and their regulatory sequences. A variety of approaches are currently available for editing genes and modifying gene expression in zebrafish, including RNAi, Cre/lox, and CRISPR-Cas9. However, the lac operator-repressor system, an E. coli lac operon component which has been adapted for use in many other species and is a valuable, flexible tool for inducible modulation of gene expression studies, has not been previously tested in zebrafish. Results: Here we demonstrate that the lac operator-repressor system robustly decreases expression of firefly luciferase in cultured zebrafish fibroblast cells. Our work establishes the lac operator-repressor system as a promising tool for the manipulation of gene expression in whole zebrafish. Conclusions: Our results lay the groundwork for the development of lac-based reporter assays in zebrafish, and adds to the tools available for investigating dynamic gene expression in embryogenesis. We believe this work will catalyze the development of new reporter assay systems to investigate uncharacterized regulatory elements and their cell-type specific activities.},
  journal = {Frontiers in Genetics},
  volume = {12},
  note = {{PMID:} 34220959}
}

Broad noncoding transcription suggests genome surveillance by RNA polymerase V

@article{Tsuzuki2020RNAPolymeraseV,
  author = {*Tsuzuki, Masayuki and *Sethuraman, Shriya and Coke, Adriana N. and Rothi, M. Hafiz and Boyle, Alan P. and Wierzbicki, Andrzej T.},
  title = {{Broad noncoding transcription suggests genome surveillance by RNA polymerase V}},
  year = {2020},
  volume = {117},
  doi = {10.1073/pnas.2014419117},
  number = {48},
  publisher = {National Academy of Sciences},
  pages = {30799--30804},
  abstract = {Eukaryotic genomes are pervasively transcribed, yet most transcribed sequences lack conservation or known biological functions. We show that a specialized plant-specific RNA polymerase V broadly transcribes the Arabidopsis genome. We propose a model where Pol V transcription surveils the genome and is required to recognize and repress newly inserted or reactivated transposons. Our results indicate that pervasive transcription of nonconserved sequences may serve an essential role in maintenance of genome integrity.Eukaryotic genomes are pervasively transcribed, yet most transcribed sequences lack conservation or known biological functions. In Arabidopsis thaliana, RNA polymerase V (Pol V) produces noncoding transcripts, which base pair with small interfering RNA (siRNA) and allow specific establishment of RNA-directed DNA methylation (RdDM) on transposable elements. Here, we show that Pol V transcribes much more broadly than previously expected, including subsets of both heterochromatic and euchromatic regions. At already established RdDM targets, Pol V and siRNA work together to maintain silencing. In contrast, some euchromatic sequences do not give rise to siRNA but are covered by low levels of Pol V transcription, which is needed to establish RdDM de novo if a transposon is reactivated. We propose a model where Pol V surveils the genome to make it competent to silence newly activated or integrated transposons. This indicates that pervasive transcription of nonconserved sequences may serve an essential role in maintenance of genome integrity.High-throughput sequencing data have been deposited in Gene Expression Omnibus (GEO) database (accession no. GSE146913).},
  issn = {0027-8424},
  url = {https://www.pnas.org/content/early/2020/11/11/2014419117},
  pdf = {http://boylelab.org/pubs/PNAS_2020_Tsuzuki.pdf},
  journal = {Proceedings of the National Academy of Sciences},
  note = {{PMID:} 33199612}
}

DNA methylation directs nucleosome positioning in RNA-mediated transcriptional silencing

@article{Rothi2020MethylationNucleosomePositioning,
  author = {Rothi, M Hafiz and Sethuraman, Shriya and Dolata, Jakub and Boyle, Alan P and Wierzbicki, Andrzej T},
  title = {{DNA methylation directs nucleosome positioning in RNA-mediated transcriptional silencing}},
  year = {2020},
  doi = {10.1101/2020.10.29.359794},
  publisher = {Cold Spring Harbor Laboratory},
  abstract = {{Repressive chromatin modifications are instrumental in regulation of gene expression and transposon silencing. In Arabidopsis thaliana, transcriptional silencing is performed by the RNA-directed DNA methylation (RdDM) pathway. In this process, two specialized RNA polymerases, Pol IV and Pol V, produce non-coding RNAs, which recruit several RNA-binding proteins and lead to the establishment of repressive chromatin marks. An important feature of chromatin is nucleosome positioning, which has also been implicated in RdDM. We show that RdDM affects nucleosomes via the SWI/SNF chromatin remodeling complex. This leads to the establishment of nucleosomes on methylated regions, which counteracts the general depletion of DNA methylation on nucleosomal regions. Nucleosome placement by RdDM has no detectable effects on the pattern of DNA methylation. Instead, DNA methylation by RdDM and other pathways affects nucleosome positioning. We propose a model where DNA methylation serves as one of the determinants of nucleosome positioning.Competing Interest StatementThe authors have declared no competing interest.}},
  url = {https://www.biorxiv.org/content/early/2020/10/29/2020.10.29.359794},
  pdf = {https://www.biorxiv.org/content/early/2020/10/29/2020.10.29.359794.full.pdf},
  journal = {bioRxiv}
}

MapGL: Inferring evolutionary gain and loss of short genomic sequence features by phylogenetic maximum parsimony

@article{Diehl2020MapGL,
  author = {Diehl, Adam G and Boyle, Alan P},
  title = {{MapGL: Inferring evolutionary gain and loss of short genomic sequence features by phylogenetic maximum parsimony}},
  year = {2020},
  month = {Sep},
  day = {22},
  volume = {21},
  pages = {416},
  doi = {10.1186/s12859-020-03742-9},
  abstract = {Comparative genomics studies are growing in number partly because of their unique ability to provide insight into shared and divergent biology between species. Of particular interest is the use of phylogenetic methods to infer the evolutionary history of cis-regulatory sequence features, which contribute strongly to phenotypic divergence and are frequently gained and lost in eutherian genomes. Understanding the mechanisms by which cis-regulatory element turnover generate emergent phenotypes is crucial to our understanding of adaptive evolution. Ancestral reconstruction methods can place species-specific cis-regulatory features in their evolutionary context, thus increasing our understanding of the process of regulatory sequence turnover. However, applying these methods to gain and loss of cis-regulatory features currently requires complex workflows which represent a potential barrier to widespread adoption by a broad scientific community. MapGL simplifies phylogenetic inference of the evolutionary history of short genomic sequence features by combining the necessary steps into a single piece of software with a simple set of inputs and outputs.},
  url = {https://bmcbioinformatics.biomedcentral.com/articles/10.1186/s12859-020-03742-9},
  pdf = {http://boylelab.org/pubs/BMC_Bioinformatics_2020_Diehl.pdf},
  journal = {BMC Bioinformatics},
  note = {{PMID:} 32962625}
}

TRACE: transcription factor footprinting using chromatin accessibility data and DNA sequence

@article{Ouyang2020TRACE,
  author = {Ouyang, Ningxin and Boyle, Alan P},
  title = {{TRACE: transcription factor footprinting using chromatin accessibility data and DNA sequence}},
  year = {2020},
  month = {Jul},
  day = {6},
  volume = {30},
  number = {7},
  pages = {1040--1046},
  doi = {10.1101/gr.258228.119},
  abstract = {Transcription is tightly regulated by cis-regulatory DNA elements where transcription factors can bind. Thus, identification of transcription factor binding sites is key to understanding gene expression and whole regulatory networks within a cell. The standard approaches for TFBSs prediction such as position weight matrices (PWMs) and chromatin immunoprecipitation followed by sequencing (ChIP-seq) are widely used but have their drawbacks such as high false positive rates and limited antibody availability, respectively. Several computational footprinting algorithms have been developed to detect TFBSs by investigating chromatin accessibility patterns, but also have their limitations. To improve on these methods, we have developed a footprinting method to predict Transcription factor footpRints in Active Chromatin Elements (TRACE). Trace incorporates DNase-seq data and PWMs within a multivariate Hidden Markov Model (HMM) to detect footprint-like regions with matching motifs. Trace is an unsupervised method that accurately annotates binding sites for specific TFs automatically with no requirement on pre-generated candidate binding sites or ChIP-seq training data. Compared to published footprinting algorithms, TRACE has the best overall performance with the distinct advantage of targeting multiple motifs in a single model.},
  url = {https://genome.cshlp.org/content/early/2020/07/02/gr.258228.119},
  pdf = {http://boylelab.org/pubs/Genome_Research_2020_Ouyang.pdf},
  journal = {Genome Research},
  note = {{PMID:} 32660981}
}

Perspectives on ENCODE

@article{ENCODEConsortium2020Perspectives,
  author = {{The ENCODE Project Consortium}},
  title = {{Perspectives on ENCODE}},
  journal = {Nature},
  year = {2020},
  month = {Jul},
  day = {30},
  volume = {583},
  number = {7818},
  pages = {693--698},
  abstract = {The Encylopedia of DNA Elements (ENCODE) Project launched in 2003 with the long-term goal of developing a comprehensive map of functional elements in the human genome. These included genes, biochemical regions associated with gene regulation (for example, transcription factor binding sites, open chromatin, and histone marks) and transcript isoforms. The marks serve as sites for candidate cis-regulatory elements (cCREs) that may serve functional roles in regulating gene expression 1. The project has been extended to model organisms, particularly the mouse. In the third phase of ENCODE, nearly a million and more than 300,000 cCRE annotations have been generated for human and mouse, respectively, and these have provided a valuable resource for the scientific community.},
  doi = {10.1038/s41586-020-2449-8},
  url = {https://doi.org/10.1038/s41586-020-2449-8},
  pdf = {http://boylelab.org/pubs/Nature_2020b_ENCODE.pdf},
  note = {{PMID:} 32728248}
}

Expanded encyclopaedias of DNA elements in the human and mouse genomes

@article{ENCODEConsortium2020ExpandedEncyclopedias,
  author = {{The ENCODE Project Consortium}},
  title = {{Expanded encyclopaedias of DNA elements in the human and mouse genomes}},
  journal = {Nature},
  year = {2020},
  month = {Jul},
  day = {30},
  volume = {583},
  number = {7818},
  pages = {699--710},
  abstract = {The human and mouse genomes contain instructions that specify RNAs and proteins and govern the timing, magnitude, and cellular context of their production. To better delineate these elements, phase III of the Encyclopedia of DNA Elements (ENCODE) Project has expanded analysis of the cell and tissue repertoires of RNA transcription, chromatin structure and modification, DNA methylation, chromatin looping, and occupancy by transcription factors and RNA-binding proteins. Here we summarize these efforts, which have produced 5,992 new experimental datasets, including systematic determinations across mouse fetal development. All data are available through the ENCODE data portal (https://www.encodeproject.org), including phase II ENCODE1 and Roadmap Epigenomics2 data. We have developed a registry of 926,535 human and 339,815 mouse candidate cis-regulatory elements, covering 7.9 and 3.4{\%} of their respective genomes, by integrating selected datatypes associated with gene regulation, and constructed a web-based server (SCREEN; http://screen.encodeproject.org) to provide flexible, user-defined access to this resource. Collectively, the ENCODE data and registry provide an expansive resource for the scientific community to build a better understanding of the organization and function of the human and mouse genomes.},
  issn = {1476--4687},
  doi = {10.1038/s41586-020-2493-4},
  url = {https://doi.org/10.1038/s41586-020-2493-4},
  pdf = {http://boylelab.org/pubs/Nature_2020_ENCODE.pdf},
  note = {{PMID:} 32728249}
}

Transposable elements contribute to cell and species-specific chromatin looping and gene regulation in mammalian genomes

@article{Diehl2020TransposableElementLoops,
  author = {Diehl, Adam G and Ouyang, Ningxin and Boyle, Alan P},
  title = {Transposable elements contribute to cell and species-specific chromatin looping and gene regulation in mammalian genomes},
  year = {2020},
  doi = {10.1038/s41467-020-15520-5},
  volume = {11},
  number = {1},
  pages = {1796},
  month = mar,
  abstract = {Chromatin looping is important for gene regulation, and studies of 3D chromatin structure across species and cell types have improved our understanding of the principles governing chromatin looping. However, 3D genome evolution and its relationship with natural selection remains largely unexplored. In mammals, the CTCF protein defines the boundaries of most chromatin loops, and variations in CTCF occupancy are associated with looping divergence. While many CTCF binding sites fall within transposable elements (TEs), their contribution to 3D chromatin structural evolution is unknown. Here we report the relative contributions of TE-driven CTCF binding site expansions to conserved and divergent chromatin looping in human and mouse. We demonstrate that TE-derived CTCF binding divergence may explain a large fraction of variable loops. These variable loops contribute significantly to corresponding gene expression variability across cells and species, possibly by refining sub-TAD-scale loop contacts responsible for cell-type-specific enhancer-promoter interactions.},
  url = {http://www.nature.com/articles/s41467-020-15520-5},
  pdf = {http://boylelab.org/pubs/Nature_Communications_2020_Diehl.pdf},
  journal = {Nature Communications},
  note = {{PMID:} 32286261}
}

Poly-Enrich: count-based methods for gene set enrichment testing with genomic regions

@article{Lee2020PolyEnrich,
  author = {Lee, Christopher T and Cavalcante, Raymond G and Lee, Chee and Qin, Tingting and Patil, Snehal and Wang, Shuze and Tsai, Zing T Y and Boyle, Alan P and Sartor, Maureen A},
  title = {{Poly-Enrich: count-based methods for gene set enrichment testing with genomic regions}},
  journal = {NAR Genomics and Bioinformatics},
  volume = {2},
  number = {1},
  year = {2020},
  month = feb,
  abstract = {{Gene set enrichment (GSE) testing enhances the biological interpretation of ChIP-seq data and other large sets of genomic regions. Our group has previously introduced two GSE methods for genomic regions: ChIP-Enrich for narrow regions and Broad-Enrich for broad regions. Here, we introduce Poly-Enrich, which has wider applicability, additional capabilities and models the number of peaks assigned to a gene using a generalized additive model with a negative binomial family to determine gene set enrichment, while adjusting for gene locus length. As opposed to ChIP-Enrich, Poly-Enrich works well even when nearly all genes have a peak, illustrated by using Poly-Enrich to characterize pathways and types of genic regions enriched with different families of repetitive elements. By comparing Poly-Enrich and ChIP-Enrich results with ENCODE ChIP-seq data, we found that the optimal test depends more on the pathway being regulated than on properties of the transcription factors. Using known transcription factor functions, we discovered clusters of related biological processes consistently better modeled with Poly-Enrich. This suggests that the regulation of certain processes may be modified by multiple binding events, better modeled by a count-based method. Our new hybrid method automatically uses the optimal method for each gene set, with correct FDR-adjustment.}},
  doi = {10.1093/nargab/lqaa006},
  url = {https://doi.org/10.1093/nargab/lqaa006},
  note = {{PMID:} 32051932},
  pdf = {http://boylelab.org/pubs/NAR_Genomics_and_Bioinformatics_2020_Lee.pdf}
}

Predicting the effects of SNPs on transcription factor binding affinity.

@article{Nishizaki2019SNPBindingAffinity,
  author = {Nishizaki, Sierra S and Ng, Natalie and Dong, Shengcheng and Porter, Robert S and Morterud, Cody and Williams, Colten and Asman, Courtney and Switzenberg, Jessica A and Boyle, Alan P},
  title = {{Predicting the effects of SNPs on transcription factor binding affinity.}},
  journal = {Bioinformatics},
  year = {2019},
  volume = {50},
  pages = {2434},
  month = aug,
  doi = {10.1093/bioinformatics/btz612},
  abstract = {MOTIVATION:GWAS have revealed that 88% of disease associated SNPs reside in noncoding regions. However, noncoding SNPs remain understudied, partly because they are challenging to prioritize for experimental validation. To address this deficiency, we developed the SNP effect matrix pipeline (SEMpl). RESULTS:SEMpl estimates transcription factor binding affinity by observing differences in ChIP-seq signal intensity for SNPs within functional transcription factor binding sites genome-wide. By cataloging the effects of every possible mutation within the transcription factor binding site motif, SEMpl can predict the consequences of SNPs to transcription factor binding. This knowledge can be used to identify potential disease-causing regulatory loci. AVAILABILITY AND IMPLEMENTATION:SEMpl is available from https://github.com/Boyle-Lab/SEM_CPP. SUPPLEMENTARY INFORMATION:Supplementary data are available at Bioinformatics online.},
  url = {https://academic.oup.com/bioinformatics/advance-article/doi/10.1093/bioinformatics/btz612/5543098},
  pdf = {http://boylelab.org/pubs/Bioinformatics_2019_Nishizaki.pdf},
  note = {{PMID:} 31373606}
}

CGIMP: Real-time exploration and covariate projection for self-organizing map datasets

@article{Diehl2019CGIMP,
  author = {Diehl, Adam G and Boyle, Alan P},
  title = {{CGIMP: Real-time exploration and covariate projection for self-organizing map datasets}},
  journal = {Journal of Open Source Software},
  year = {2019},
  volume = {4},
  number = {39},
  pages = {1520},
  month = jul,
  doi = {10.21105/joss.01520},
  url = {http://joss.theoj.org/papers/10.21105/joss.01520},
  pdf = {http://boylelab.org/pubs/JOSS_2019_Diehl.pdf}
}

The ENCODE Blacklist: Identification of Problematic Regions of the Genome.

@article{Amemiya2019ENCODEBlacklist,
  author = {Amemiya, Haley M and {\dag}Kundaje, Anshul and {\dag}Boyle, Alan P},
  title = {{The ENCODE Blacklist: Identification of Problematic Regions of the Genome.}},
  journal = {Scientific Reports},
  year = {2019},
  volume = {9},
  number = {1},
  pages = {9354},
  month = jun,
  doi = {10.1038/s41598-019-45839-z},
  abstract = {Functional genomics assays based on high-throughput sequencing greatly expand our ability to understand the genome. Here, we define the ENCODE blacklist- a comprehensive set of regions in the human, mouse, worm, and fly genomes that have anomalous, unstructured, or high signal in next-generation sequencing experiments independent of cell line or experiment. The removal of the ENCODE blacklist is an essential quality measure when analyzing functional genomics data.},
  url = {http://eutils.ncbi.nlm.nih.gov/entrez/eutils/elink.fcgi?dbfrom=pubmed&id=31249361&retmode=ref&cmd=prlinks},
  pdf = {http://boylelab.org/pubs/Sci_Rep_2019_Amemiya.pdf},
  note = {{PMID:} 31249361}
}

Cell specificity of regulatory annotations and their genetic effects on gene expression

@article{Varshney2019CellSpecificAnnotations,
  title = {Cell specificity of regulatory annotations and their genetic effects on gene expression},
  author = {Varshney, Arushi and VanRenterghem, Hadley and Orchard, Peter and {\dag}Boyle, Alan P and {\dag}Stitzel, Michael L and {\dag}Ucar, Duygu and Parker, Stephen CJ},
  year = {2019},
  doi = {10.1534/genetics.118.301525},
  journal = {Genetics},
  volume = {211},
  number = {2},
  pages = {549--562},
  abstract = {Epigenomic signatures from histone marks and transcription factor (TF)-binding sites have been used to annotate putative gene regulatory regions. However, a direct comparison of these diverse annotations is missing, and it is unclear how genetic variation within these annotations affects gene expression. Here, we compare five widely used annotations of active regulatory elements that represent high densities of one or more relevant epigenomic marks-"super" and "typical" (nonsuper) enhancers, stretch enhancers, high-occupancy target (HOT) regions, and broad domains-across the four matched human cell types for which they are available. We observe that stretch and super enhancers cover cell type-specific enhancer "chromatin states," whereas HOT regions and broad domains comprise more ubiquitous promoter states. Expression quantitative trait loci (eQTL) in stretch enhancers have significantly smaller effect sizes compared to those in HOT regions. Strikingly, chromatin accessibility QTL in stretch enhancers have significantly larger effect sizes compared to those in HOT regions. These observations suggest that stretch enhancers could harbor genetically primed chromatin to enable changes in TF binding, possibly to drive cell type-specific responses to environmental stimuli. Our results suggest that current eQTL studies are relatively underpowered or could lack the appropriate environmental context to detect genetic effects in the most cell type-specific "regulatory annotations," which likely contributes to infrequent colocalization of eQTL with genome-wide association study signals.},
  url = {http://eutils.ncbi.nlm.nih.gov/entrez/eutils/elink.fcgi?dbfrom=pubmed&id=30593493&retmode=ref&cmd=prlinks},
  pdf = {http://boylelab.org/pubs/Genetics_2019_Varshney.pdf},
  note = {{PMID:} 30593493}
}

Integration of Multiple Epigenomic Marks Improves Prediction of Variant Impact in Saturation Mutagenesis Reporter Assay.

@article{Shigaki2019EpigenomicVariantImpact,
  author = {Shigaki, Dustin and Adato, Orit and Adhikar, Aashish N and Dong, Shengcheng and Hawkins-Hooker, Alex and Inoue, Fumitaka and Juven-Gershon, Tamar and Kenlay, Henry and Martin, Beth and Patra, Ayoti and Penzar, Dmitry P and Schubach, Max and Xiong, Chenling and Yan, Zhongxia and Boyle, Alan P and Kreimer, Anat and Kulakovskiy, Ivan V. and Reid, John and Unger, Ron and Yosef, Nir and Shendure, Jay and Ahituv, Nadav and Kircher, Martin and Beer, Michael A},
  title = {{Integration of Multiple Epigenomic Marks Improves Prediction of Variant Impact in Saturation Mutagenesis Reporter Assay.}},
  journal = {Human mutation},
  year = {2019},
  volume = {33},
  number = {8},
  pages = {831},
  doi = {10.1002/humu.23797},
  abstract = {Integrative analysis of high-throughput reporter assays, machine learning, and profiles of epigenomic chromatin state in a broad array of cells and tissues has the potential to significantly improve our understanding of non-coding regulatory element function and its contribution to human disease. Here we report results from the CAGI 5 regulation saturation challenge where participants were asked to predict the impact of nucleotide substitution at every base pair within five disease associated human enhancers and nine disease associated promoters. A library of mutations covering all bases was generated by saturation mutagenesis and altered activity was assessed in a massively parallel reporter assay (MPRA) in relevant cell lines. Reporter expression was measured relative to plasmid DNA to determine the impact of variants. The challenge was to predict the functional effects of variants on reporter expression. Comparative analysis of the full range of submitted prediction results identifies the most successful models of transcription factor binding sites, machine learning algorithms, and ways to choose among or incorporate diverse datatypes and cell-types for training computational models. These results have the potential to improve the design of future studies on more diverse sets of regulatory elements and aid the interpretation of disease associated genetic variation. This article is protected by copyright. All rights reserved.},
  url = {https://onlinelibrary.wiley.com/doi/full/10.1002/humu.23797},
  pdf = {http://boylelab.org/pubs/Hum._Mutat._2019_Shigaki.pdf},
  note = {{PMID:} 31106481}
}

Predicting functional variants in enhancer and promoter elements using RegulomeDB

@article{Dong2019RegulomeDB,
  author = {Dong, Shengcheng and Boyle, Alan P},
  title = {{Predicting functional variants in enhancer and promoter elements using RegulomeDB}},
  year = {2019},
  volume = {33},
  number = {8},
  pages = {831},
  journal = {Human Mutation},
  doi = {10.1002/humu.23791},
  abstract = {Here we present a computational model, Score of Unified Regulatory Features (SURF), that predicts functional variants in enhancer and promoter elements. SURF is trained on data from massively parallel reporter assays and predicts the effect of variants on reporter expression levels. It achieved the top performance in the Fifth Critical Assessment of Genome Interpretation "Regulation Saturation" challenge. We also show that features queried through RegulomeDB, which are direct annotations from functional genomics data, help improve prediction accuracy beyond transfer learning features from DNA sequence-based deep learning models. Some of the most important features include DNase footprints, especially when coupled with complementary ChIP-seq data. Furthermore, we found our model achieved good performance in predicting allele-specific transcription factor binding events. As an extension to the current scoring system in RegulomeDB, we expect our computational model to prioritize variants in regulatory regions, thus help the understanding of functional variants in noncoding regions that lead to disease.},
  url = {https://onlinelibrary.wiley.com/doi/full/10.1002/humu.23791},
  pdf = {http://boylelab.org/pubs/Hum._Mutat._2019_Dong.pdf},
  note = {{PMID:} 31228310}
}

Conserved and species-specific transcription factor co-binding patterns drive divergent gene regulation in human and mouse

@article{Diehl2018TranscriptionFactorCobinding,
  author = {Diehl, Adam G and Boyle, Alan P},
  title = {Conserved and species-specific transcription factor co-binding patterns drive divergent gene regulation in human and mouse},
  year = {2018},
  volume = {46},
  number = {4},
  pages = {1878--1894},
  doi = {10.1093/nar/gky018},
  abstract = {The mouse has been widely used as a model system in which to study human genetic mechanisms. However, part of the difficulty in translating findings from mouse is that, despite high levels of gene conservation, regulatory control networks between human and mouse have been extensively rewired. To understand common themes of regulatory control we look beyond physical sharing of regulatory sequence, where extensive turnover of individual transcription factor binding sites complicates cross-species prediction of specific functions, and instead look at conserved properties of the regulatory code itself. We define regulatory conservation in terms of a grammar with shared, species-specific, and tissue-specific segments, and show that this grammar is more predictive of shared chromatin states and gene expression profiles than shared occupancy alone. Furthermore, we demonstrate a marked enrichment of disease associated variation in conserved grammatical patterns. These findings offer new understanding of transcriptional regulatory mechanisms shared between human and mouse.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/29361190},
  pdf = {http://boylelab.org/pubs/Nucleic_Acids_Res._2018_Diehl.pdf},
  journal = {Nucleic Acids Research},
  note = {{PMID:} 29361190}
}

Genome-wide Study of Atrial Fibrillation Identifies Seven Risk Loci and Highlights Biological Pathways and Regulatory Elements Involved in Cardiac Development.

@article{Nielsen2017AtrialFibrillation,
  author = {Nielsen, Jonas B and Fritsche, Lars G and Zhou, Wei and Teslovich, Tanya M and Holmen, Oddgeir L and Gustafsson, Stefan and Gabrielsen, Maiken E and Schmidt, Ellen M and Beaumont, Robin and Wolford, Brooke N and Lin, Maoxuan and Brummett, Chad M and Preuss, Michael H and Refsgaard, Lena and Bottinger, Erwin P and Graham, Sarah E and Surakka, Ida and Chu, Yunhan and Skogholt, Anne Heidi and Dalen, H{\aa}vard and Boyle, Alan P and Oral, Hakan and Herron, Todd J and Kitzman, Jacob and Jalife, Jos{\'e} and Svendsen, Jesper H and Olesen, Morten S and Nj{\o}lstad, Inger and L{\o}chen, Maja-Lisa and Baras, Aris and Gottesman, Omri and Marcketta, Anthony and O'Dushlaine, Colm and Ritchie, Marylyn D and Wilsgaard, Tom and Loos, Ruth J F and Frayling, Timothy M and Boehnke, Michael and Ingelsson, Erik and Carey, David J and Dewey, Frederick E and Kang, Hyun M and Abecasis, Gon{\c c}alo R and Hveem, Kristian and Willer, Cristen J},
  title = {{Genome-wide Study of Atrial Fibrillation Identifies Seven Risk Loci and Highlights Biological Pathways and Regulatory Elements Involved in Cardiac Development.}},
  journal = {American Journal of Human Genetics},
  year = {2017},
  month = dec,
  volume = {102},
  number = {1},
  pages = {103--115},
  doi = {10.1016/j.ajhg.2017.12.003},
  abstract = {Atrial fibrillation (AF) is a common cardiac arrhythmia and a major risk factor for stroke, heart failure, and premature death. The pathogenesis of AF remains poorly understood, which contributes to the current lack of highly effective treatments. To understand the genetic variation and biology underlying AF, we undertook a genome-wide association study (GWAS) of 6,337 AF individuals and 61,607 AF-free individuals from Norway, including replication in an additional 30,679 AF individuals and 278,895 AF-free individuals. Through genotyping and dense imputation mapping from whole-genome sequencing, we tested almost nine million genetic variants across the genome and identified seven risk loci, including two novel loci. One novel locus (lead single-nucleotide variant [SNV] rs12614435; p = 6.76~{\texttimes} 10-18) comprised intronic and several highly correlated missense variants situated in the I-, A-, and M-bands of titin, which is the largest protein in humans and responsible for the passive elasticity of heart and skeletal muscle. The other novel locus (lead SNV rs56202902; p~=~1.54~{\texttimes} 10-11) covered a large, gene-dense chromosome 1 region that has previously been linked to cardiac conduction. Pathway and functional enrichment analyses suggested that many AF-associated genetic variants act through a mechanism of impaired muscle cell differentiation and tissue formation during fetal heart development.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/29290336},
  pdf = {http://boylelab.org/pubs/American_Journal_of_Human_Genetics_2018_Nielsen.pdf},
  note = {{PMID:} 29290336}
}

A proximity-based graph clustering method for the identification and application of transcription factor clusters.

@article{Spadafore2017TranscriptionFactorClusters,
  author = {Spadafore, Maxwell and Najarian, Kayvan and Boyle, Alan P},
  title = {{A proximity-based graph clustering method for the identification and application of transcription factor clusters.}},
  journal = {BMC Bioinformatics},
  year = {2017},
  volume = {18},
  number = {1},
  pages = {530},
  month = nov,
  doi = {10.1186/s12859-017-1935-y},
  abstract = {BACKGROUND:Transcription factors (TFs) form a complex regulatory network within the cell that is crucial to cell functioning and human health. While methods to establish where a TF binds to DNA are well established, these methods provide no information describing how TFs interact with one another when they do bind. TFs tend to bind the genome in clusters, and current methods to identify these clusters are either limited in scope, unable to detect relationships beyond motif similarity, or not applied to TF-TF interactions. METHODS:Here, we present a proximity-based graph clustering approach to identify TF clusters using either ChIP-seq or motif search data. We use TF co-occurrence to construct a filtered, normalized adjacency matrix and use the Markov Clustering Algorithm to partition the graph while maintaining TF-cluster and cluster-cluster interactions. We then apply our graph structure beyond clustering, using it to increase the accuracy of motif-based TFBS searching for an example TF. RESULTS:We show that our method produces small, manageable clusters that encapsulate many known, experimentally validated transcription factor interactions and that our method is capable of capturing interactions that motif similarity methods might miss. Our graph structure is able to significantly increase the accuracy of motif TFBS searching, demonstrating that the TF-TF connections within the graph correlate with biological TF-TF interactions. CONCLUSION:The interactions identified by our method correspond to biological reality and allow for fast exploration of TF clustering and regulatory dynamics.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/29187152},
  pdf = {http://boylelab.org/pubs/BMC_Bioinformatics_2017_Spadafore.pdf},
  note = {{PMID:} 29187152}
}

Protein-altering and regulatory genetic variants near GATA4 implicated in bicuspid aortic valve.

@article{Yang2017GATA4BicuspidAorticValve,
  author = {*Yang, Bo and *Zhou, Wei and *Jiao, Jiao and Nielsen, Jonas B and Mathis, Michael R and Heydarpour, Mahyar and Lettre, Guillaume and Folkersen, Lasse and Prakash, Siddharth and Schurmann, Claudia and Fritsche, Lars and Farnum, Gregory A and Lin, Maoxuan and Othman, Mohammad and Hornsby, Whitney and Driscoll, Anisa and Levasseur, Alexandra and Thomas, Marc and Farhat, Linda and Dub{\'e}, Marie-Pierre and Isselbacher, Eric M and Franco-Cereceda, Anders and Guo, Dong-chuan and Bottinger, Erwin P and Deeb, G Michael and Booher, Anna and Kheterpal, Sachin and Chen, Y Eugene and Kang, Hyun Min and Kitzman, Jacob and Cordell, Heather J and Keavney, Bernard D and Goodship, Judith A and Ganesh, Santhi K and Abecasis, Gon{\c c}alo and Eagle, Kim A and Boyle, Alan P and Loos, Ruth J F and {\dag}Eriksson, Per and {\dag}Tardif, Jean-Claude and {\dag}Brummett, Chad M and {\dag}Milewicz, Dianna M and {\dag}Body, Simon C and {\dag}Willer, Cristen J},
  title = {{Protein-altering and regulatory genetic variants near GATA4 implicated in bicuspid aortic valve.}},
  journal = {Nature Communications},
  year = {2017},
  volume = {8},
  pages = {15481},
  month = may,
  doi = {10.1038/ncomms15481},
  abstract = {Bicuspid aortic valve (BAV) is a heritable congenital heart defect and an important risk factor for valvulopathy and aortopathy. Here we report a genome-wide association scan of 466 BAV cases and 4,660 age, sex and ethnicity-matched controls with replication in up to 1,326 cases and 8,103 controls. We identify association with a noncoding variant 151 kb from the gene encoding the cardiac-specific transcription factor, GATA4, and near-significance for p.Ser377Gly in GATA4. GATA4 was interrupted by CRISPR-Cas9 in induced pluripotent stem cells from healthy donors. The disruption of GATA4 significantly impaired the transition from endothelial cells into mesenchymal cells, a critical step in heart valve development.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/28541271},
  pdf = {http://boylelab.org/pubs/Nat_Commun_2017_Yang.pdf},
  note = {{PMID:} 28541271}
}

Mining the Unknown: Assigning Function to Noncoding Single Nucleotide Polymorphisms.

@article{Nishizaki2017NoncodingSNPFunction,
  author = {Nishizaki, Sierra S and Boyle, Alan P},
  title = {{Mining the Unknown: Assigning Function to Noncoding Single Nucleotide Polymorphisms.}},
  journal = {Trends in Genetics},
  year = {2017},
  volume = {33},
  number = {1},
  pages = {34--45},
  month = jan,
  doi = {10.1016/j.tig.2016.10.008},
  abstract = {One of the formative goals of genetics research is to understand how genetic variation leads to phenotypic differences and human disease. Genome-wide association studies (GWASs) bring us closer to this goal by linking variation with disease faster than ever before. Despite this, GWASs alone are unable to pinpoint disease-causing single nucleotide polymorphisms (SNPs). Noncoding SNPs, which represent the majority of GWAS SNPs, present a particular challenge. To address this challenge, an array of computational tools designed to prioritize and predict the function of noncoding GWAS SNPs have been developed. However, fewer than 40% of GWAS publications from 2015 utilized these tools. We discuss several leading methods for annotating noncoding variants and how they can be integrated into research pipelines in hopes that they will be broadly applied in future GWAS analyses.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/27939749},
  pdf = {http://boylelab.org/pubs/Trends_Genet_2017_Nishizaki.pdf},
  note = {{PMID:} 27939749}
}

Deciphering ENCODE.

@article{Diehl2016DecipheringENCODE,
  author = {Diehl, Adam G and Boyle, Alan P},
  title = {{Deciphering ENCODE.}},
  journal = {Trends in Genetics},
  year = {2016},
  volume = {32},
  number = {4},
  pages = {238--249},
  month = mar,
  doi = {10.1016/j.tig.2016.02.002},
  abstract = {The ENCODE project represents a major leap from merely describing and comparing genomic sequences to surveying them for direct indicators of function. The astounding quantity of data produced by the ENCODE consortium can serve as a map to locate specific landmarks, guide hypothesis generation, and lead us to principles and mechanisms underlying genome biology. Despite its broad appeal, the size and complexity of the repository can be intimidating to prospective users. We present here some background about the ENCODE data, survey the resources available for accessing them, and describe a few simple principles to help prospective users choose the data type(s) that best suit their needs, where to get them, and how to use them to their best advantage.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/26962025},
  pdf = {http://boylelab.org/pubs/Trends_Genet_2016_Diehl.pdf},
  note = {{PMID:} 26962025}
}

Mango: A bias correcting ChIA-PET analysis pipeline

@article{Phanstiel2015Mango,
  author = {Phanstiel, Douglas H and Boyle, Alan P and Heidari, Nastaran and Snyder, Michael P},
  title = {{Mango: A bias correcting ChIA-PET analysis pipeline}},
  year = {2015},
  doi = {10.1093/bioinformatics/btv336},
  abstract = {Motivation: Chromatin Interaction Analysis by Paired-End Tag sequencing (ChIA-PET) is an established method for detecting genome-wide looping interactions at high resolution. Current ChIA-PET analysis software packages either fail to correct for non-specific interactions due to genomic proximity or only address a fraction of the steps required for data processing. We present Mango, a complete ChIA-PET data analysis pipeline that provides statistical confidence estimates for interactions and corrects for major sources of bias including differential peak enrichment and genomic proximity.Results: Comparison to the existing software packages, ChIA-PET Tool and ChiaSig revealed that Mango interactions exhibit much better agreement with high-resolution Hi-C data. Importantly, Mango executes all steps required for processing ChIA-PET datasets, whereas ChiaSig only completes 20% of the required steps. Application of Mango to multiple available ChIA-PET datasets permitted the independent rediscovery of known trends in chromatin loops including enrichment of CTCF, RAD21, SMC3 and ZNF143 at the anchor regions of interactions and strong bias for convergent CTCF motifs.Availability and implementation: Mango is open source and distributed through github at https://github.com/dphansti/mango.Contact: mpsnyder@standford.eduSupplementary information: Supplementary data are available at Bioinformatics online.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/26034063},
  journal = {Bioinformatics},
  pdf = {http://boylelab.org/pubs/Bioinformatics_2015_Phanstiel.pdf},
  note = {{PMID:} 26034063}
}

A comparative encyclopedia of DNA elements in the mouse genome.

@article{Yue2014MouseENCODE,
  author = {*Yue, Feng and *Cheng, Yong and *Breschi, Alessandra and *Vierstra, Jeff and *Wu, Weisheng and *Ryba, Tyrone and *Sandstrom, Richard and *Ma, Zhihai and *Davis, Carrie and *Pope, Benjamin D and *Shen, Yin and Pervouchine, Dmitri D and Djebali, Sarah and Thurman, Robert E and Kaul, Rajinder and Rynes, Eric and Kirilusha, Anthony and Marinov, Georgi K and Williams, Brian A and Trout, Diane and Amrhein, Henry and Fisher-Aylor, Katherine and Antoshechkin, Igor and DeSalvo, Gilberto and See, Lei-Hoon and Fastuca, Meagan and Drenkow, Jorg and Zaleski, Chris and Dobin, Alex and Prieto, Pablo and Lagarde, Julien and Bussotti, Giovanni and Tanzer, Andrea and Denas, Olgert and Li, Kanwei and Bender, M A and Zhang, Miaohua and Byron, Rachel and Groudine, Mark T and McCleary, David and Pham, Long and Ye, Zhen and Kuan, Samantha and Edsall, Lee and Wu, Yi-Chieh and Rasmussen, Matthew D and Bansal, Mukul S and Kellis, Manolis and Keller, Cheryl A and Morrissey, Christapher S and Mishra, Tejaswini and Jain, Deepti and Dogan, Nergiz and Harris, Robert S and Cayting, Philip and Kawli, Trupti and Boyle, Alan P and Euskirchen, Ghia and Kundaje, Anshul and Lin, Shin and Lin, Yiing and Jansen, Camden and Malladi, Venkat S and Cline, Melissa S and Erickson, Drew T and Kirkup, Vanessa M and Learned, Katrina and Sloan, Cricket A and Rosenbloom, Kate R and Lacerda de Sousa, Beatriz and Beal, Kathryn and Pignatelli, Miguel and Flicek, Paul and Lian, Jin and Kahveci, Tamer and Lee, Dongwon and Kent, W James and Ramalho Santos, Miguel and Herrero, Javier and Notredame, Cedric and Johnson, Audra and Vong, Shinny and Lee, Kristen and Bates, Daniel and Neri, Fidencio and Diegel, Morgan and Canfield, Theresa and Sabo, Peter J and Wilken, Matthew S and Reh, Thomas A and Giste, Erika and Shafer, Anthony and Kutyavin, Tanya and Haugen, Eric and Dunn, Douglas and Reynolds, Alex P and Neph, Shane and Humbert, Richard and Hansen, R Scott and De Bruijn, Marella and Selleri, Licia and Rudensky, Alexander and Josefowicz, Steven and Samstein, Robert and Eichler, Evan E and Orkin, Stuart H and Levasseur, Dana and Papayannopoulou, Thalia and Chang, Kai-Hsin and Skoultchi, Arthur and Gosh, Srikanta and Disteche, Christine and Treuting, Piper and Wang, Yanli and Weiss, Mitchell J and Blobel, Gerd A and Cao, Xiaoyi and Zhong, Sheng and Wang, Ting and Good, Peter J. and Lowdon, Rebecca F. and Adams, Leslie B and Zhou, Xiao-Qiao and Pazin, Michael J and Feingold, Elise A. and Wold, Barbara and Taylor, James and Mortazavi, Ali and Weissman, Sherman M and Stamatoyannopoulos, John A and Snyder, Michael P and Guigo, Roderic and Gingeras, Thomas R. and Gilbert, David M and Hardison, Ross C and Beer, Michael A and Ren, Bing and {Mouse ENCODE Consortium}},
  title = {{A comparative encyclopedia of DNA elements in the mouse genome.}},
  journal = {Nature},
  year = {2014},
  volume = {515},
  number = {7527},
  pages = {355--364},
  month = nov,
  doi = {10.1038/nature13992},
  abstract = {The laboratory mouse shares the majority of its protein-coding genes with humans, making it the premier model organism in biomedical research, yet the two mammals differ in significant ways. To gain greater insights into both shared and species-specific transcriptional and cellular regulatory programs in the mouse, the Mouse ENCODE Consortium has mapped transcription, DNase I hypersensitivity, transcription factor binding, chromatin modifications and replication domains throughout the mouse genome in diverse cell and tissue types. By comparing with the human genome, we not only confirm substantial conservation in the newly annotated potential functional sequences, but also find a large degree of divergence of sequences involved in transcriptional regulation, chromatin state and higher order chromatin organization. Our results illuminate the wide range of evolutionary forces acting on genes and their regulatory regions, and provide a general resource for research into mammalian biology and mechanisms of human diseases.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/25409824},
  pdf = {http://boylelab.org/pubs/Nature_2014_Yue.pdf},
  note = {{PMID:} 25409824}
}

Principles of regulatory information conservation between mouse and human.

@article{Cheng2014RegulatoryConservation,
  author = {*Cheng, Yong and *Ma, Zhihai and Kim, Bong-Hyun and Wu, Weisheng and Cayting, Philip and Boyle, Alan P and Sundaram, Vasavi and Xing, Xiaoyun and Dogan, Nergiz and Li, Jingjing and Euskirchen, Ghia and Lin, Shin and Lin, Yiing and Visel, Axel and Kawli, Trupti and Yang, Xinqiong and Patacsil, Dorrelyn and Keller, Cheryl A and Giardine, Belinda and {Mouse ENCODE Consortium} and Kundaje, Anshul and Wang, Ting and Pennacchio, Len A and Weng, Zhiping and {\dag}Hardison, Ross C and {\dag}Snyder, Michael P},
  title = {{Principles of regulatory information conservation between mouse and human.}},
  journal = {Nature},
  year = {2014},
  volume = {515},
  number = {7527},
  pages = {371--375},
  month = nov,
  doi = {10.1038/nature13985},
  abstract = {To broaden our understanding of the evolution of gene regulation mechanisms, we generated occupancy profiles for 34 orthologous transcription factors (TFs) in human-mouse erythroid progenitor, lymphoblast and embryonic stem-cell lines. By combining the genome-wide transcription factor occupancy repertoires, associated epigenetic signals, and co-association patterns, here we deduce several evolutionary principles of gene regulatory features operating since the mouse and human lineages diverged. The genomic distribution profiles, primary binding motifs, chromatin states, and DNA methylation preferences are well conserved for TF-occupied sequences. However, the extent to which orthologous DNA segments are bound by orthologous TFs varies both among TFs and with genomic location: binding at promoters is more highly conserved than binding at distal elements. Notably, occupancy-conserved TF-occupied sequences tend to be pleiotropic; they function in several tissues and also co-associate with many TFs. Single nucleotide variants at sites with potential regulatory functions are enriched in occupancy-conserved TF-occupied sequences.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/25409826},
  pdf = {http://boylelab.org/pubs/Nature_2014_Cheng.pdf},
  note = {{PMID:} 25409826}
}

Comparative analysis of regulatory information and circuits across distant species.

@article{Boyle2014ComparativeRegulatoryCircuits,
  author = {*Boyle, Alan P and *Araya, Carlos L and Brdlik, Cathleen and Cayting, Philip and Cheng, Chao and Cheng, Yong and Gardner, Kathryn and Hillier, LaDeana W and Janette, Judith and Jiang, Lixia and Kasper, Dionna and Kawli, Trupti and Kheradpour, Pouya and Kundaje, Anshul and Li, Jingyi Jessica and Ma, Lijia and Niu, Wei and Rehm, E Jay and Rozowsky, Joel and Slattery, Matthew and Spokony, Rebecca and Terrell, Robert and Vafeados, Dionne and Wang, Daifeng and Weisdepp, Peter and Wu, Yi-Chieh and Xie, Dan and Yan, Koon-Kiu and Feingold, Elise A. and Good, Peter J. and Pazin, Michael J and Huang, Haiyan and Bickel, Peter J and Brenner, Steven E. and Reinke, Valerie and Waterston, Robert H and Gerstein, Mark and {\dag}White, Kevin P and {\dag}Kellis, Manolis and {\dag}Snyder, Michael},
  title = {{Comparative analysis of regulatory information and circuits across distant species.}},
  journal = {Nature},
  year = {2014},
  volume = {512},
  number = {7515},
  pages = {453--456},
  month = aug,
  doi = {10.1038/nature13668},
  abstract = {Despite the large evolutionary distances between metazoan species, they can show remarkable commonalities in their biology, and this has helped to establish fly and worm as model organisms for human biology. Although studies of individual elements and factors have explored similarities in gene regulation, a large-scale comparative analysis of basic principles of transcriptional regulatory features is lacking. Here we map the genome-wide binding locations of 165 human, 93 worm and 52 fly transcription regulatory factors, generating a total of 1,019 data sets from diverse cell types, developmental stages, or conditions in the three species, of which 498 (48.9%) are presented here for the first time. We find that structural properties of regulatory networks are remarkably conserved and that orthologous regulatory factor families recognize similar binding motifs in vivo and show some similar co-associations. Our results suggest that gene-regulatory properties previously observed for individual factors are general principles of metazoan regulation that are remarkably well-preserved despite extensive functional divergence of individual network connections. The comparative maps of regulatory circuitry provided here will drive an improved understanding of the regulatory underpinnings of model organism biology and how these relate to human biology, development and disease.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/25164757},
  pdf = {http://boylelab.org/pubs/Nature_2014_Boyle.pdf},
  note = {{PMID:} 25164757}
}

Regulatory analysis of the C. elegans genome with spatiotemporal resolution.

@article{Araya2014CElegansRegulation,
  author = {Araya, Carlos L and Kawli, Trupti and Kundaje, Anshul and Jiang, Lixia and Wu, Beijing and Vafeados, Dionne and Terrell, Robert and Weissdepp, Peter and Gevirtzman, Louis and Mace, Daniel and Niu, Wei and Boyle, Alan P and Xie, Dan and Ma, Lijia and Murray, John I. and Reinke, Valerie and Waterston, Robert H and Snyder, Michael},
  title = {{Regulatory analysis of the C. elegans genome with spatiotemporal resolution.}},
  journal = {Nature},
  year = {2014},
  volume = {512},
  number = {7515},
  pages = {400--405},
  month = aug,
  doi = {10.1038/nature13497},
  abstract = {Discovering the structure and dynamics of transcriptional regulatory events in the genome with cellular and temporal resolution is crucial to understanding the regulatory underpinnings of development and disease. We determined the genomic distribution of binding sites for 92 transcription factors and regulatory proteins across multiple stages of Caenorhabditis elegans development by performing 241 ChIP-seq (chromatin immunoprecipitation followed by sequencing) experiments. Integration of regulatory binding and cellular-resolution expression data produced a spatiotemporally resolved metazoan transcription factor binding map. Using this map, we explore developmental regulatory circuits that encode combinatorial logic at the levels of co-binding and co-expression of transcription factors, characterizing the genomic coverage and clustering of regulatory binding, the binding preferences of, and biological processes regulated by, transcription factors, the global transcription factor co-associations and genomic subdomains that suggest shared patterns of regulation, and identifying key transcription factors and transcription factor co-associations for fate specification of individual lineages and cell types.},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/25164749},
  pdf = {http://boylelab.org/pubs/Nature_2014_Araya.pdf},
  note = {{PMID:} 25164749}
}

Sushi.R: flexible, quantitative and integrative genomic visualizations for publication-quality multi-panel figures.

@article{Phanstiel2014SushiR,
  author = {Phanstiel, Douglas H and Boyle, Alan P and Araya, Carlos L and Snyder, Michael P},
  title = {{Sushi.R: flexible, quantitative and integrative genomic visualizations for publication-quality multi-panel figures.}},
  journal = {Bioinformatics},
  year = {2014},
  month = jun,
  doi = {10.1093/bioinformatics/btu379},
  abstract = {Motivation: Interpretation and communication of genomic data require flexible and quantitative tools to analyze and visualize diverse data types, and yet, a comprehensive tool to display all common genomic data types in publication quality figures does not exist to date. To address this shortcoming, we present Sushi.R, an R/Bioconductor package that allows flexible integration of genomic visualizations into highly customizable, publication-ready, multi-panel figures from common genomic data formats including Browser Extensible Data (BED), bedGraph and Browser Extensible Data Paired-End (BEDPE). Sushi.R is open source and made publicly available through GitHub (https://github.com/dphansti/Sushi) and Bioconductor (http://bioconductor.org/packages/release/bioc/html/Sushi.html).},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/24903420},
  pdf = {http://boylelab.org/pubs/Bioinformatics_2014_Phanstiel.pdf},
  note = {{PMID:} 24903420}
}

Extensive variation in chromatin states across humans

@article{Kasowski2013ChromatinStateVariation,
  title = {Extensive variation in chromatin states across humans},
  author = {*Kasowski, Maya and *Kyriazopoulou-Panagiotopoulou, Sofia and *Grubert, Fabian and *Zaugg, Judith B and *Kundaje, Anshul and Liu, Yuling and Boyle, Alan P and Zhang, Qiangfeng Cliff and Zakharia, Fouad and Spacek, Damek V and Li, Jingjing and Xie, Dan and Steinmetz, Lars M and Hogenesch, John B and Kellis, Manolis and Batzoglou, Serafim and Snyder, Michael},
  journal = {Science},
  year = {2013},
  month = {Nov},
  volume = {342},
  number = {6159},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/24136358},
  pdf = {http://boylelab.org/pubs/Science_2013_Kasowski.pdf},
  doi = {10.1126/science.1242510},
  pages = {750--752},
  note = {{PMID:} 24136358}
}

Dynamic trans-acting factor colocalization in human cells

@article{Xie2013FactorColocalization,
  title = {Dynamic trans-acting factor colocalization in human cells},
  author = {*Xie, Dan and *Boyle, Alan P and *Wu, Linfeng and Kawli, Trupti and Zhai, Jie and Snyder, Michael},
  journal = {Cell},
  pages = {713--724},
  volume = {155},
  number = {3},
  year = {2013},
  month = oct,
  url = {http://www.ncbi.nlm.nih.gov/pubmed/24243024},
  pdf = {http://boylelab.org/pubs/Cell_2013_Xie.pdf},
  doi = {10.1016/j.cell.2013.09.043},
  abstract = {Different trans-acting factors (TFs) collaborate and act in concert at distinct loci to perform accurate regulation of their target genes. To date, the cobinding of TF pairs has been investigated in a limited context both in terms of the number of factors within a cell type and across cell types and the extent of combinatorial colocalizations. Here, we use an approach to analyze TF colocalization within a cell type and across multiple cell lines at an unprecedented level. We extend this approach with large-scale mass spectrometry analysis of immunoprecipitations of 50 TFs. Our combined approach reveals large numbers of interesting TF-TF associations. We observe extensive change in TF colocalizations both within a cell type exposed to different conditions and across multiple cell types. We show distinct functional annotations and properties of different TF cobinding patterns and provide insights into the complex regulatory landscape of the cell.},
  note = {{PMID:} 24243024}
}

Linking disease associations with regulatory information in the human genome

@article{Schaub2012DiseaseRegulatoryInformation,
  title = {Linking disease associations with regulatory information in the human genome},
  volume = {22},
  issn = {1549-5469},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/22955986},
  pdf = {http://boylelab.org/pubs/Genome_Res._2012_Schaub.pdf},
  doi = {10.1101/gr.136127.111},
  abstract = {Genome-wide association studies have been successful in identifying single nucleotide polymorphisms {(SNPs)} associated with a large number of phenotypes. However, an associated {SNP} is likely part of a larger region of linkage disequilibrium. This makes it difficult to precisely identify the {SNPs} that have a biological link with the phenotype. We have systematically investigated the association of multiple types of {ENCODE} data with disease-associated {SNPs} and show that there is significant enrichment for functional {SNPs} among the currently identified associations. This enrichment is strongest when integrating multiple sources of functional information and when highest confidence disease-associated {SNPs} are used. We propose an approach that integrates multiple types of functional data generated by the {ENCODE} Consortium to help identify "functional {SNPs"} that may be associated with the disease phenotype. Our approach generates putative functional annotations for up to 80\% of all previously reported associations. We show that for most associations, the functional {SNP} most strongly supported by experimental evidence is a {SNP} in linkage disequilibrium with the reported association rather than the reported {SNP} itself. Our results show that the experimental data sets generated by the {ENCODE} Consortium can be successfully used to suggest functional hypotheses for variants associated with diseases and other phenotypes.},
  number = {9},
  journal = {Genome Research},
  author = {Schaub, Marc A and Boyle, Alan P and Kundaje, Anshul and {\dag}Batzoglou, Serafim and {\dag}Snyder, Michael},
  month = sep,
  year = {2012},
  note = {{PMID:} 22955986},
  pages = {1748--1759}
}

Architecture of the human regulatory network derived from ENCODE data

@article{Gerstein2012ENCODERegulatoryNetwork,
  title = {Architecture of the human regulatory network derived from {ENCODE} data},
  volume = {489},
  issn = {1476-4687},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/22955619},
  pdf = {http://boylelab.org/pubs/Nature_2012_Gerstein.pdf},
  doi = {10.1038/nature11245},
  abstract = {Transcription factors bind in a combinatorial fashion to specify the on-and-off states of genes; the ensemble of these binding events forms a regulatory network, constituting the wiring diagram for a cell. To examine the principles of the human transcriptional regulatory network, we determined the genomic binding information of 119 transcription-related factors in over 450 distinct experiments. We found the combinatorial, co-association of transcription factors to be highly context specific: distinct combinations of factors bind at specific genomic locations. In particular, there are significant differences in the binding proximal and distal to genes. We organized all the transcription factor binding into a hierarchy and integrated it with other genomic information (for example, {microRNA} regulation), forming a dense meta-network. Factors at different levels have different properties; for instance, top-level transcription factors more strongly influence expression and middle-level ones co-regulate targets to mitigate information-flow bottlenecks. Moreover, these co-regulations give rise to many enriched network motifs (for example, noise-buffering feed-forward loops). Finally, more connected network components are under stronger selection and exhibit a greater degree of allele-specific activity (that is, differential binding to the two parental alleles). The regulatory information obtained in this study will be crucial for interpreting personal genome sequences and understanding basic principles of human biology and disease.},
  number = {7414},
  journal = {Nature},
  author = {*Gerstein, Mark B and *Kundaje, Anshul and *Hariharan, Manoj and *Landt, Stephen G and *Yan, Koon-Kiu and *Cheng, Chao and *Mu, Xinmeng Jasmine and *Khurana, Ekta and *Rozowsky, Joel and *Alexander, Roger and *Min, Renqiang and *Alves, Pedro and Abyzov, Alexej and Addleman, Nick and Bhardwaj, Nitin and Boyle, Alan P and Cayting, Philip and Charos, Alexandra and Chen, David Z and Cheng, Yong and Clarke, Declan and Eastman, Catharine and Euskirchen, Ghia and Frietze, Seth and Fu, Yao and Gertz, Jason and Grubert, Fabian and Harmanci, Arif and Jain, Preti and Kasowski, Maya and Lacroute, Phil and Leng, Jing and Lian, Jin and Monahan, Hannah and {O'Geen}, Henriette and Ouyang, Zhengqing and Partridge, E Christopher and Patacsil, Dorrelyn and Pauli, Florencia and Raha, Debasish and Ramirez, Lucia and Reddy, Timothy E and Reed, Brian and Shi, Minyi and Slifer, Teri and Wang, Jing and Wu, Linfeng and Yang, Xinqiong and Yip, Kevin Y and Zilberman-Schapira, Gili and Batzoglou, Serafim and Sidow, Arend and Farnham, Peggy J and Myers, Richard M and Weissman, Sherman M and Snyder, Michael},
  month = sep,
  year = {2012},
  note = {{PMID:} 22955619},
  pages = {91--100}
}

An integrated encyclopedia of DNA elements in the human genome

@article{ENCODEConsortium2012IntegratedEncyclopedia,
  title = {An integrated encyclopedia of {DNA} elements in the human genome},
  volume = {489},
  issn = {1476-4687},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/22955616},
  pdf = {http://boylelab.org/pubs/Nature_2012_ENCODE_Project_Consortium.pdf},
  doi = {10.1038/nature11247},
  abstract = {The human genome encodes the blueprint of life, but the function of the vast majority of its nearly three billion bases is unknown. The Encyclopedia of {DNA} Elements {(ENCODE)} project has systematically mapped regions of transcription, transcription factor association, chromatin structure and histone modification. These data enabled us to assign biochemical functions for 80\% of the genome, in particular outside of the well-studied protein-coding regions. Many discovered candidate regulatory elements are physically associated with one another and with expressed genes, providing new insights into the mechanisms of gene regulation. The newly identified elements also show a statistical correspondence to sequence variants linked to human disease, and can thereby guide interpretation of this variation. Overall, the project provides new insights into the organization and regulation of our genes and genome, and is an expansive resource of functional annotations for biomedical research.},
  number = {7414},
  journal = {Nature},
  author = {{The ENCODE Project Consortium}},
  month = sep,
  year = {2012},
  note = {{PMID:} 22955616},
  pages = {57--74}
}

Annotation of functional variation in personal genomes using RegulomeDB

@article{Boyle2012RegulomeDB,
  title = {Annotation of functional variation in personal genomes using {RegulomeDB}},
  volume = {22},
  issn = {1549-5469},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/22955989},
  pdf = {http://boylelab.org/pubs/Genome_Res._2012_Boyle.pdf},
  doi = {10.1101/gr.137323.112},
  abstract = {As the sequencing of healthy and disease genomes becomes more commonplace, detailed annotation provides interpretation for individual variation responsible for normal and disease phenotypes. Current approaches focus on direct changes in protein coding genes, particularly nonsynonymous mutations that directly affect the gene product. However, most individual variation occurs outside of genes and, indeed, most markers generated from genome-wide association studies {(GWAS)} identify variants outside of coding segments. Identification of potential regulatory changes that perturb these sites will lead to a better localization of truly functional variants and interpretation of their effects. We have developed a novel approach and database, {RegulomeDB}, which guides interpretation of regulatory variants in the human genome. {RegulomeDB} includes high-throughput, experimental data sets from {ENCODE} and other sources, as well as computational predictions and manual annotations to identify putative regulatory potential and identify functional variants. These data sources are combined into a powerful tool that scores variants to help separate functional variants from a large pool and provides a small set of putative sites with testable hypotheses as to their function. We demonstrate the applicability of this tool to the annotation of noncoding variants from 69 full sequenced genomes as well as that of a personal genome, where thousands of functionally associated variants were identified. Moreover, we demonstrate a {GWAS} where the database is able to quickly identify the known associated functional variant and provide a hypothesis as to its function. Overall, we expect this approach and resource to be valuable for the annotation of human genome sequences.},
  number = {9},
  journal = {Genome Research},
  author = {Boyle, Alan P and Hong, Eurie L and Hariharan, Manoj and Cheng, Yong and Schaub, Marc A and Kasowski, Maya and Karczewski, Konrad J and Park, Julie and Hitz, Benjamin C and Weng, Shuai and Cherry, J Michael and Snyder, Michael},
  month = sep,
  year = {2012},
  note = {{PMID:} 22955989},
  pages = {1790--1797}
}

Personal omics profiling reveals dynamic molecular and medical phenotypes

@article{Chen2012PersonalOmics,
  title = {Personal omics profiling reveals dynamic molecular and medical phenotypes},
  volume = {148},
  issn = {1097-4172},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/22424236},
  pdf = {http://boylelab.org/pubs/Cell_2012_Chen.pdf},
  doi = {10.1016/j.cell.2012.02.009},
  abstract = {Personalized medicine is expected to benefit from combining genomic information with regular monitoring of physiological states by multiple high-throughput methods. Here, we present an integrative personal omics profile {(iPOP)}, an analysis that combines genomic, transcriptomic, proteomic, metabolomic, and autoantibody profiles from a single individual over a 14 month period. Our {iPOP} analysis revealed various medical risks, including type 2 diabetes. It also uncovered extensive, dynamic changes in diverse molecular components and biological pathways across healthy and diseased conditions. Extremely high-coverage genomic and transcriptomic data, which provide the basis of our {iPOP}, revealed extensive heteroallelic changes during healthy and diseased states and an unexpected {RNA} editing mechanism. This study demonstrates that longitudinal {iPOP} can be used to interpret healthy and diseased states by connecting genomic information with additional dynamic omics activity.},
  number = {6},
  journal = {Cell},
  author = {*Chen, Rui and *Mias, George I and *{Li-Pook-Than}, Jennifer and *Jiang, Lihua and Lam, Hugo Y K and Chen, Rong and Miriami, Elana and Karczewski, Konrad J and Hariharan, Manoj and Dewey, Frederick E and Cheng, Yong and Clark, Michael J and Im, Hogune and Habegger, Lukas and Balasubramanian, Suganthi and {O'Huallachain}, Maeve and Dudley, Joel T and Hillenmeyer, Sara and Haraksingh, Rajini and Sharon, Donald and Euskirchen, Ghia and Lacroute, Phil and Bettinger, Keith and Boyle, Alan P and Kasowski, Maya and Grubert, Fabian and Seki, Scott and Garcia, Marco and {Whirl-Carrillo}, Michelle and Gallardo, Mercedes and Blasco, Maria A and Greenberg, Peter L and Snyder, Phyllis and Klein, Teri E and Altman, Russ B and Butte, Atul J and Ashley, Euan A and Gerstein, Mark and Nadeau, Kari C and Tang, Hua and Snyder, Michael},
  month = mar,
  year = {2012},
  note = {{PMID:} 22424236},
  pages = {1293--1307}
}

Open chromatin defined by DNaseI and FAIRE identifies regulatory elements that shape cell-type identity

@article{Song2011OpenChromatin,
  title = {Open chromatin defined by {DNaseI} and {FAIRE} identifies regulatory elements that shape cell-type identity},
  issn = {1549-5469},
  volume = {21},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/21750106},
  pdf = {http://boylelab.org/pubs/Genome_Res._2011_Song.pdf},
  doi = {10.1101/gr.121541.111},
  abstract = {The human body contains thousands of unique cell types, each with specialized functions. Cell identity is governed in large part by gene transcription programs, which are determined by regulatory elements encoded in {DNA.} To identify regulatory elements active in seven cell lines representative of diverse human cell types, we used {DNase-seq} and {FAIRE-seq} to map "open chromatin." Over 870,000 {DNaseI} or {FAIRE} sites, which correspond tightly to nucleosome depleted regions, were identified across the seven cell lines, covering nearly 9\% of the genome. The combination of {DNaseI} and {FAIRE} is more effective than either assay alone in identifying likely regulatory elements, as judged by coincidence with transcription factor binding locations determined in the same cells. Open chromatin common to all seven cell types tended to be at or near transcription start sites and to be coincident with {CTCF} binding sites, while open chromatin sites found in only one cell type were typically located away from transcription start sites, and contained {DNA} motifs recognized by regulators of cell-type identity. We show that open chromatin regions bound by {CTCF} are potent insulators. We identified clusters of open regulatory elements {(COREs)} that were physically near each other and whose appearance was coordinated among one or more cell types. Gene expression and {RNA} Pol {II} binding data support the hypothesis that {COREs} control gene activity required for the maintenance of cell-type identity. This publicly available atlas of regulatory elements may prove valuable in identifying non-coding {DNA} sequence variants that are causally linked to human disease.},
  journal = {Genome Research},
  author = {*Song, Lingyun and *Zhang, Zhancheng and *Grasfeder, Linda L and *Boyle, Alan P and *Giresi, Paul G and *Lee, {Bum-Kyu} and *Sheffield, Nathan C and Graff, Stefan and Huss, Mikael and Keefe, Damian and Liu, Zheng and London, Darin and {McDaniell}, Ryan M and Shibata, Yoichiro and Showers, Kimberly A and Simon, Jeremy M and Vales, Teresa and Wang, Tianyuan and Winter, Deborah and Zhang, Zhuzhu and Clarke, Neil D and {\dag}Birney, Ewan and {\dag}Iyer, Vishy R and {\dag}Crawford, Gregory E and {\dag}Lieb, Jason D and {\dag}Furey, Terrence S},
  month = jul,
  year = {2011},
  number = {10},
  note = {{PMID:} 21750106},
  pages = {1757--1767}
}

A User's Guide to the Encyclopedia of DNA Elements (ENCODE)

@article{ENCODEConsortium2011UsersGuide,
  title = {A User's Guide to the Encyclopedia of {DNA} Elements {(ENCODE)}},
  volume = {9},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/21526222},
  pdf = {http://boylelab.org/pubs/PLoS_Biol_2011_ENCODE_Project_Consortium.pdf},
  doi = {10.1371/journal.pbio.1001046},
  abstract = {The Encyclopedia of {DNA} Elements {(ENCODE)} Project was created to enable the scientific and medical communities to interpret the human genome sequence and to use it to understand human biology and improve health. The {ENCODE} Consortium, a 
large group of scientists from around the world, uses a variety of experimental methods to identify and describe the regions of the 3 billion base-pair human genome that are important for function. Using experimental, computational, and statistical analyses, we 
aimed to discover and describe genes, transcripts, and transcriptional regulatory regions, as well as {DNA} binding proteins that interact with regulatory regions in the genome, including transcription factors, different versions of histones and other markers, 
and {DNA} methylation patterns that define states of the genome in various cell types. The {ENCODE} Project has developed standards for each experiment type to ensure high-quality, reproducible data and novel algorithms to facilitate analysis. All data and 
derived results are made available through a freely accessible database. This article provides an overview of the complete project and the resources it is generating, as well as examples to illustrate the application of {ENCODE} data as a user's guide to 
facilitate the interpretation of the human genome.},
  number = {4},
  journal = {{PLoS} Biology},
  author = {{The ENCODE Project Consortium}},
  month = apr,
  year = {2011},
  pages = {e1001046},
  note = {{PMID:} 21526222}
}

High-resolution genome-wide in vivo footprinting of diverse transcription factors in human cells

@article{Boyle2011GenomeWideFootprinting,
  title = {High-resolution genome-wide in vivo footprinting of diverse transcription factors in human cells},
  volume = {21},
  issn = {1549-5469},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/21106903},
  pdf = {http://boylelab.org/pubs/Genome_Res._2011_Boyle.pdf},
  doi = {10.1101/gr.112656.110},
  abstract = {Regulation of gene transcription in diverse cell 
types is largely determined by varied sets of cis-elements where 
transcription factors bind. Here we demonstrate that data from a single 
high-throughput {DNaseI} hypersensitivity assay can delineate hundreds 
of thousands of base-pair resolution in vivo footprints in human cells 
that precisely mark individual transcription {factor-DNA} interactions. 
These annotations provide a unique resource for the investigation of 
cis-regulatory elements. We find that footprints for specific 
transcription factors correlate with {ChIP-seq} enrichment and can 
accurately identify functional vs. non-functional transcription factor 
motifs. We also find that footprints reveal a unique evolutionary 
conservation pattern that differentiates functional footprinted bases 
from surrounding {DNA.} Finally, detailed analysis of {CTCF} footprints 
suggests multiple modes of binding and a novel {DNA} binding motif 
upstream of the primary binding site.},
  journal = {Genome Research},
  author = {Alan P Boyle and Lingyun Song and {Bum-Kyu} Lee and 
Darin London and Damian Keefe and Ewan Birney and Vishwanath R Iyer and 
Gregory E {\dag}Crawford and Terrence S {\dag}Furey},
  month = mar,
  year = {2011},
  note = {{PMID:} 21106903},
  pages = {456--464}
}

Evidence-ranked motif identification

@article{Georgiev2010MotifIdentification,
  title = {Evidence-ranked motif identification},
  volume = {11},
  issn = {1465-6914},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/20156354},
  pdf = {http://boylelab.org/pubs/Genome_Biol._2010_Georgiev.pdf},
  doi = {10.1186/gb-2010-11-2-r19},
  abstract = {{ABSTRACT:} {cERMIT} is a computationally efficient 
motif discovery tool based on analyzing genome-wide quantitative 
regulatory evidence. Instead of pre-selecting promising candidate 
sequences, it utilizes information across all sequence regions to search 
for high-scoring motifs. We apply {cERMIT} on a range of direct binding 
and overexpression data sets; it substantially outperforms 
state-of-the-art approaches on curated {ChIP-chip} datasets, and easily 
scales to current mammalian {ChIP-seq} experiments with data on 
thousands of non-coding regions.},
  number = {2},
  journal = {Genome Biology},
  author = {Stoyan Georgiev and Alan P Boyle and Karthik Jayasurya 
and Sayan Mukherjee and Uwe Ohler},
  month = feb,
  year = {2010},
  note = {{PMID:} 20156354},
  pages = {R19}
}

Global epigenomic analysis of primary human pancreatic islets provides insights into type 2 diabetes susceptibility loci

@article{Stitzel2010PancreaticIsletEpigenome,
  title = {Global epigenomic analysis of primary human pancreatic 
islets provides insights into type 2 diabetes susceptibility loci},
  volume = {12},
  issn = {1932-7420},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/21035756},
  pdf = {http://boylelab.org/pubs/Cell_Metab._2010_Stitzel.pdf},
  doi = {10.1016/j.cmet.2010.09.012},
  abstract = {Identifying cis-regulatory elements is important to 
understanding how human pancreatic islets modulate gene expression in 
physiologic or pathophysiologic (e.g., diabetic) conditions. We 
conducted genome-wide analysis of {DNase} I hypersensitive sites, 
histone H3 lysine methylation modifications {(K4me1,} K4me3, K79me2), 
and {CCCTC} factor {(CTCF)} binding in human islets. This identified 
.18,000 putative promoters (several hundred unannotated and 
islet-active). Surprisingly, active promoter modifications were absent 
at genes encoding islet-specific hormones, suggesting a distinct 
regulatory mechanism. Of 34,039 distal (nonpromoter) regulatory 
elements, 47\% are islet unique and 22\% are {CTCF} bound. In the 18 
type 2 diabetes {(T2D)-associated} loci, we identified 118 putative 
regulatory elements and confirmed enhancer activity for 12 of 33 tested. 
Among six regulatory elements harboring {T2D-associated} variants, two 
exhibit significant allele-specific differences in activity. These 
findings present a global snapshot of the human islet epigenome and 
should provide functional context for noncoding variants emerging from 
genetic studies of {T2D} and other islet disorders.},
  number = {5},
  journal = {Cell Metabolism},
  author = {Michael L *Stitzel and Praveen *Sethupathy and Daniel S 
Pearson and Peter S Chines and Lingyun Song and Michael R Erdos and Ryan 
Welch and Stephen C J Parker and Alan P Boyle and Laura J Scott and 
Elliott H Margulies and Michael Boehnke and Terrence S Furey and Gregory 
E Crawford and Francis S Collins},
  month = nov,
  year = {2010},
  note = {{PMID:} 21035756},
  pages = {443--455}
}

Heritable individual-specific and allele-specific chromatin signatures in humans

@article{McDaniell2010ChromatinSignatures,
  title = {Heritable individual-specific and allele-specific 
chromatin signatures in humans},
  volume = {328},
  issn = {1095-9203},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/20299549},
  pdf = {http://boylelab.org/pubs/Science_2010_McDaniell.pdf},
  documenturl = {http://f1000.com/3410956},
  doi = {10.1126/science.1184655},
  abstract = {The extent to which variation in chromatin structure 
and transcription factor binding may influence gene expression, and thus 
underlie or contribute to variation in phenotype, is unknown. To address 
this question, we cataloged both individual-to-individual variation and 
differences between homologous chromosomes within the same individual 
(allele-specific variation) in chromatin structure and transcription 
factor binding in lymphoblastoid cells derived from individuals of 
geographically diverse ancestry. Ten percent of active chromatin sites 
were individual-specific; a similar proportion were allele-specific. 
Both individual-specific and allele-specific sites were commonly 
transmitted from parent to child, which suggests that they are heritable 
features of the human genome. Our study shows that heritable chromatin 
status and transcription factor binding differ as a result of genetic 
variation and may underlie phenotypic variation in humans.},
  number = {5975},
  journal = {Science},
  author = {Ryan {McDaniell} and {Bum-Kyu} Lee and Lingyun Song 
and Zheng Liu and Alan P Boyle and Michael R Erdos and Laura J Scott and 
Mario A Morken and Katerina S Kucera and Anna Battenhouse and Damian 
Keefe and Francis S Collins and Huntington F Willard and Jason D Lieb 
and Terrence S Furey and Gregory E {\dag}Crawford and Vishwanath R {\dag}Iyer and 
Ewan {\dag}Birney},
  month = apr,
  year = {2010},
  note = {{PMID:} 20299549},
  keywords = {African Continental Ancestry Group, Alleles, Binding 
Sites, Cell Line, Chromatin, Chromatin Immunoprecipitation, Chromosomes, 
Human, Chromosomes, Human, X, Deoxyribonuclease I, European Continental 
Ancestry Group, Female, Gene Expression Regulation, Genetic Variation, 
Humans, Male, Nuclear Family, Polymorphism, Single Nucleotide, Protein 
Binding, Regulatory Elements, Transcriptional, Repressor Proteins, 
Sequence Analysis, {DNA,} Transcription Factors, X Chromosome 
Inactivation},
  pages = {235--239}
}

Both noncoding and protein-coding RNAs contribute to gene expression evolution in the primate brain

@article{Babbitt2010PrimateBrainRNAEvolution,
  title = {Both noncoding and protein-coding {RNAs} contribute to 
gene expression evolution in the primate brain},
  volume = {2},
  issn = {1759-6653},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/20333225},
  pdf = {http://boylelab.org/pubs/Genome_Biol_Evol_2010_Babbitt.pdf},
  doi = {10.1093/gbe/evq002},
  abstract = {Despite striking differences in cognition and 
behavior between humans and our closest primate relatives, several 
studies have found little evidence for adaptive change in protein-coding 
regions of genes expressed primarily in the brain. Instead, changes in 
gene expression may underlie many cognitive and behavioral differences. 
Here, we used digital gene expression: tag profiling (here called 
{Tag-Seq,} also called {DGE:tag} profiling) to assess changes in global 
transcript abundance in the frontal cortex of the brains of 3 humans, 3 
chimpanzees, and 3 rhesus macaques. A substantial fraction of 
transcripts we identified as differentially transcribed among species 
were not assayed in previous studies based on microarrays. 
Differentially expressed tags within coding regions are enriched for 
gene functions involved in synaptic transmission, transport, oxidative 
phosphorylation, and lipid metabolism. Importantly, because {Tag-Seq} 
technology provides strand-specific information about all polyadenlyated 
transcripts, we were able to assay expression in noncoding intragenic 
regions, including both sense and antisense noncoding transcripts 
(relative to nearby genes). We find that many noncoding transcripts are 
conserved in both location and expression level between species, 
suggesting a possible functional role. Lastly, we examined the overlap 
between differential gene expression and signatures of positive 
selection within putative promoter regions, a sign that these 
differences represent adaptations during human evolution. Comparative 
approaches may provide important insights into genes responsible for 
differences in cognitive functions between humans and nonhuman primates, 
as well as highlighting new candidate genes for studies investigating 
neurological disorders.},
  journal = {Genome Biology and Evolution},
  author = {Courtney C Babbitt and Olivier Fedrigo and Adam D 
Pfefferle and Alan P Boyle and Julie E Horvath and Terrence S Furey and 
Gregory A Wray},
  year = {2010},
  note = {{PMID:} 20333225},
  pages = {67--79}
}

DNaseI hypersensitivity at gene-poor, FSH dystrophy-linked 4q35.2

@article{Xu2009FSHDChromatin,
  title = {{DNaseI} hypersensitivity at gene-poor, {FSH} 
dystrophy-linked 4q35.2},
  volume = {37},
  issn = {1362-4962},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/19820107},
  pdf = {http://boylelab.org/pubs/Nucleic_Acids_Res._2009_Xu.pdf},
  doi = {10.1093/nar/gkp833},
  abstract = {A subtelomeric region, 4q35.2, is implicated in 
facioscapulohumeral muscular dystrophy {(FSHD),} a dominant disease 
thought to involve local pathogenic changes in chromatin. {FSHD} 
patients have too few copies of a tandem 3.3-kb repeat {(D4Z4)} at 
4q35.2. No phenotype is associated with having few copies of an almost 
identical repeat at 10q26.3. Standard expression analyses have not given 
definitive answers as to the genes involved. To investigate the 
pathogenic effects of short {D4Z4} arrays on gene expression in the very 
gene-poor 4q35.2 and to find chromatin landmarks there for transcription 
control, unannotated genes and chromatin structure, we mapped 
{DNaseI-hypersensitive} {(DH)} sites in {FSHD} and control myoblasts. 
Using custom tiling arrays {(DNase-chip),} we found unexpectedly many 
{DH} sites in the two large gene deserts in this {4-Mb} region. One site 
was seen preferentially in {FSHD} myoblasts. Several others were mapped 
{\textgreater}0.7 Mb from genes known to be active in the muscle lineage 
and were also observed in cultured fibroblasts, but not in lymphoid, 
myeloid or hepatic cells. Their selective occurrence in cells derived 
from mesoderm suggests functionality. Our findings indicate that the 
gene desert regions of 4q35.2 may have functional significance, possibly 
also to {FSHD,} despite their paucity of known genes.},
  number = {22},
  journal = {Nucleic Acids Research},
  author = {Xueqing Xu and Koji Tsumagari and Janet Sowden and Rabi Tawil and Alan P Boyle and Lingyun Song and Terrence S Furey and Gregory E Crawford and Melanie Ehrlich},
  month = dec,
  year = {2009},
  note = {{PMID:} 19820107},
  pages = {7381--7393}
}

High-resolution mapping studies of chromatin and gene regulatory elements

@article{Boyle2009ChromatinMapping,
  title = {High-resolution mapping studies of chromatin and gene regulatory elements},
  volume = {1},
  issn = {1750-1911},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/20514362},
  pdf = {http://boylelab.org/pubs/Epigenomics_2009_Boyle.pdf},
  doi = {10.2217/epi.09.29},
  abstract = {Microarray and high-throughput sequencing technologies have enabled the development of comprehensive assays to identify locations of particular chromatin structures and regulatory elements. It is now possible to create genome-wide maps of {DNA} methylation, trans-factor binding sites, histone variants and histone tail modifications, nucleosome positions, regions of open chromatin, and chromosome locations and interactions. This review provides a summary of these new assays that are changing the way in which molecular biology research is being performed. While the generation of large amounts of data from these experiments is becoming increasingly easier, the development of corresponding analysis methods has progressed more slowly. It will likely be years before the full extent of the information contained in these data is fully appreciated.},
  number = {2},
  journal = {Epigenomics},
  author = {Alan P Boyle and Terrence S Furey},
  year = {2009},
  note = {{PMID:} 20514362},
  pages = {319--329}
}

F-Seq: a feature density estimator for high-throughput sequence tags.

@article{Boyle2008FSeq,
  title = {{F-Seq:} a feature density estimator for high-throughput sequence tags.},
  volume = {24},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/18784119},
  pdf = {http://boylelab.org/pubs/Bioinformatics_2008_Boyle.pdf},
  doi = {10.1093/bioinformatics/btn480},
  abstract = {Tag sequencing using high-throughput sequencing technologies are now regularly employed to identify specific sequence features, such as transcription factor binding sites {(ChIP-seq)} or regions of open chromatin {(DNase-seq).} To intuitively summarize and display individual sequence data as an accurate and interpretable signal, we developed {F-Seq,} a software package that generates a continuous tag sequence density estimation allowing identification of biologically meaningful sites whose output can be displayed directly in the {UCSC} Genome Browser. {AVAILABILITY:} The software is written in the Java language and is available on all major computing platforms for download at http://www.genome.duke.edu/labs/furey/software/fseq.},
  number = {21},
  journal = {Bioinformatics},
  author = {Alan P Boyle and Justin Guinney and Gregory E Crawford and Terrence S Furey},
  month = nov,
  year = {2008},
  note = {{PMID:} 18784119},
  pages = {2537--2538}
}

High-resolution mapping and characterization of open chromatin across the genome.

@article{Boyle2008OpenChromatin,
  title = {High-resolution mapping and characterization of open chromatin across the genome.},
  volume = {132},
  url = {http://www.ncbi.nlm.nih.gov/pubmed/18243105},
  pdf = {http://boylelab.org/pubs/Cell_2008_Boyle.pdf},
  doi = {10.1016/j.cell.2007.12.014},
  abstract = {Mapping {DNase} I hypersensitive {(HS)} sites is an accurate method of identifying the location of genetic regulatory elements, including promoters, enhancers, silencers, insulators, and locus control regions. We employed high-throughput sequencing and whole-genome tiled array strategies to identify {DNase} I {HS} sites within human primary {CD4+} T cells. Combining these two technologies, we have created a comprehensive and accurate genome-wide open chromatin map. Surprisingly, only 16\%-21\% of the identified 94,925 {DNase} I {HS} sites are found in promoters or first exons of known genes, but nearly half of the most open sites are in these regions. In conjunction with expression, motif, and chromatin immunoprecipitation data, we find evidence of cell-type-specific characteristics, including the ability to identify transcription start sites and locations of different chromatin marks utilized in these cells. In addition, and unexpectedly, our analyses have uncovered detailed features of nucleosome structure.},
  number = {2},
  journal = {Cell},
  author = {Alan P Boyle and Sean Davis and Hennady P Shulha and Paul Meltzer and Elliott H Margulies and Zhiping Weng and Terrence S {\dag}Furey and Gregory E {\dag}Crawford},
  month = jan,
  year = {2008},
  note = {{PMID:} 18243105},
  pages = {311--322}
}

Identification of Regulatory Elements in Archaea using Self-Organizing Maps

@inproceedings{Boyle2004ArchaeaRegulatoryElements,
  title = {Identification of Regulatory Elements in Archaea using Self-Organizing Maps},
  pdf = {http://boylelab.org/pubs/RECOMB_2004_Boyle.pdf},
  booktitle = {Proc RECOMB},
  author = {Alan P Boyle and John A Boyle and Susan M Bridges},
  month = mar,
  year = {2004}
}

Global analysis of microbial translation initiation regions

@inproceedings{Boyle2003TranslationInitiation,
  title = {Global analysis of microbial translation initiation regions},
  url = {http://www.thefreelibrary.com/Global+analysis+of+microbial+translation+initiation+regions-a0105160258},
  pdf = {http://boylelab.org/pubs/MAS_2003_Boyle.pdf},
  abstract = {The availability of genomic sequences from multiple bacteria has allowed global comparisons of patterns. Here we present a graphical comparison of normalized base frequencies in the vicinity of translation  starts for both eubacteria and archae. The results show that most eubacterial Open Reading Frames (ORFs) are preceded by a distinctly recognizable Shine-Dalgarno (SD) sequence pattern. However, some eubacteria deviate from this arrangement and have diminished SD patterns or completely lack this sequence. On the other hand, some archae seem to use both SD sequences and leaderless transcripts in their translation initiation  process. This is dependent on the position of a gene within an operon. Most archae seem to have other regular sequences located upstream from the typical SD location. Both eubacteria and archae have a surprising repetitive pattern seen within the averaged ORFs. The eubacterial and archaeal averaged patterns are slightly different from each other, and individual organisms within each domain vary from the averages. Nevertheless, the existence of such a periodicity within ORFs may allow the development of new techniques to identify actual genes from ORFs.},
  volume = {48},
  number = {3},
  pages = {138--150},
  booktitle = {Journal of the Mississippi Academy of Sciences},
  author = {Alan P Boyle and John A Boyle},
  month = jul,
  year = {2003}
}

Clustering of archael gene regulatory regions

@inproceedings{Boyle2003ArchaeaRegulatoryClustering,
  title = {Clustering of archael gene regulatory regions},
  volume = {17},
  number = {5},
  pages = {A985-A985},
  booktitle = {FASEB Journal},
  author = {Alan P Boyle and Susan Bridges},
  month = mar,
  year = {2003}
}

Visualization of aligned genomic open reading frame data

@article{Boyle2003GenomicORFVisualization,
  title = {Visualization of aligned genomic open reading frame data},
  issn = {1539-3429},
  pdf = {http://boylelab.org/pubs/BAMBED_2002_Boyle.pdf},
  abstract = {Students can better appreciate the value of genomic data if they are asked to use the data themselves. However, in general the enormous volume of data involved makes detailed examination difficult. Here we present a web site that allows students to study one particular aspect of sequenced genomes. They are able to align the open reading frames (ORFs) of any available genome that is of reasonable size. The ORFs may be aligned using either the start codon or the stop codon as the starting points. Results will readily show the presence of common ribosome binding sites as well as reveal interesting order within the ORFs that is nonexistent outside of them. Students will be able to ask various questions involving comparisons of genomes and see the results presented in both a tabular and graphic format. An example problem is presented under "Results."},
  volume = {31},
  number = {1},
  pages = {64--68},
  doi = {10.1002/bmb.2003.494031010144},
  journal = {Biochemistry and Molecular Biology Education},
  author = {Alan P Boyle and John A Boyle},
  month = jan,
  year = {2003}
}

Interactive clustering for exploration of genomic data

@inproceedings{Wan2002InteractiveGenomicClustering,
  title = {Interactive clustering for exploration of genomic data},
  pdf = {http://boylelab.org/pubs/ANNIE_2002_Wan.pdf},
  abstract = {The complete genomic sequences for many organisms, particularly primitive organisms with relatively small genomes (prokaryotes), are now available. We describe an approach that supports interactive exploration of patterns in genomic data by combining use of positional weight matrices, the k-means clustering algorithm, and a visualization tool. Users interact with the system by examining a visualization of the "average" pattern found in each cluster for the sequence under consideration and determine if further clustering or modified clustering is desired. The effectiveness of this approach is demonstrated by a study of promoter sequences in archaea.},
  volume = {12},
  pages = {753--758},
  booktitle = {Proceedings of the Artificial Neural Networks in Engineering Conference},
  author = {Xiufeng Wan and John A Boyle and Susan M Bridges and Alan P Boyle},
  address = {St. Louis, MO},
  month = nov,
  year = {2002}
}