@inproceedings{09b5d92c3a564e75803f857bc569832a,
title = "GeneComp, a New Reference-Based Compressor for SAM Files",
abstract = "The affordability of DNA sequencing has led to unprecedented volumes of genomic data. These data must be stored, processed, and analyzed. The most popular format for genomic data is the SAM format, which contains information such as alignment, quality values, etc. These files are large (on the order of terabytes), which necessitates compression. In this work we propose a new reference-based compressor for SAM files, which can accommodate different levels of compression, based on the specific needs of the user. In particular, the proposed compressor GeneComp allows the user to perform lossy compression of the quality scores, which have been proven to occupy more than half of the compressed file (when losslessly compressed). We show that the proposed compressor GeneComp overall achieves better compression ratios than previously proposed algorithms when working on lossless mode.",
keywords = "compression, genomic data, sam file",
author = "Reggy Long and Mikel Hernaez and Idoia Ochoa and Tsachy Weissman",
note = "Publisher Copyright: {\textcopyright} 2017 IEEE.; 2017 Data Compression Conference, DCC 2017 ; Conference date: 04-04-2017 Through 07-04-2017",
year = "2017",
month = may,
day = "8",
doi = "10.1109/DCC.2017.76",
language = "English (US)",
series = "Data Compression Conference Proceedings",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "330--339",
editor = "Ali Bilgin and Joan Serra-Sagrista and Marcellin, {Michael W.} and Storer, {James A.}",
booktitle = "Proceedings - DCC 2017, 2017 Data Compression Conference",
address = "United States",
}