From 9d583aa755235aaf3f2789716467ebd6f15594b0 Mon Sep 17 00:00:00 2001 From: noamteyssier <22600644+noamteyssier@users.noreply.github.com> Date: Wed, 26 Aug 2026 13:45:08 -0700 Subject: [PATCH] docs: simplify documentation and focus around cbq - remove examples focused around vbq and focus on cbq - reduce large multiline comments into simpler one-liners - update main readme to focus on cbq - update changelog --- CHANGELOG.md | 1 + README.md | 24 +- src/bq/header.rs | 145 ++------- src/bq/mod.rs | 216 ++------------ src/bq/reader.rs | 338 +++------------------ src/bq/writer.rs | 163 +--------- src/cbq/core/block.rs | 32 +- src/cbq/core/header.rs | 2 +- src/cbq/core/index.rs | 2 +- src/cbq/mod.rs | 51 ++-- src/cbq/write.rs | 31 +- src/error.rs | 63 +--- src/lib.rs | 74 ++--- src/parallel.rs | 33 +-- src/policy.rs | 80 +---- src/record/binseq_record.rs | 34 +-- src/record/sequencing_record.rs | 38 +-- src/utils/fastx.rs | 104 +------ src/vbq/header.rs | 259 +++------------- src/vbq/index.rs | 303 ++----------------- src/vbq/mod.rs | 58 +--- src/vbq/reader.rs | 507 +++----------------------------- src/vbq/writer.rs | 427 ++------------------------- src/write.rs | 89 ++---- 24 files changed, 412 insertions(+), 2662 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4e72fdb..de50e2f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Large cleanup of internal and public API. - Runtime dependencies removed (`byteorder`, `num_cpus`, `auto_impl`, and `memchr`). - Deprecated items removed from the public API. +- Documentation overhaul: docs now center on CBQ as the recommended variant (VBQ is documented as superseded), examples use CBQ throughout, and docstrings are heavily simplified across the crate. ### Removed diff --git a/README.md b/README.md index 0774464..9211c5c 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# BINSEQ Format Specification +# BINSEQ [![MIT licensed](https://img.shields.io/badge/license-MIT-blue.svg)](./LICENSE.md) ![actions status](https://github.com/arcinstitute/binseq/workflows/CI/badge.svg) @@ -7,25 +7,23 @@ ## Overview -BINSEQ is a binary file format family designed for efficient storage and processing of DNA sequences. -They make use of two-bit encoding for nucleotides and are optimized for high-performance parallel processing. +BINSEQ is a binary file format family for efficient storage and processing of DNA sequences. +It uses two-bit encoding for nucleotides and is optimized for high-performance parallel processing. -BINSEQ has three variants: +The recommended variant is **CBQ** (`*.cbq`): a columnar, block-compressed format for variable-length records with optional quality scores and headers. +It is lossless by default (native `N` support), compresses well, and decodes fast. +For details on its structure see the [documentation](https://docs.rs/binseq/latest/binseq/cbq/). -1. **BQ**: (`*.bq`) files are for _fixed-length_ records **without** quality scores. -2. **VBQ**: (`*.vbq`) files are for _variable-length_ records **with optional** quality scores and headers. -3. **CBQ**: (`*.cbq`) files are for _columnar variable-length_ records **with optional** quality scores and headers. +Two earlier variants remain supported: -All variants support both single and paired sequences. +- **BQ** (`*.bq`): fixed-length records without quality scores. Minimal and fast for uniform reads. +- **VBQ** (`*.vbq`): variable-length records with optional quality scores and headers. Superseded by CBQ, which improves on its compression and decoding speed; new projects should use CBQ. -**Note:** For most use cases, the newest variant _CBQ_ is recommended due to its flexibility, storage efficiency, and decoding speed. -It supersedes _VBQ_ in terms of performance and storage efficiency, at a small cost in encoding speed. -VBQ will still be supported but newer projects should consider using _CBQ_ instead. -For information on the structure of _CBQ_ files, see the [documentation](https://docs.rs/binseq/latest/binseq/cbq/). +All variants support both single and paired sequences. ## Getting Started -This is a **library** for reading and writing BINSEQ files, for a **command-line interface** see [bqtools](https://github.com/arcinstitute/bqtools). +This is a **library** for reading and writing BINSEQ files; for a **command-line interface** see [bqtools](https://github.com/arcinstitute/bqtools). To get started please refer to our [documentation](https://docs.rs/binseq/latest/binseq/). For example programs which make use of the library check out our [examples directory](https://github.com/arcinstitute/binseq/tree/main/examples). diff --git a/src/bq/header.rs b/src/bq/header.rs index f55d3ab..995e100 100644 --- a/src/bq/header.rs +++ b/src/bq/header.rs @@ -1,8 +1,4 @@ -//! Header module for the binseq library -//! -//! This module provides the header structure and functionality for binary sequence files. -//! The header contains metadata about the binary sequence data, including format version, -//! sequence length, and other information necessary for proper interpretation of the data. +//! Fixed-size file header for BQ files. use bitnuc_deprec::BitSize; use std::io::{Read, Write}; @@ -12,30 +8,17 @@ use crate::{ utils::read_u32_le, }; -/// Current magic number: "BSEQ" in ASCII (in little-endian byte order) -/// -/// This is used to identify binary sequence files and verify file integrity. -#[allow(clippy::unreadable_literal)] -const MAGIC: u32 = 0x51455342; +/// The magic bytes at the start of a BQ file on disk. +pub const FILE_MAGIC: [u8; 4] = *b"BSEQ"; +const MAGIC: u32 = u32::from_le_bytes(FILE_MAGIC); -/// The magic bytes as they appear at the start of a BQ file on disk. -/// -/// Used to identify BQ files by content rather than by file extension. -pub const FILE_MAGIC: [u8; 4] = MAGIC.to_le_bytes(); - -/// Current format version of the binary sequence file format -/// -/// This version number allows for future format changes while maintaining backward compatibility. +/// Current format version const FORMAT: u8 = 1; /// Size of the header in bytes -/// -/// The header has a fixed size to ensure consistent reading and writing of binary sequence files. pub const SIZE_HEADER: usize = 32; -/// Reserved bytes in the header -/// -/// These bytes are reserved for future use and should be set to a consistent value. +/// Reserved bytes in the header (future use) pub const RESERVED: [u8; 17] = [42; 17]; #[derive(Debug, Clone, Copy)] @@ -98,48 +81,28 @@ impl FileHeaderBuilder { } } -/// Header structure for binary sequence files -/// -/// The `FileHeader` contains metadata about the binary sequence data stored in a file, -/// including format information, sequence lengths, and space for future extensions. -/// -/// The total size of this structure is 32 bytes, with a fixed layout to ensure -/// consistent reading and writing across different platforms. +/// Fixed 32-byte header for BQ files. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct FileHeader { - /// Magic number to identify the file format - /// - /// 4 bytes + /// Magic number identifying the file format (4 bytes) pub magic: u32, - /// Version of the file format - /// - /// 1 byte + /// Format version (1 byte) pub format: u8, - /// Length of all sequences in the file - /// - /// 4 bytes + /// Primary sequence length (4 bytes) pub slen: u32, - /// Length of secondary sequences in the file - /// - /// 4 bytes + /// Secondary sequence length (4 bytes) pub xlen: u32, - /// Number of bits per nucleotide (currently 2 or 4) - /// - /// 1 byte + /// Bits per nucleotide, 2 or 4 (1 byte) pub bits: BitSize, - /// All records have a flag attribute - /// - /// 1 byte + /// Whether all records carry a flag attribute (1 byte) pub flags: bool, - /// Reserve remaining bytes for future use - /// - /// 17 bytes + /// Reserved for future use (17 bytes) pub reserved: [u8; 17], } impl FileHeader { @@ -149,26 +112,7 @@ impl FileHeader { self.xlen > 0 } - /// Parses a header from a fixed-size byte array - /// - /// This method validates the magic number and format version before constructing - /// a header instance. If validation fails, appropriate errors are returned. - /// - /// # Arguments - /// - /// * `buffer` - A byte array of exactly `SIZE_HEADER` bytes containing the header data - /// - /// # Returns - /// - /// * `Ok(FileHeader)` - A valid header parsed from the buffer - /// * `Err(Error)` - If the buffer contains invalid header data - /// - /// # Errors - /// - /// Returns an error if: - /// * The magic number is incorrect - /// * The format version is unsupported - /// * The reserved bytes are invalid + /// Parses and validates a header from a fixed-size byte array pub fn from_bytes(buffer: &[u8; SIZE_HEADER]) -> Result { let magic = read_u32_le(&buffer[0..4]); if magic != MAGIC { @@ -200,26 +144,7 @@ impl FileHeader { }) } - /// Parses a header from an arbitrarily sized buffer - /// - /// This method extracts the header from the beginning of a buffer that may be larger - /// than the header size. It checks that the buffer is at least as large as the header - /// before attempting to parse it. - /// - /// # Arguments - /// - /// * `buffer` - A byte slice containing at least `SIZE_HEADER` bytes - /// - /// # Returns - /// - /// * `Ok(FileHeader)` - A valid header parsed from the buffer - /// * `Err(Error)` - If the buffer is too small or contains invalid header data - /// - /// # Errors - /// - /// Returns an error if: - /// * The buffer is smaller than `SIZE_HEADER` - /// * The header data is invalid (see `from_bytes` for validation details) + /// Parses a header from the first `SIZE_HEADER` bytes of a buffer pub fn from_buffer(buffer: &[u8]) -> Result { let mut bytes = [0u8; SIZE_HEADER]; if buffer.len() < SIZE_HEADER { @@ -229,23 +154,7 @@ impl FileHeader { Self::from_bytes(&bytes) } - /// Writes the header to a writer - /// - /// This method serializes the header to its binary representation and writes it - /// to the provided writer. - /// - /// # Arguments - /// - /// * `writer` - Any type that implements the `Write` trait - /// - /// # Returns - /// - /// * `Ok(())` - If the header was successfully written - /// * `Err(Error)` - If writing to the writer failed - /// - /// # Errors - /// - /// Returns an error if writing to the writer fails (typically an I/O error). + /// Serializes the header and writes it to a writer pub fn write_bytes(&self, writer: &mut W) -> Result<()> { let mut buffer = [0u8; SIZE_HEADER]; buffer[0..4].copy_from_slice(&self.magic.to_le_bytes()); @@ -259,25 +168,7 @@ impl FileHeader { Ok(()) } - /// Reads a header from a reader - /// - /// This method reads exactly `SIZE_HEADER` bytes from the provided reader and - /// parses them into a header structure. - /// - /// # Arguments - /// - /// * `reader` - Any type that implements the `Read` trait - /// - /// # Returns - /// - /// * `Ok(FileHeader)` - A valid header read from the reader - /// * `Err(Error)` - If reading from the reader failed or the header data is invalid - /// - /// # Errors - /// - /// Returns an error if: - /// * Reading from the reader fails (typically an I/O error) - /// * The header data is invalid (see `from_bytes` for validation details) + /// Reads and parses a header from a reader pub fn from_reader(reader: &mut R) -> Result { let mut buffer = [0u8; SIZE_HEADER]; reader.read_exact(&mut buffer)?; diff --git a/src/bq/mod.rs b/src/bq/mod.rs index 3811e48..2c6e2f8 100644 --- a/src/bq/mod.rs +++ b/src/bq/mod.rs @@ -1,160 +1,62 @@ -//! # bq +//! # BQ Format //! -//! *.bq files are BINSEQ variants for **fixed-length** records and **does not support quality scores**. -//! -//! For variable-length records and optional quality scores use the [`cbq`](crate::cbq) or [`vbq`](crate::vbq) modules. -//! -//! This module contains the utilities for reading, writing, and interacting with BQ files. +//! BQ (`*.bq`) is the BINSEQ variant for **fixed-length** records without +//! quality scores. Its uniform record size gives constant-time random access +//! with minimal overhead. For variable-length records and optional quality +//! scores use [`cbq`](crate::cbq). //! //! For detailed information on the file format, see our [paper](https://www.biorxiv.org/content/10.1101/2025.04.08.647863v1). //! //! ## Usage //! //! ### Reading +//! //! ```rust //! use binseq::{bq, BinseqRecord}; -//! use rand::{thread_rng, Rng}; //! -//! let path = "./data/subset.bq"; -//! let reader = bq::MmapReader::new(path).unwrap(); -//! -//! // We can easily determine the number of records in the file +//! let reader = bq::MmapReader::new("./data/subset.bq").unwrap(); //! let num_records = reader.num_records(); //! -//! // We have random access to any record within the range -//! let random_index = thread_rng().gen_range(0..num_records); -//! let record = reader.get(random_index).unwrap(); +//! // Random access to any record in the file +//! let record = reader.get(num_records / 2).unwrap(); //! -//! // We can easily decode the (2bit)encoded sequence back to a sequence of bytes +//! // Decode the 2-bit encoded sequence back to bytes //! let mut sbuf = Vec::new(); -//! let mut xbuf = Vec::new(); -//! -//! record.decode_s(&mut sbuf); -//! if record.is_paired() { -//! record.decode_x(&mut xbuf); -//! } +//! record.decode_s(&mut sbuf).unwrap(); //! ``` //! //! ### Writing //! -//! #### Writing unpaired sequences -//! //! ```rust //! use binseq::{bq, SequencingRecordBuilder}; //! use std::io::Cursor; //! -//! // Create an in-memory buffer for output -//! let output_handle = Cursor::new(Vec::new()); -//! -//! // Initialize our BQ header (64 bp, only primary) +//! // BQ is fixed-length: sequence lengths are set in the header //! let header = bq::FileHeaderBuilder::new().slen(64).build().unwrap(); -//! -//! // Initialize our BQ writer //! let mut writer = bq::WriterBuilder::default() //! .header(header) -//! .build(output_handle) +//! .build(Cursor::new(Vec::new())) //! .unwrap(); //! -//! // Generate a random sequence //! let seq = [b'A'; 64]; -//! -//! // Build a record and write it to the file //! let record = SequencingRecordBuilder::default() //! .s_seq(&seq) -//! .flag(0) -//! .build() -//! .unwrap(); -//! writer.push(record).unwrap(); -//! -//! // Flush the writer -//! writer.flush().unwrap(); -//! ``` -//! -//! #### Writing paired sequences -//! -//! ```rust -//! use binseq::{bq, SequencingRecordBuilder}; -//! use std::io::Cursor; -//! -//! // Create an in-memory buffer for output -//! let output_handle = Cursor::new(Vec::new()); -//! -//! // Initialize our BQ header (64 bp and 128bp) -//! let header = bq::FileHeaderBuilder::new().slen(64).xlen(128).build().unwrap(); -//! -//! // Initialize our BQ writer -//! let mut writer = bq::WriterBuilder::default() -//! .header(header) -//! .build(output_handle) -//! .unwrap(); -//! -//! // Generate paired sequences -//! let primary = [b'A'; 64]; -//! let secondary = [b'C'; 128]; -//! -//! // Build a paired record and write it to the file -//! let record = SequencingRecordBuilder::default() -//! .s_seq(&primary) -//! .x_seq(&secondary) -//! .flag(0) //! .build() //! .unwrap(); //! writer.push(record).unwrap(); -//! -//! // Flush the writer //! writer.flush().unwrap(); //! ``` //! -//! # Example: Streaming Access +//! Paired records work the same way: set `xlen` on the header and `x_seq` on +//! the record. For streaming over arbitrary readers/writers (e.g. network +//! sockets) see [`StreamReader`](crate::bq::StreamReader) and the +//! `network_streaming` example. //! -//! ``` -//! use binseq::{Policy, Result, BinseqRecord, SequencingRecordBuilder}; -//! use binseq::bq::{FileHeaderBuilder, StreamReader, WriterBuilder}; -//! use std::io::{BufReader, BufWriter, Cursor}; -//! -//! fn main() -> Result<()> { -//! // Create a header for sequences of length 100 -//! let header = FileHeaderBuilder::new().slen(100).build()?; -//! -//! // Create a buffered writer over any `Write` destination -//! let mut writer = WriterBuilder::default() -//! .header(header) -//! .build(BufWriter::new(Cursor::new(Vec::new())))?; -//! -//! // Write sequences -//! let sequence = b"ACGT".repeat(25); // 100 nucleotides -//! let record = SequencingRecordBuilder::default() -//! .s_seq(&sequence) -//! .flag(0) -//! .build()?; -//! writer.push(record)?; -//! -//! // Get the inner buffer -//! let buffer = writer.into_inner().into_inner().map_err(std::io::Error::from)?; -//! let data = buffer.into_inner(); -//! -//! // Create a stream reader -//! let mut reader = StreamReader::new(BufReader::new(Cursor::new(data))); -//! -//! // Process records as they arrive -//! while let Some(record) = reader.next_record() { -//! // Process each record -//! let record = record?; -//! let flag = record.flag(); -//! } -//! -//! Ok(()) -//! } -//! ``` -//! -//! ## BQ file format +//! ## File Format //! -//! A BQ file consists of two sections: +//! A BQ file is a fixed-size header followed by densely packed records. //! -//! 1. Fixed-size header (32 bytes) -//! 2. Record data section -//! -//! ### Header Format (32 bytes total) +//! ### Header (32 bytes) //! //! | Offset | Size (bytes) | Name | Description | Type | //! | ------ | ------------ | -------- | ---------------------------- | ------ | @@ -164,77 +66,19 @@ //! | 9 | 4 | xlen | Sequence length (secondary) | uint32 | //! | 13 | 19 | reserved | Reserved for future use | bytes | //! -//! ### Record Format -//! -//! Each record consists of a: -//! -//! 1. Flag field (8 bytes, uint64) -//! 2. Sequence data (ceil(N/32) \* 8 bytes, where N is sequence length) -//! -//! The flag field is implementation-defined and can be used for filtering, metadata, or other purposes. The placement of the flag field at the start of each record enables efficient filtering without reading sequence data. -//! -//! Total record size = 8 + (ceil(N/32) \* 8) bytes, where N is sequence length -//! -//! ## Encoding -//! -//! - Each nucleotide is encoded using 2 bits: -//! - A = 00 -//! - C = 01 -//! - G = 10 -//! - T = 11 -//! - Non-ATCG characters are **unsupported**. -//! - Sequences are stored in Little-Endian order -//! - The final u64 of sequence data is padded with zeros if the sequence length is not divisible by 32 -//! -//! See [`bitnuc`] for 2bit implementation details. -//! -//! ## bq implementation Notes -//! -//! - Sequences are stored in u64 chunks, each holding up to 32 bases -//! - Random access to any record can be calculated as: -//! - record_size = 8 + (ceil(sequence_length/32) \* 8) -//! - record_start = 16 + (record_index \* record_size) -//! - Total number of records can be calculated as: (file_size - 16) / record_size -//! - Flag field placement allows for efficient filtering strategies: -//! - Records can be skipped based on flag values without reading sequence data -//! - Flag checks can be vectorized for parallel processing -//! - Memory access patterns are predictable for better cache utilization -//! -//! ## Example Storage Requirements -//! -//! Common sequence lengths: -//! -//! - 32bp reads: -//! - Sequence: 1 \* 8 = 8 bytes (fits in one u64) -//! - Flag: 8 bytes -//! - Total per record: 16 bytes -//! - 100bp reads: -//! - Sequence: 4 \* 8 = 32 bytes (requires four u64s) -//! - Flag: 8 bytes -//! - Total per record: 40 bytes -//! - 150bp reads: -//! - Sequence: 5 \* 8 = 40 bytes (requires five u64s) -//! - Flag: 8 bytes -//! - Total per record: 48 bytes -//! -//! ## Validation -//! -//! Implementations should verify: +//! ### Records //! -//! 1. Correct magic number -//! 2. Compatible version number -//! 3. Sequence length is greater than 0 -//! 4. File size minus header (32 bytes) is divisible by the record size +//! Each record is a flag field (8 bytes, uint64, implementation-defined) followed +//! by the encoded sequence data (`ceil(N/32) * 8` bytes for sequence length `N`). +//! The leading flag enables filtering without touching sequence data, and the +//! uniform record size makes random access a constant-time offset calculation. //! -//! ## Future Considerations +//! ### Encoding //! -//! - The 19 reserved bytes in the header allow for future format extensions -//! - The 64-bit flag field provides space for implementation-specific features such as: -//! - Quality score summaries -//! - Filtering flags -//! - Read group identifiers -//! - Processing state -//! - Count data +//! Nucleotides are 2-bit encoded (A=00, C=01, G=10, T=11) into little-endian +//! u64 words of 32 bases each, zero-padded in the final word. Non-ACGT characters +//! are unsupported; see [`Policy`](crate::Policy) for handling options and +//! [`bitnuc`] for implementation details. mod header; mod reader; diff --git a/src/bq/reader.rs b/src/bq/reader.rs index 30a6b86..f72fd66 100644 --- a/src/bq/reader.rs +++ b/src/bq/reader.rs @@ -1,11 +1,4 @@ -//! Binary sequence reader module -//! -//! This module provides functionality for reading binary sequence files using either: -//! 1. Memory mapping for efficient access to entire files -//! 2. Streaming for processing data as it arrives -//! -//! It supports both sequential and parallel processing of records, -//! with configurable record layouts for different sequence types. +//! BQ readers: memory-mapped ([`MmapReader`]) and streaming ([`StreamReader`]). use std::fs::File; use std::io::Read; @@ -23,25 +16,19 @@ use crate::{ error::{ReadError, Result}, }; -/// A reference to a binary sequence record in a memory-mapped file +/// A zero-copy view of a single record in a memory-mapped BQ file. /// -/// This struct provides a view into a single record within a binary sequence file, -/// allowing access to the record's components (sequence data, flags, etc.) without -/// copying the data from the memory-mapped file. -/// -/// The record's data is stored in a compact binary format where: -/// - The first u64 contains flags -/// - Subsequent u64s contain the primary sequence data -/// - If present, final u64s contain the extended sequence data +/// Layout: an optional leading flag u64, then primary sequence u64s, +/// then extended sequence u64s (if paired). #[derive(Clone, Copy)] pub struct RefRecord<'a> { - /// The position (index) of this record in the file (0-based record index, not byte offset) + /// 0-based record index (not byte offset) id: u64, - /// The underlying u64 buffer representing the record's binary data + /// The record's binary data as u64 words buffer: &'a [u64], /// Reusable default quality buffer qbuf: &'a [u8], - /// The configuration that defines the layout and size of record components + /// Layout configuration for the record config: RecordConfig, /// Cached index string for the sequence header header_buf: [u8; 20], @@ -51,12 +38,6 @@ pub struct RefRecord<'a> { impl<'a> RefRecord<'a> { /// Creates a new record reference /// - /// # Arguments - /// - /// * `id` - The record's position in the file (0-based record index, not byte offset) - /// * `buffer` - The u64 slice containing the record's binary data - /// * `config` - Configuration defining the record's layout - /// /// # Panics /// /// Panics if the buffer length doesn't match the expected size from the config @@ -81,12 +62,9 @@ impl BinseqRecord for RefRecord<'_> { fn index(&self) -> u64 { self.id } - /// Clear the buffer and fill it with the sequence header fn sheader(&self) -> &[u8] { &self.header_buf[..self.header_len] } - - /// Clear the buffer and fill it with the extended header fn xheader(&self) -> &[u8] { self.sheader() } @@ -169,7 +147,7 @@ impl BinseqRecord for BatchRecord<'_> { dbuf.extend_from_slice(self.xseq()); Ok(()) } - /// Override this method since we can make use of block information + /// Overridden to slice the precomputed batch-decoded buffer fn sseq(&self) -> &[u8] { let config = &self.inner.config; let scalar = config.scalar(); @@ -181,7 +159,7 @@ impl BinseqRecord for BatchRecord<'_> { } self.dbuf.get(lbound..rbound).unwrap_or_default() } - /// Override this method since we can make use of block information + /// Overridden to slice the precomputed batch-decoded buffer fn xseq(&self) -> &[u8] { let config = &self.inner.config; let scalar = config.scalar(); @@ -201,45 +179,24 @@ impl BinseqRecord for BatchRecord<'_> { } } -/// Configuration for binary sequence record layout -/// -/// This struct defines the size and layout of binary sequence records, -/// including both primary sequence data and optional extended data. -/// It handles the translation between sequence lengths in base pairs -/// and the number of u64 chunks needed to store the compressed data. +/// Size and layout of BQ records, mapping base-pair lengths to u64 chunks. #[derive(Clone, Copy)] pub(crate) struct RecordConfig { - /// The primary sequence length in base pairs + /// Primary sequence length in base pairs slen: u64, - /// The extended sequence length in base pairs + /// Extended sequence length in base pairs xlen: u64, - /// The number of u64 chunks needed to store the primary sequence - /// (each u64 stores 32 nucleotides) + /// u64 chunks needed for the primary sequence schunk: u64, - /// The number of u64 chunks needed to store the extended sequence - /// (each u64 stores 32 values) + /// u64 chunks needed for the extended sequence xchunk: u64, - /// The bitsize of the record + /// Bits per nucleotide bitsize: BitSize, /// Whether flags are present flags: bool, } impl RecordConfig { /// Creates a new record configuration - /// - /// This constructor initializes a configuration for a binary sequence record - /// with specified primary and extended sequence lengths. - /// - /// # Arguments - /// - /// * `slen` - The length of primary sequences in the file - /// * `xlen` - The length of secondary/extended sequences in the file - /// * `bitsize` - The bitsize of the record - /// * `flags` - Whether flags are present - /// - /// # Returns - /// - /// A new `RecordConfig` instance with the specified sequence lengths pub fn new(slen: usize, xlen: usize, bitsize: BitSize, flags: bool) -> Self { let (schunk, xchunk) = match bitsize { BitSize::Two => (slen.div_ceil(32), xlen.div_ceil(32)), @@ -255,18 +212,7 @@ impl RecordConfig { } } - /// Creates a new record configuration from a header - /// - /// This constructor initializes a configuration based on a header that contains - /// the sequence lengths for primary and extended sequences. - /// - /// # Arguments - /// - /// * `header` - A reference to a `FileHeader` containing sequence lengths - /// - /// # Returns - /// - /// A new `RecordConfig` instance with the sequence lengths from the header + /// Creates a record configuration from a file header pub fn from_header(header: &FileHeader) -> Self { Self::new( header.slen as usize, @@ -276,44 +222,32 @@ impl RecordConfig { ) } - /// Returns the primary sequence length in base pairs - /// - /// This method returns the length of the primary sequence in base pairs. + /// Primary sequence length in base pairs pub fn slen(&self) -> usize { self.slen as usize } - /// Returns the extended sequence length in base pairs - /// - /// This method returns the length of the extended sequence in base pairs. + /// Extended sequence length in base pairs pub fn xlen(&self) -> usize { self.xlen as usize } - /// Returns the number of u64 chunks needed to store the primary sequence - /// - /// This method returns the number of u64 chunks required to store the primary - /// sequence, where each u64 stores 32 nucleotides. + /// u64 chunks needed for the primary sequence pub fn schunk(&self) -> usize { self.schunk as usize } - /// Returns the number of u64 chunks needed to store the extended sequence - /// - /// This method returns the number of u64 chunks required to store the extended - /// sequence, where each u64 stores 32 values. + /// u64 chunks needed for the extended sequence pub fn xchunk(&self) -> usize { self.xchunk as usize } - /// Returns the full record size in bytes (u8): - /// 8 * (schunk + xchunk + 1 (flag)) + /// Full record size in bytes pub fn record_size_bytes(&self) -> usize { 8 * self.record_size_u64() } - /// Returns the full record size in u64 - /// schunk + xchunk + 1 (flag) + /// Full record size in u64 words (schunk + xchunk + optional flag word) pub fn record_size_u64(&self) -> usize { if self.flags { (self.schunk + self.xchunk + 1) as usize @@ -331,16 +265,9 @@ impl RecordConfig { } } -/// A memory-mapped reader for binary sequence files -/// -/// This reader provides efficient access to binary sequence files by memory-mapping -/// them instead of performing traditional I/O operations. It supports both -/// sequential access to individual records and parallel processing of records -/// across multiple threads. -/// -/// The reader ensures thread-safety through the use of `Arc` for sharing the -/// memory-mapped data between threads. +/// A memory-mapped reader for BQ files /// +/// Supports random access to individual records and parallel processing. /// Records are returned as [`RefRecord`] which implement the [`BinseqRecord`] trait. /// /// # Examples @@ -350,27 +277,20 @@ impl RecordConfig { /// use binseq::Result; /// /// fn main() -> Result<()> { -/// let path = "./data/subset.bq"; -/// let reader = MmapReader::new(path)?; -/// -/// // Calculate the number of records in the file -/// let num_records = reader.num_records(); -/// println!("Number of records: {}", num_records); -/// -/// // Get the record at index 20 (0-indexed) +/// let reader = MmapReader::new("./data/subset.bq")?; +/// println!("Number of records: {}", reader.num_records()); /// let record = reader.get(20)?; -/// /// Ok(()) /// } /// ``` pub struct MmapReader { - /// Memory mapped file contents, wrapped in Arc for thread-safe sharing + /// Memory-mapped file contents mmap: Arc, - /// Binary sequence file header containing format information + /// File header header: FileHeader, - /// Configuration defining the layout of records in the file + /// Record layout configuration config: RecordConfig, /// Reusable buffer for quality scores @@ -381,27 +301,7 @@ pub struct MmapReader { } impl MmapReader { - /// Creates a new memory-mapped reader for a binary sequence file - /// - /// This method opens the file, memory-maps its contents, and validates - /// the file structure to ensure it contains valid binary sequence data. - /// - /// # Arguments - /// - /// * `path` - Path to the binary sequence file - /// - /// # Returns - /// - /// * `Ok(MmapReader)` - A new reader if the file is valid - /// * `Err(Error)` - If the file is invalid or cannot be opened - /// - /// # Errors - /// - /// Returns an error if: - /// * The file cannot be opened - /// * The file is not a regular file - /// * The file header is invalid - /// * The file size doesn't match the expected size based on the header + /// Opens and memory-maps a BQ file, validating its header and size. pub fn new>(path: P) -> Result { // Verify input file is a file before attempting to map let file = File::open(path)?; @@ -436,17 +336,12 @@ impl MmapReader { } /// Returns the total number of records in the file - /// - /// This is calculated by subtracting the header size from the total file size - /// and dividing by the size of each record. #[must_use] pub fn num_records(&self) -> usize { (self.mmap.len() - SIZE_HEADER) / self.config.record_size_bytes() } - /// Returns a copy of the binary sequence file header - /// - /// The header contains format information and sequence length specifications. + /// Returns a copy of the file header #[must_use] pub fn header(&self) -> FileHeader { self.header @@ -470,20 +365,7 @@ impl MmapReader { vec![self.default_quality_score; self.header.slen.max(self.header.xlen) as usize] } - /// Returns a reference to a specific record - /// - /// # Arguments - /// - /// * `idx` - The index of the record to retrieve (0-based) - /// - /// # Returns - /// - /// * `Ok(RefRecord)` - A reference to the requested record - /// * `Err(Error)` - If the index is out of bounds - /// - /// # Errors - /// - /// Returns an error if the requested index is beyond the number of records in the file + /// Returns a reference to the record at a 0-based index pub fn get(&self, idx: usize) -> Result> { if idx > self.num_records() { return Err(ReadError::OutOfRange { @@ -500,10 +382,7 @@ impl MmapReader { Ok(RefRecord::new(idx as u64, buffer, &self.qbuf, self.config)) } - /// Returns a slice of the buffer containing the underlying u64 for that range - /// of records. - /// - /// Note: range 10..40 will return all u64s in the mmap between the record index 10 and 40 + /// Returns the underlying u64 slice spanning a range of record indices pub fn get_buffer_slice(&self, range: Range) -> Result<&[u64]> { if range.end > self.num_records() { return Err(ReadError::OutOfRange { @@ -522,24 +401,19 @@ impl MmapReader { } } -/// A reader for streaming binary sequence data from any source that implements Read +/// A streaming BQ reader over any `Read` source /// -/// Unlike `MmapReader` which requires the entire file to be accessible at once, -/// `StreamReader` processes data as it becomes available, making it suitable for: -/// - Processing data as it arrives over a network -/// - Handling very large files that exceed available memory -/// - Pipeline processing where data is flowing continuously -/// -/// The reader maintains an internal buffer and can handle partial record reconstruction -/// across chunk boundaries. +/// Unlike [`MmapReader`], data is processed as it arrives, so it suits +/// network streams, pipes, and files larger than memory. Records spanning +/// buffer refills are handled transparently. pub struct StreamReader { - /// The source reader for binary sequence data + /// The source reader reader: R, - /// Binary sequence file header containing format information + /// File header (populated on first read) header: Option, - /// Configuration defining the layout of records in the file + /// Record layout configuration (populated with the header) config: Option, /// Buffer for storing incoming data @@ -567,35 +441,12 @@ pub struct StreamReader { } impl StreamReader { - /// Creates a new `StreamReader` with the default buffer size - /// - /// This constructor initializes a `StreamReader` that will read from the provided - /// source, using an 8K default buffer size. - /// - /// # Arguments - /// - /// * `reader` - The source to read binary sequence data from - /// - /// # Returns - /// - /// A new `StreamReader` instance + /// Creates a new `StreamReader` with the default (8K) buffer size pub fn new(reader: R) -> Self { Self::with_capacity(reader, 8192) } - /// Creates a new `StreamReader` with a specified buffer capacity - /// - /// This constructor initializes a `StreamReader` with a custom buffer size, - /// which can be tuned based on the expected usage pattern. - /// - /// # Arguments - /// - /// * `reader` - The source to read binary sequence data from - /// * `capacity` - The size of the internal buffer in bytes - /// - /// # Returns - /// - /// A new `StreamReader` instance with the specified buffer capacity + /// Creates a new `StreamReader` with a specified buffer capacity in bytes pub fn with_capacity(reader: R, capacity: usize) -> Self { Self { reader, @@ -618,26 +469,11 @@ impl StreamReader { self.default_quality_score = score; } - /// Reads and validates the header from the underlying reader - /// - /// This method reads the binary sequence file header and validates it. - /// It caches the header internally for future use. - /// - /// # Returns - /// - /// * `Ok(&FileHeader)` - A reference to the validated header - /// * `Err(Error)` - If reading or validating the header fails + /// Reads, validates, and caches the file header /// /// # Panics /// /// Panics if the header is missing when expected in the stream. - /// - /// # Errors - /// - /// Returns an error if: - /// * There is an I/O error when reading from the source - /// * The header data is invalid - /// * End of stream is reached before the full header can be read pub fn read_header(&mut self) -> Result<&FileHeader> { if self.header.is_none() { // Ensure we have enough data for the header @@ -660,21 +496,7 @@ impl StreamReader { .expect("header was just populated above")) } - /// Fills the internal buffer with more data from the reader - /// - /// This method reads more data from the underlying reader, handling - /// the case where some unprocessed data remains in the buffer. - /// - /// # Returns - /// - /// * `Ok(())` - If the buffer was successfully filled with new data - /// * `Err(Error)` - If reading from the source fails - /// - /// # Errors - /// - /// Returns an error if: - /// * There is an I/O error when reading from the source - /// * End of stream is reached (no more data available) + /// Refills the internal buffer, shifting any unprocessed bytes to the front fn fill_buffer(&mut self) -> Result<()> { // Move remaining data to beginning of buffer if needed if self.buffer_pos > 0 && self.buffer_pos < self.buffer_len { @@ -696,27 +518,11 @@ impl StreamReader { Ok(()) } - /// Retrieves the next record from the stream - /// - /// This method reads and processes the next complete record from the stream. - /// It handles the case where a record spans multiple buffer fills. - /// - /// # Returns - /// - /// * `Ok(Some(RefRecord))` - The next record was successfully read - /// * `Ok(None)` - End of stream was reached (no more records) - /// * `Err(Error)` - If an error occurred during reading + /// Retrieves the next record from the stream, or `None` at end of stream /// /// # Panics /// /// Panics if the configuration is missing when expected in the stream. - /// - /// # Errors - /// - /// Returns an error if: - /// * There is an I/O error when reading from the source - /// * The header has not been read yet - /// * The data format is invalid pub fn next_record(&mut self) -> Option>> { // Ensure header is read if self.header.is_none() @@ -769,45 +575,16 @@ impl StreamReader { } /// Consumes the stream reader and returns the inner reader - /// - /// This method is useful when you need access to the underlying reader - /// after processing is complete. - /// - /// # Returns - /// - /// The inner reader that was used by this `StreamReader` pub fn into_inner(self) -> R { self.reader } } -/// Default batch size for parallel processing -/// -/// This constant defines how many records each thread processes at a time -/// during parallel processing operations. +/// Number of records each thread processes at a time during parallel processing pub(crate) const BATCH_SIZE: usize = 1024; -/// Parallel processing implementation for memory-mapped readers impl ParallelReader for MmapReader { - /// Processes all records in parallel using multiple threads - /// - /// This method distributes the records across the specified number of threads - /// and processes them using the provided processor. Each thread receives its - /// own clone of the processor and processes a contiguous chunk of records. - /// - /// # Arguments - /// - /// * `processor` - The processor to use for handling records - /// * `num_threads` - The number of threads to use for processing - /// - /// # Type Parameters - /// - /// * `P` - A type that implements `ParallelProcessor` and can be cloned - /// - /// # Returns - /// - /// * `Ok(())` - If all records were processed successfully - /// * `Err(Error)` - If an error occurred during processing + /// Processes all records in parallel across the given number of threads fn process_parallel( self, processor: P, @@ -817,26 +594,7 @@ impl ParallelReader for MmapReader { self.process_parallel_range(processor, num_threads, 0..num_records) } - /// Process records in parallel within a specified range - /// - /// This method allows parallel processing of a subset of records within the file, - /// defined by a start and end index. The range is distributed across the specified - /// number of threads. - /// - /// # Arguments - /// - /// * `processor` - The processor to use for each record - /// * `num_threads` - The number of threads to spawn - /// * `range` - The range of record indices to process - /// - /// # Type Parameters - /// - /// * `P` - A type that implements `ParallelProcessor` and can be cloned - /// - /// # Returns - /// - /// * `Ok(())` - If all records were processed successfully - /// * `Err(Error)` - If an error occurred during processing + /// Processes a range of record indices in parallel fn process_parallel_range( self, processor: P, diff --git a/src/bq/writer.rs b/src/bq/writer.rs index 641073f..b5fbeba 100644 --- a/src/bq/writer.rs +++ b/src/bq/writer.rs @@ -1,11 +1,4 @@ -//! Binary sequence writer module -//! -//! This module provides functionality for writing nucleotide sequences to binary files -//! in a compact 2-bit format. It includes support for: -//! - Single and paired sequence writing -//! - Invalid nucleotide handling with configurable policies -//! - Efficient buffering and encoding -//! - Headless mode for parallel writing +//! BQ writer for encoding nucleotide sequences into fixed-length binary records. use std::io::Write; @@ -23,11 +16,7 @@ fn write_buffer(writer: &mut W, ebuf: &[u64]) -> Result<()> { Ok(()) } -/// Builder for creating configured `Writer` instances -/// -/// This builder provides a flexible way to create writers with various -/// configurations. It follows the builder pattern, allowing for optional -/// settings to be specified in any order. +/// Builder for configured [`Writer`] instances /// /// # Examples /// @@ -48,9 +37,9 @@ fn write_buffer(writer: &mut W, ebuf: &[u64]) -> Result<()> { pub struct WriterBuilder { /// Required header defining sequence lengths and format header: Option, - /// Optional policy for handling invalid nucleotides + /// Policy for handling invalid nucleotides policy: Option, - /// Optional headless mode for parallel writing scenarios + /// Headless mode (skip writing the header) for parallel writing headless: Option, } impl WriterBuilder { @@ -85,19 +74,10 @@ impl WriterBuilder { } } -/// High-level writer for binary sequence files -/// -/// This writer provides a convenient interface for writing nucleotide sequences -/// to binary files in a compact format. It handles sequence encoding, invalid -/// nucleotide processing, and file format compliance. -/// -/// The writer can operate in two modes: -/// - Normal mode: Writes the header followed by records -/// - Headless mode: Writes only records (useful for parallel writing) +/// Writer for BQ files /// -/// # Type Parameters -/// -/// * `W` - The underlying writer type that implements `Write` +/// Headless mode skips writing the file header, for parallel writing where +/// only one writer should emit it. #[derive(Clone)] pub struct Writer { /// The underlying writer for output @@ -109,44 +89,11 @@ pub struct Writer { /// Encoder for converting sequences to binary format encoder: Encoder, - /// Whether this writer is in headless mode - /// When true, the header is not written to the output + /// Whether this writer is in headless mode (header not written) headless: bool, } impl Writer { - /// Creates a new `Writer` instance with specified configuration - /// - /// This is a low-level constructor. For a more convenient way to create a - /// `Writer`, use the `WriterBuilder` struct. - /// - /// # Arguments - /// - /// * `inner` - The underlying writer to write to - /// * `header` - The header defining sequence lengths and format - /// * `policy` - The policy for handling invalid nucleotides - /// * `headless` - Whether to skip writing the header (for parallel writing) - /// - /// # Returns - /// - /// * `Ok(Writer)` - A new writer instance - /// * `Err(Error)` - If writing the header fails - /// - /// # Examples - /// - /// ``` - /// # use binseq::bq::{FileHeaderBuilder, Writer}; - /// # use binseq::{Result, Policy}; - /// # fn main() -> Result<()> { - /// let header = FileHeaderBuilder::new().slen(100).build()?; - /// let writer = Writer::new( - /// Vec::new(), - /// header, - /// Policy::default(), - /// false - /// )?; - /// # Ok(()) - /// # } - /// ``` + /// Creates a new `Writer`; prefer [`WriterBuilder`] for most uses. pub fn new(mut inner: W, header: FileHeader, policy: Policy, headless: bool) -> Result { if !headless { header.write_bytes(&mut inner)?; @@ -174,42 +121,10 @@ impl Writer { self.encoder.policy() } - /// Writes a record using the unified [`SequencingRecord`] API - /// - /// This method provides a consistent interface with VBQ and CBQ writers. - /// Note that BQ format does not support quality scores or headers - these - /// fields from the record will be ignored. - /// - /// # Arguments - /// - /// * `record` - A [`SequencingRecord`] containing the sequence data to write - /// - /// # Returns - /// - /// * `Ok(true)` if the record was written successfully - /// * `Ok(false)` if the record was skipped due to invalid nucleotides - /// * `Err(_)` if writing failed - /// - /// # Examples + /// Writes a [`SequencingRecord`], returning `false` if it was skipped + /// due to invalid nucleotides. /// - /// ``` - /// # use binseq::bq::{FileHeaderBuilder, WriterBuilder}; - /// # use binseq::{Result, SequencingRecordBuilder}; - /// # fn main() -> Result<()> { - /// let header = FileHeaderBuilder::new().slen(8).build()?; - /// let mut writer = WriterBuilder::default() - /// .header(header) - /// .build(Vec::new())?; - /// - /// let record = SequencingRecordBuilder::default() - /// .s_seq(b"ACGTACGT") - /// .flag(42) - /// .build()?; - /// - /// writer.push(record)?; - /// # Ok(()) - /// # } - /// ``` + /// BQ does not store quality scores or headers; those fields are ignored. pub fn push(&mut self, record: SequencingRecord) -> Result { let has_flag = self.header.flags; if has_flag { @@ -267,76 +182,28 @@ impl Writer { } /// Consumes the writer and returns the underlying writer - /// - /// This is useful when you need to access the underlying writer after - /// writing is complete, for example to get the contents of a `Vec`. - /// - /// # Examples - /// - /// ``` - /// # use binseq::bq::{FileHeaderBuilder, WriterBuilder}; - /// # use binseq::Result; - /// # fn main() -> Result<()> { - /// let header = FileHeaderBuilder::new().slen(100).build()?; - /// let writer = WriterBuilder::default() - /// .header(header) - /// .build(Vec::new())?; - /// - /// // After writing sequences... - /// let bytes = writer.into_inner(); - /// # Ok(()) - /// # } - /// ``` pub fn into_inner(self) -> W { self.inner } /// Gets a mutable reference to the underlying writer - /// - /// This allows direct access to the underlying writer while retaining - /// ownership of the `Writer`. pub fn by_ref(&mut self) -> &mut W { &mut self.inner } /// Flushes any buffered data to the underlying writer - /// - /// # Returns - /// - /// * `Ok(())` - If the flush was successful - /// * `Err(Error)` - If flushing failed pub fn flush(&mut self) -> Result<()> { self.inner.flush()?; Ok(()) } - /// Checks if this writer is in headless mode - /// - /// In headless mode, the writer does not write the header to the output. - /// This is useful for parallel writing scenarios where only one writer - /// should write the header. - /// - /// # Returns - /// - /// `true` if the writer is in headless mode, `false` otherwise + /// Checks if this writer is in headless mode (header not written) pub fn is_headless(&self) -> bool { self.headless } - /// Ingests the contents of another writer's buffer - /// - /// This method is used in parallel writing scenarios to combine the output - /// of multiple writers. It takes the contents of another writer's buffer - /// and writes them to this writer's output. - /// - /// # Arguments - /// - /// * `other` - Another writer whose underlying writer is a `Vec` - /// - /// # Returns - /// - /// * `Ok(())` - If the contents were successfully ingested - /// * `Err(Error)` - If writing the contents failed + /// Drains another writer's buffer into this writer's output, for + /// combining parallel writers. pub fn ingest(&mut self, other: &mut Writer>) -> Result<()> { let other_inner = other.by_ref(); self.inner.write_all(other_inner)?; diff --git a/src/cbq/core/block.rs b/src/cbq/core/block.rs index d4768fd..d83f351 100644 --- a/src/cbq/core/block.rs +++ b/src/cbq/core/block.rs @@ -30,10 +30,7 @@ pub struct ColumnarBlock { /// Position of all N's in the sequence pub(crate) npos: Vec, - /// Reusable buffer for encoding sequences - /// - /// Byte-native 2-bit packing, zero-padded to whole 8-byte words to - /// remain bit-identical with the legacy u64-per-32-bases on-disk layout. + /// Reusable 2-bit encoding buffer, zero-padded to whole 8-byte words (legacy u64 on-disk layout) ebuf: Vec, /// An Elias-Fano encoding for the N-positions @@ -56,16 +53,10 @@ pub struct ColumnarBlock { l_seq_offsets: Vec, l_header_offsets: Vec, - /// Number of records in the block - /// - /// A record is a logical unit of data. - /// If the records are paired sequences this is the number of pairs. + /// Number of records in the block (a pair counts as one record) pub(crate) num_records: usize, - /// Number of sequences in the block - /// - /// This is the same as the number of records for unpaired sequences. - /// For paired sequences it will be twice the number of records. + /// Number of sequences in the block (twice `num_records` when paired) pub(crate) num_sequences: usize, /// Total nucleotides in this block @@ -77,9 +68,7 @@ pub struct ColumnarBlock { qbuf: Vec, default_quality_score: u8, - /// The file header (used for block configuration) - /// - /// Not to be confused with the `BlockHeader` + /// The file header used for block configuration (not the `BlockHeader`) pub(crate) header: FileHeader, } impl ColumnarBlock { @@ -349,10 +338,7 @@ impl ColumnarBlock { Ok(()) } - /// Returns the expected length of the encoded sequence buffer in bytes - /// - /// This is deterministically calculated based on the sequence length and the encoding scheme. - /// The on-disk format packs 32 bases per 8-byte word, so the buffer is padded to whole words. + /// Expected encoded sequence buffer length in bytes (32 bases per 8-byte word, padded to whole words). fn ebuf_len(&self) -> usize { self.nuclen.div_ceil(32) * 8 } @@ -437,12 +423,10 @@ impl ColumnarBlock { Ok(()) } - /// Decompress all columns back to native representation + /// Decompress all columns back to native representation. /// - /// Note: `resize` can be only be used with `copy_decode` if passing - /// as `&mut [T]`. Passing a resized `&mut Vec` will lead to an - /// append operation, not an overwrite. If passing `&mut Vec`, the - /// `Vec` will be resized automatically by `copy_decode`. + /// Note: `copy_decode` into a `&mut Vec` appends; pre-resized buffers + /// must be passed as `&mut [T]` to be overwritten. pub fn decompress_columns(&mut self) -> Result<()> { // decompress sequence lengths { diff --git a/src/cbq/core/header.rs b/src/cbq/core/header.rs index 5860a58..519fd1d 100644 --- a/src/cbq/core/header.rs +++ b/src/cbq/core/header.rs @@ -127,7 +127,7 @@ impl FileHeader { } } -/// A convenience struct for building a [`FileHeader`](crate::cbq::FileHeader) using a builder pattern. +/// Builder for a [`FileHeader`](crate::cbq::FileHeader). #[derive(Default)] pub struct FileHeaderBuilder { compression_level: Option, diff --git a/src/cbq/core/index.rs b/src/cbq/core/index.rs index 29ed26b..4a40ac4 100644 --- a/src/cbq/core/index.rs +++ b/src/cbq/core/index.rs @@ -176,7 +176,7 @@ impl Index { } } -/// A struct representing a block range in a CBQ file and stored in the [`Index`](crate::cbq::Index) +/// A block range stored in the [`Index`](crate::cbq::Index). /// /// This is stored identically in memory and on disk. #[derive(Clone, Copy, Debug, PartialEq, Eq, Zeroable, Pod, Default)] diff --git a/src/cbq/mod.rs b/src/cbq/mod.rs index f59b90a..3ce22fa 100644 --- a/src/cbq/mod.rs +++ b/src/cbq/mod.rs @@ -1,35 +1,35 @@ //! # CBQ Format //! -//! CBQ is a high-performance binary format built around blocked columnar storage. -//! It optimizes for storage efficiency and parallel processing of records. +//! CBQ is the recommended BINSEQ variant: a binary format built around blocked +//! columnar storage, optimized for storage efficiency and parallel processing. //! //! ## Overview //! -//! CBQ was built to solve the rough edges of VBQ. -//! It keeps the blocked structure of VBQ, but instead of interleaving the internal data of all records in the block, it stores each attribute in a separate column. -//! Each of these columns are then ZSTD compressed and optionally decoded when reading. +//! Records are grouped into blocks; within each block every record attribute +//! (lengths, sequences, quality scores, headers, flags) is stored as a separate +//! column, and each column is ZSTD-compressed independently. Compared to the +//! row-based VBQ this gives better compression ratios, faster reads +//! (pay-per-use decompression), and simpler record parsing. //! -//! It was built to be performant, efficient, and lossless by default. +//! Sequences are two-bit encoded and lossless by default: the positions of +//! ambiguous nucleotides (`N`) are tracked in an Elias-Fano encoded column and +//! backfilled on decode. //! -//! This has a few benefits and advantages over VBQ: +//! ## Usage //! -//! 1. Better compression ratios for each individual attribute. -//! 2. Significantly faster throughput for reading (easier decompression + pay-per-use decompression). -//! 3. Simple record parsing and manipulation. -//! -//! Notably this format *only* performs two-bit encoding of sequences. -//! However, it tracks the positions of all ambiguous nucleotides (`N`) within the sequence. -//! When it is decoded and the two-bit encoded sequence is decoded back to nucleotides, the `N` positions are backfilled with `N`. -//! -//! To make use of the sparse-but-clustered nature of the `N`-positions, we make use of an Elias-Fano encoding of the `N`-positions. -//! This encoding is then used to efficiently store and retrieve the positions of `N`s within the sequence. +//! Write CBQ files through [`BinseqWriterBuilder`](crate::BinseqWriterBuilder) +//! (or [`ColumnarBlockWriter`](crate::cbq::ColumnarBlockWriter) directly), and read +//! them with [`MmapReader`](crate::cbq::MmapReader) via the +//! [`ParallelProcessor`](crate::ParallelProcessor) trait — see the crate-level example. //! //! ## File Structure //! -//! A CBQ file consists of a [`FileHeader`](cbq::FileHeader), followed by record blocks and an embedded [`Index`](cbq::Index). -//! Each record block is composed of a [`BlockHeader`](cbq::BlockHeader) which provides metadata about the block, and a [`ColumnarBlock`](cbq::ColumnarBlock) containing the actual data. -//! -//! The [`IndexHeader`](cbq::IndexHeader) and [`IndexFooter`](cbq::IndexFooter) are used to locate and access the data within the file when reading as memory mapped. +//! A CBQ file consists of a [`FileHeader`](crate::cbq::FileHeader), followed by record +//! blocks and an embedded [`Index`](crate::cbq::Index). Each record block is a +//! [`BlockHeader`](crate::cbq::BlockHeader) with block metadata followed by a +//! [`ColumnarBlock`](crate::cbq::ColumnarBlock) of data. The +//! [`IndexHeader`](crate::cbq::IndexHeader) and [`IndexFooter`](crate::cbq::IndexFooter) +//! locate the index for memory-mapped reading. //! //! ```text //! ┌───────────────────┐ @@ -53,16 +53,15 @@ //! //! ## Block Format //! -//! The blocks on-disk are stored as ZSTD compressed data. -//! Each column is ZSTD compressed and stored contiguously next to each other. -//! -//! The [BlockHeader](cbq::BlockHeader) contains the compressed sizes of each of the columns as well as the relevant information for their uncompressed sizes. +//! Each column is ZSTD-compressed and stored contiguously; the +//! [`BlockHeader`](crate::cbq::BlockHeader) records the compressed and +//! uncompressed sizes of every column. //! //! ```text //! [BlockHeader][col1][col2][col3]...[BlockHeader][col1][col2][col3]... //! ``` //! -//! The order of columns in the block is as follows: +//! Column order: //! //! 1. `z_seq_len` - sequence lengths //! 2. `z_header_len` - header lengths (optional) diff --git a/src/cbq/write.rs b/src/cbq/write.rs index 90cb821..c9129b5 100644 --- a/src/cbq/write.rs +++ b/src/cbq/write.rs @@ -62,9 +62,7 @@ impl ColumnarBlockWriter { Ok(writer) } - /// Sets the compression level for Writer - /// - /// Note: only used on init, shouldn't be set by the user + /// Initializes the zstd context (compression level + long-distance matching). fn init_compressor(&mut self) -> Result<()> { // Initialize the compressor with the compression level self.cctx @@ -84,10 +82,9 @@ impl ColumnarBlockWriter { self.block.header } - /// Push a record to the writer + /// Push a record to the writer. /// - /// Returns `Ok(true)` if the record was written successfully. - /// CBQ handles N's explicitly in its encoding, so records are never skipped. + /// Always returns `Ok(true)` on success: CBQ encodes N's explicitly, so records are never skipped. pub fn push(&mut self, record: SequencingRecord) -> Result { if !self.block.can_fit(&record) { self.flush()?; @@ -127,10 +124,8 @@ impl ColumnarBlockWriter { /// Ingest only the *completed* (already-compressed) blocks from `other`. /// /// Unlike [`ingest`](Self::ingest), this never touches either writer's - /// incomplete block, so it performs no zstd compression. The work done - /// under a global lock is reduced to a `write_all` of pre-compressed bytes - /// plus a header copy — compression has already been paid for on the worker - /// thread when `other`'s blocks were flushed in `push`. + /// incomplete block and performs no zstd compression — under a global lock + /// it only writes pre-compressed bytes and copies headers. /// /// `other` keeps building its incomplete block across calls; only its /// completed-block buffer and headers are drained. @@ -152,12 +147,9 @@ impl ColumnarBlockWriter { Ok(()) } - /// Ingests only the *incomplete* (non-compressed) blocks from the `other`. - /// - /// This should not be used in isolation and should be handled from the [`ingest`](Self::ingest) API only - /// to avoid any mistakes. + /// Ingests the *incomplete* (uncompressed) block from `other`. /// - /// [`ingest_completed`](Self::ingest_completed) should always be called first. + /// Only called via [`ingest`](Self::ingest), after [`ingest_completed`](Self::ingest_completed). fn ingest_incompleted(&mut self, other: &mut ColumnarBlockWriter>) -> Result<()> { if other.block.num_records == 0 { return Ok(()); // short-circuit @@ -184,19 +176,14 @@ impl ColumnarBlockWriter { } } -/// Specialized implementation when using a local `Vec` as the inner data structure +/// Methods specific to `Vec`-backed writers. impl ColumnarBlockWriter> { #[must_use] pub fn inner_data(&self) -> &[u8] { &self.inner } - /// Clears only the completed-block state (compressed bytes + headers), - /// leaving the incomplete block intact. - /// - /// Used by [`ingest_completed`](ColumnarBlockWriter::ingest_completed) so a - /// worker thread can keep accumulating records into its in-progress block - /// across batches. + /// Clears the completed-block state (compressed bytes + headers), leaving the incomplete block intact. pub fn clear_completed_data(&mut self) { self.inner.clear(); self.headers.clear(); diff --git a/src/error.rs b/src/error.rs index 27d4089..89c7a37 100644 --- a/src/error.rs +++ b/src/error.rs @@ -61,16 +61,10 @@ pub enum Error { #[derive(thiserror::Error, Debug)] pub enum HeaderError { /// The magic number in the header does not match the expected value - /// - /// # Arguments - /// * `u32` - The invalid magic number that was found #[error("Invalid magic number: {0}")] InvalidMagicNumber(u32), /// The format version in the header is not supported - /// - /// # Arguments - /// * `u8` - The unsupported version number that was found #[error("Invalid format version: {0}")] InvalidFormatVersion(u8), @@ -83,10 +77,6 @@ pub enum HeaderError { InvalidBitSize(u8), /// The size of the data does not match what was specified in the header - /// - /// # Arguments - /// * First `usize` - The actual number of bytes provided - /// * Second `usize` - The expected number of bytes according to the header #[error("Invalid number of bytes provided: {0}. Expected: {1}")] InvalidSize(usize, usize), @@ -103,19 +93,12 @@ pub enum ReadError { IncompatibleFile, /// The file appears to be truncated or corrupted - /// - /// # Arguments - /// * `usize` - The byte position where the truncation was detected #[error( "Number of bytes in file does not match expectation - possibly truncated at byte pos {0}" )] FileTruncation(usize), /// Attempted to access a record index that is beyond the available range - /// - /// # Arguments - /// * First `usize` - The requested record index - /// * Second `usize` - The maximum available record index #[error("Requested record index ({requested_index}) is out of record range ({max_index})")] OutOfRange { requested_index: usize, @@ -130,25 +113,18 @@ pub enum ReadError { EndOfStream, /// A partial record was encountered at the end of a stream - /// - /// # Arguments - /// * `usize` - The number of bytes read in the partial record #[error("Partial record at end of stream ({0} bytes)")] PartialRecord(usize), - /// When a block header contains an invalid magic number - /// - /// The first parameter is the invalid magic number, the second is the position in the file + /// A block header contains an invalid magic number #[error("Unexpected Block Magic Number found: {0} at position {1}")] InvalidBlockMagicNumber(u64, usize), - /// When trying to read a block but reaching the end of the file unexpectedly - /// - /// The parameter is the position in the file where the read was attempted + /// Reached the end of the file unexpectedly while reading a block #[error("Unable to find an expected full block at position {0}")] UnexpectedEndOfFile(usize), - /// When the file metadata doesn't match the expected VBQ format + /// The file metadata doesn't match the expected VBQ format #[error("Unexpected file metadata")] InvalidFileType, @@ -183,18 +159,11 @@ pub enum WriteError { obs_extended: bool, }, - /// The length of the sequence being written does not match what was specified in the header - /// - /// # Fields - /// * `expected` - The sequence length specified in the header - /// * `got` - The actual length of the sequence being written + /// The length of the sequence being written does not match the header #[error("Sequence length ({got}) does not match the header ({expected})")] UnexpectedSequenceLength { expected: u32, got: usize }, /// The sequence contains invalid nucleotide characters - /// - /// # Arguments - /// * `String` - Description of the invalid nucleotides found #[error("Invalid nucleotides found in sequence: {0}")] InvalidNucleotideSequence(String), @@ -202,28 +171,23 @@ pub enum WriteError { #[error("Missing header in writer builder")] MissingHeader, - /// When a record is too large to fit in a block of the configured size - /// - /// The first parameter is the record size, the second is the maximum block size + /// A record is too large to fit in a block of the configured size #[error( "Encountered a record with embedded size {0} but the maximum block size is {1}. Rerun with increased block size." )] RecordSizeExceedsMaximumBlockSize(usize, usize), - /// When trying to ingest blocks with different sizes than expected - /// - /// The first parameter is the expected size, the second is the found size + + /// Attempted to ingest blocks with a different size than expected #[error( "Incompatible block sizes encountered in BlockWriter Ingest. Found ({1}) Expected ({0})" )] IncompatibleBlockSizes(usize, usize), - /// When trying to ingest data with an incompatible header - /// - /// The first parameter is the expected header, the second is the found header + /// Attempted to ingest data with an incompatible header #[error("Incompatible headers found in vbq::Writer::ingest. Found ({1:?}) Expected ({0:?})")] IncompatibleHeaders(crate::vbq::FileHeader, crate::vbq::FileHeader), - /// When building a `SequencingRecord` without a primary sequence + /// A `SequencingRecord` was built without a primary sequence #[error("SequencingRecordBuilder requires a primary sequence (s_seq)")] MissingSequence, @@ -239,14 +203,9 @@ pub enum WriteError { } /// Errors related to VBQ file indexing -/// -/// These errors occur when there are issues with the index of a VBQ file, -/// such as corruption or mismatches with the underlying file. #[derive(thiserror::Error, Debug)] pub enum IndexError { - /// When the magic number in the index doesn't match the expected value - /// - /// The parameter is the invalid magic number that was found + /// The magic number in the index doesn't match the expected value #[error("Invalid magic number: {0}")] InvalidMagicNumber(u64), @@ -299,7 +258,7 @@ pub enum CbqError { #[derive(thiserror::Error, Debug)] pub enum FormatError { - /// When the BINSEQ format could not be determined from a file's magic bytes + /// The BINSEQ format could not be determined from a file's magic bytes #[error("Unable to determine BINSEQ format from magic bytes in file: {0}")] UnrecognizedMagicBytes(String), } diff --git a/src/lib.rs b/src/lib.rs index 5511d60..d8bb826 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -2,44 +2,17 @@ #![cfg_attr(docsrs, doc(auto_cfg))] #![doc = include_str!("../README.md")] //! -//! # BINSEQ +//! # Library //! -//! The `binseq` library provides efficient APIs for working with the [BINSEQ](https://www.biorxiv.org/content/10.1101/2025.04.08.647863v2) file format family. +//! The `binseq` library provides efficient APIs for working with the [BINSEQ](https://www.biorxiv.org/content/10.1101/2025.04.08.647863v2) file format family: //! -//! It offers methods to read and write BINSEQ files, providing: +//! - **[`cbq`]** — the recommended variant: columnar, block-compressed, lossless by default +//! - **[`bq`]** and **[`vbq`]** — earlier variants, still fully supported +//! - [`BinseqRecord`], [`BinseqReader`], and [`BinseqWriter`] — variant-agnostic record, reader, and writer abstractions +//! - [`ParallelProcessor`] — parallel processing of records with arbitrary per-record logic +//! - [`Policy`] — configurable handling of invalid nucleotides when encoding BQ/VBQ (CBQ stores `N` natively) //! -//! - Compact multi-bit encoding and decoding of nucleotide sequences through [`bitnuc`](https://docs.rs/bitnuc/latest/bitnuc/) -//! - Support for both single and paired-end sequences -//! - Abstract [`BinseqRecord`] trait for representing records from all variants -//! - Abstract [`BinseqReader`] enum for processing records from all variants -//! - Abstract [`BinseqWriter`] enum for writing records to all variants -//! - Parallel processing capabilities for arbitrary tasks through the [`ParallelProcessor`] trait. -//! - Configurable [`Policy`] for handling invalid nucleotides (BQ/VBQ, CBQ natively supports `N` nucleotides) -//! -//! ## Recent additions (v0.9.0): -//! -//! ### New variant: CBQ -//! **[`cbq`]** is a new variant of BINSEQ that solves many of the pain points around VBQ. -//! The CBQ format is a columnar-block-based format that offers improved compression and faster processing speeds compared to VBQ. -//! It natively supports `N` nucleotides and avoids the need for additional 4-bit encoding. -//! -//! ### Improved interface for writing records -//! **[`BinseqWriter`]** provides a unified interface for writing records generically to BINSEQ files. -//! This makes use of the new [`SequencingRecord`] which provides a cleaner builder API for writing records to BINSEQ files. -//! -//! ## Recent VBQ Format Changes (v0.7.0+) -//! -//! The VBQ format has undergone significant improvements: -//! -//! - **Embedded Index**: VBQ files now contain their index data embedded at the end of the file, -//! improving portability. -//! - **Headers Support**: Optional sequence identifiers/headers can be stored with each record. -//! - **Extended Capacity**: u64 indexing supports files with more than 4 billion records. -//! - **Multi-bit Encoding**: Support for both 2-bit and 4-bit nucleotide encodings. -//! -//! Legacy VBQ files are automatically migrated to the new format when accessed. -//! -//! # Example: Memory-mapped Access +//! # Example: Parallel Processing //! //! ``` //! use binseq::Result; @@ -47,32 +20,29 @@ //! //! #[derive(Clone, Default)] //! pub struct Processor { -//! // Define fields here +//! // per-thread state //! } //! //! impl ParallelProcessor for Processor { //! fn process_record(&mut self, record: B) -> Result<()> { -//! // Implement per-record logic here +//! // per-record logic //! Ok(()) //! } //! //! fn on_batch_complete(&mut self) -> Result<()> { -//! // Implement per-batch logic here +//! // per-batch logic (e.g. flush to a shared writer) //! Ok(()) //! } //! } //! //! fn main() -> Result<()> { -//! // provide an input path (*.bq or *.vbq) -//! let path = "./data/subset.bq"; +//! // any BINSEQ variant (*.cbq, *.bq, *.vbq); format is sniffed from content +//! let path = "./data/subset.cbq"; //! -//! // open a reader //! let reader = BinseqReader::new(path)?; -//! -//! // initialize a processor //! let processor = Processor::default(); //! -//! // process the records in parallel with 8 threads +//! // process records in parallel with 8 threads //! reader.process_parallel(processor, 8)?; //! Ok(()) //! } @@ -80,9 +50,15 @@ #![allow(clippy::module_inception)] -/// BQ - fixed length records, no quality scores +/// CBQ - columnar variable-length records, optional quality scores and headers +pub mod cbq; + +/// BQ - fixed-length records, no quality scores pub mod bq; +/// VBQ - variable-length records, optional quality scores, compressed blocks +pub mod vbq; + /// Shared nucleotide encoder for BQ/VBQ writers mod encoder; @@ -98,13 +74,7 @@ mod policy; /// Record types and traits shared between BINSEQ variants mod record; -/// VBQ - Variable length records, optional quality scores, compressed blocks -pub mod vbq; - -/// CBQ - Columnar variable length records, optional quality scores and headers -pub mod cbq; - -/// Prelude - Commonly used types and traits +/// Commonly used types and traits pub mod prelude; /// Write operations generic over the BINSEQ variant diff --git a/src/parallel.rs b/src/parallel.rs index eaf3a88..880d983 100644 --- a/src/parallel.rs +++ b/src/parallel.rs @@ -146,22 +146,7 @@ pub trait ParallelReader { num_threads: usize, ) -> Result<()>; - /// Process records in parallel within a specified range - /// - /// This method allows parallel processing of a subset of records within the file, - /// defined by a start and end index. The range is distributed across the specified - /// number of threads. - /// - /// # Arguments - /// - /// * `processor` - The processor to use for each record - /// * `num_threads` - The number of threads to spawn - /// * `range` - The range of record indices to process - /// - /// # Returns - /// - /// * `Ok(())` - If all records were processed successfully - /// * `Err(Error)` - If an error occurred during processing + /// Process a range of record indices in parallel, distributed across `num_threads` fn process_parallel_range( self, processor: P, @@ -169,21 +154,7 @@ pub trait ParallelReader { range: Range, ) -> Result<()>; - /// Validate the specified range for the file. - /// - /// This method checks if the provided range is valid for the file, ensuring that - /// the start index is less than the end index and both indices are within the - /// bounds of the file. - /// - /// # Arguments - /// - /// * `total_records` - The total number of records in the file - /// * `range` - The range of record indices to validate - /// - /// # Returns - /// - /// * `Ok(())` - If the range is valid - /// * `Err(Error)` - If the range is invalid + /// Validate that a record range is well-formed and within the file's bounds fn validate_range(&self, total_records: usize, range: &Range) -> Result<()> { if range.start >= total_records { Err(ReadError::OutOfRange { diff --git a/src/policy.rs b/src/policy.rs index 3e9a515..3028b99 100644 --- a/src/policy.rs +++ b/src/policy.rs @@ -1,27 +1,16 @@ -//! Nucleotide sequence validation and correction policies -//! -//! This module provides policies for handling invalid nucleotides in sequences -//! during encoding operations. Different policies allow for ignoring, rejecting, -//! or correcting sequences with invalid nucleotides. +//! Policies for handling invalid nucleotides during BQ/VBQ encoding. use rand::Rng; use crate::error::{Result, WriteError}; -/// A global seed for the random number generator used in randomized policies -/// -/// This seed ensures reproducible behavior when using the `RandomDraw` policy -/// across different runs of the program. +/// Seed for the RNG used by `RandomDraw`, for reproducibility across runs pub const RNG_SEED: u64 = 42; -/// Policy for handling invalid nucleotide sequences during encoding -/// -/// When encoding sequences into binary format, non-standard nucleotides (anything -/// other than A, C, G, or T) may be encountered. This enum defines different -/// strategies for handling such invalid nucleotides. +/// Policy for handling invalid nucleotides (anything other than A, C, G, T) during encoding /// -/// The default policy is `IgnoreSequence`, which skips sequences containing -/// invalid nucleotides. +/// Defaults to `IgnoreSequence`, which skips offending sequences. +/// Only applies to BQ/VBQ; CBQ stores `N`s natively. #[derive(Debug, Clone, Copy, Default)] pub enum Policy { /// Skip sequences containing invalid nucleotides (default policy) @@ -47,16 +36,7 @@ pub enum Policy { SetToT, } impl Policy { - /// Helper method to replace invalid nucleotides with a specific nucleotide - /// - /// This internal method processes a sequence and replaces any non-standard - /// nucleotides (anything other than A, C, G, or T) with the specified value. - /// - /// # Arguments - /// - /// * `sequence` - The input sequence to process - /// * `val` - The replacement nucleotide (should be one of A, C, G, or T) - /// * `ibuf` - The output buffer to store the processed sequence + /// Replace invalid nucleotides with a specific nucleotide fn fill_with_known(sequence: &[u8], val: u8, ibuf: &mut Vec) { for &n in sequence { ibuf.push(match n { @@ -66,20 +46,7 @@ impl Policy { } } - /// Helper method to replace invalid nucleotides with random valid nucleotides - /// - /// This internal method processes a sequence and replaces any non-standard - /// nucleotides with randomly chosen valid nucleotides (A, C, G, or T). - /// - /// # Arguments - /// - /// * `sequence` - The input sequence to process - /// * `rng` - The random number generator to use for selecting replacement nucleotides - /// * `ibuf` - The output buffer to store the processed sequence - /// - /// # Type Parameters - /// - /// * `R` - A type that implements the `Rng` trait from the `rand` crate + /// Replace invalid nucleotides with randomly chosen valid nucleotides fn fill_with_random(sequence: &[u8], rng: &mut R, ibuf: &mut Vec) { for &n in sequence { ibuf.push(match n { @@ -89,40 +56,20 @@ impl Policy { } } - /// Process a sequence according to the selected policy for handling invalid nucleotides - /// - /// This method applies the policy to the given sequence, handling any invalid nucleotides - /// according to the policy's rules. It first clears the input buffer to ensure that it is empty, - /// then processes the sequence accordingly. - /// - /// # Arguments - /// - /// * `sequence` - The nucleotide sequence to be processed - /// * `ibuf` - The buffer to store the processed sequence (will be cleared first) - /// * `rng` - The random number generator (used only with the `RandomDraw` policy) + /// Apply the policy to a sequence, writing the corrected result into `ibuf` (cleared first) /// - /// # Returns - /// - /// * `Ok(true)` - If the sequence was processed and should be encoded - /// * `Ok(false)` - If the sequence should be skipped (for `IgnoreSequence` policy) - /// * `Err(Error)` - If an error occurred (for `BreakOnInvalid` policy when invalid nucleotides are found) - /// - /// # Type Parameters - /// - /// * `R` - A type that implements the `Rng` trait from the `rand` crate + /// Returns `Ok(true)` if the sequence should be encoded, `Ok(false)` if it should + /// be skipped (`IgnoreSequence`), or an error (`BreakOnInvalid`). /// /// # Examples /// /// ``` /// # use binseq::{Policy, Result}; - /// # use rand::thread_rng; /// # fn main() -> Result<()> { - /// let policy = Policy::SetToA; - /// let sequence = b"ACGTNX"; /// let mut output = Vec::new(); - /// let mut rng = thread_rng(); + /// let mut rng = rand::rng(); /// - /// let should_process = policy.handle(sequence, &mut output, &mut rng)?; + /// let should_process = Policy::SetToA.handle(b"ACGTNX", &mut output, &mut rng)?; /// /// assert!(should_process); /// assert_eq!(output, b"ACGTAA"); @@ -130,10 +77,7 @@ impl Policy { /// # } /// ``` pub fn handle(&self, sequence: &[u8], ibuf: &mut Vec, rng: &mut R) -> Result { - // First clears the input buffer to ensure that it is empty. ibuf.clear(); - - // Returns a boolean indicating whether the sequence should be processed further. match self { Self::IgnoreSequence => Ok(false), Self::BreakOnInvalid => { diff --git a/src/record/binseq_record.rs b/src/record/binseq_record.rs index 9ed9ef6..da24002 100644 --- a/src/record/binseq_record.rs +++ b/src/record/binseq_record.rs @@ -4,12 +4,8 @@ use crate::Result; /// Record trait shared between BINSEQ variants. /// -/// Exposes public methods for accessing internal data. -/// Interfaces with the [`bitnuc`] crate for decoding sequences. -/// -/// Implemented by [`bq::RefRecord`](crate::bq::RefRecord) and [`vbq::RefRecord`](crate::vbq::RefRecord). -/// -/// Used to interact with [`ParallelProcessor`](crate::ParallelProcessor) for easy parallel processing. +/// Implemented by the `RefRecord` type of each variant and consumed by +/// [`ParallelProcessor`](crate::ParallelProcessor) for parallel processing. pub trait BinseqRecord { /// Returns the bitsize of the record (number of bits per nucleotide) fn bitsize(&self) -> BitSize; @@ -68,46 +64,44 @@ pub trait BinseqRecord { Ok(()) } - /// Returns a reference to the primary decoded sequence of this record. + /// Returns the decoded primary sequence. + /// + /// # Panics /// - /// This is not available on all types that implement the `Record` trait. - /// It should be available on types that implement it in this library however. + /// The default implementation panics; all record types in this library override it. fn sseq(&self) -> &[u8] { unimplemented!("This record does not implement direct sequence access"); } - /// Returns a reference to the extended decoded sequence of this record. + /// Returns the decoded extended sequence. /// - /// This may not be available on all types that implement the `Record` trait. - /// It should be available on types that implement it in this library however. + /// # Panics + /// + /// The default implementation panics; all record types in this library override it. fn xseq(&self) -> &[u8] { unimplemented!("This record does not implement direct sequence access"); } - /// Decodes the primary sequence of this record into a newly allocated buffer. - /// - /// Not advised to use this function as it allocates a new buffer every time. + /// Decodes the primary sequence into a newly allocated buffer (prefer [`decode_s`](Self::decode_s)). fn decode_s_alloc(&self) -> Result> { let mut buf = Vec::with_capacity(self.slen() as usize); self.decode_s(&mut buf)?; Ok(buf) } - /// Decodes the extended sequence of this record into a newly allocated buffer. - /// - /// Not advised to use this function as it allocates a new buffer every time. + /// Decodes the extended sequence into a newly allocated buffer (prefer [`decode_x`](Self::decode_x)). fn decode_x_alloc(&self) -> Result> { let mut buf = Vec::with_capacity(self.xlen() as usize); self.decode_x(&mut buf)?; Ok(buf) } - /// A convenience function to check if the record is paired. + /// Returns true if the record is paired. fn is_paired(&self) -> bool { self.xlen() > 0 } - /// A convenience function to check if record has associated quality scores + /// Returns true if the record has associated quality scores. fn has_quality(&self) -> bool { !self.squal().is_empty() } diff --git a/src/record/sequencing_record.rs b/src/record/sequencing_record.rs index 7437d78..3a84a4e 100644 --- a/src/record/sequencing_record.rs +++ b/src/record/sequencing_record.rs @@ -1,9 +1,6 @@ use crate::{BitSize, Result, error::WriteError}; -/// A zero-copy record used to write sequences to binary sequence files. -/// -/// This struct provides a unified API for writing records to all binseq formats -/// (BQ, VBQ, and CBQ). It uses borrowed references for zero-copy efficiency. +/// A zero-copy (borrowed) record used to write sequences to any BINSEQ format. /// /// # Example /// @@ -52,11 +49,9 @@ impl<'a> SequencingRecord<'a> { } } - /// Returns the configured size of this record for CBQ format. + /// Returns the encoded size of this record for CBQ, given the writer configuration. /// - /// CBQ uses columnar storage so there are no per-record length prefixes. - /// This calculates the size based on writer configuration, ignoring any - /// extra data in the record that the writer won't use. + /// Extra data the writer won't use is not counted. #[inline] #[must_use] pub fn configured_size_cbq( @@ -105,22 +100,11 @@ impl<'a> SequencingRecord<'a> { size } - /// Returns the configured size of this record for VBQ format. - /// - /// VBQ uses a row-based format with length prefixes for each field. - /// This calculates the size based on writer configuration, ignoring any - /// extra data in the record that the writer won't use. + /// Returns the encoded size of this record for VBQ (row-based, length-prefixed), + /// given the writer configuration. /// - /// The VBQ record layout is: - /// - Flag (8 bytes, if `has_flags`) - /// - `s_len` (8 bytes) - /// - `x_len` (8 bytes) - /// - `s_seq` (encoded, rounded up to 8-byte words) - /// - `s_qual` (raw bytes, if `has_qualities`) - /// - `s_header_len` + `s_header` (8 + len bytes, if `has_headers` and `s_header` present) - /// - `x_seq` (encoded, rounded up to 8-byte words, if paired) - /// - `x_qual` (raw bytes, if `has_qualities` and paired) - /// - `x_header_len` + `x_header` (8 + len bytes, if `has_headers` and `x_header` present) + /// Extra data the writer won't use is not counted. + /// See [`vbq`](crate::vbq) for the record layout. #[inline] #[must_use] pub fn configured_size_vbq( @@ -206,7 +190,7 @@ impl<'a> SequencingRecord<'a> { } } -/// A convenience builder struct for creating a [`SequencingRecord`] +/// Builder for a [`SequencingRecord`] /// /// # Example /// @@ -332,11 +316,7 @@ impl<'a> SequencingRecordBuilder<'a> { self } - /// Builds the `SequencingRecord` - /// - /// # Errors - /// - /// Returns an error if the primary sequence (`s_seq`) is not set. + /// Builds the `SequencingRecord`; errors if `s_seq` is not set pub fn build(self) -> Result> { let Some(s_seq) = self.s_seq else { return Err(WriteError::MissingSequence.into()); diff --git a/src/utils/fastx.rs b/src/utils/fastx.rs index a0459c2..ad7a4a7 100644 --- a/src/utils/fastx.rs +++ b/src/utils/fastx.rs @@ -1,7 +1,4 @@ -//! FASTX encoding utilities for converting FASTX files to BINSEQ formats -//! -//! This module provides utilities for encoding FASTX (FASTA/FASTQ) files into -//! BINSEQ formats using parallel processing via the `paraseq` crate. +//! Parallel encoding of FASTX (FASTA/FASTQ) files into BINSEQ formats via `paraseq`. use std::{ io::{Read, Write}, @@ -36,8 +33,8 @@ enum FastxInput { /// Builder for encoding FASTX files to BINSEQ format /// -/// This builder is created by calling [`BinseqWriterBuilder::encode_fastx`] and -/// provides a fluent interface for configuring the input source and threading options. +/// Created by [`BinseqWriterBuilder::encode_fastx`]; configures the input source +/// and threading before running the encoding. /// /// # Example /// @@ -45,12 +42,12 @@ enum FastxInput { /// use binseq::write::{BinseqWriterBuilder, Format}; /// use std::fs::File; /// -/// // Encode from stdin to VBQ -/// let writer = BinseqWriterBuilder::new(Format::Vbq) +/// // Encode paired-end FASTQ to CBQ +/// BinseqWriterBuilder::new(Format::Cbq) /// .quality(true) /// .headers(true) -/// .encode_fastx(Box::new(File::create("output.vbq")?)) -/// .input_stdin() +/// .encode_fastx(Box::new(File::create("output.cbq")?)) +/// .input_paired("R1.fastq", "R2.fastq") /// .threads(8) /// .run()?; /// # Ok::<(), binseq::Error>(()) @@ -74,18 +71,6 @@ impl FastxEncoderBuilder { } /// Read from a single FASTX file - /// - /// # Example - /// - /// ```rust,no_run - /// # use binseq::write::{BinseqWriterBuilder, Format}; - /// # use std::fs::File; - /// BinseqWriterBuilder::new(Format::Vbq) - /// .encode_fastx(Box::new(File::create("output.vbq")?)) - /// .input("input.fastq") - /// .run()?; - /// # Ok::<(), binseq::Error>(()) - /// ``` #[must_use] pub fn input>(mut self, path: P) -> Self { self.input = Some(FastxInput::Single(path.as_ref().to_path_buf())); @@ -93,39 +78,13 @@ impl FastxEncoderBuilder { } /// Read from stdin - /// - /// # Example - /// - /// ```rust,no_run - /// # use binseq::write::{BinseqWriterBuilder, Format}; - /// # use std::fs::File; - /// BinseqWriterBuilder::new(Format::Vbq) - /// .encode_fastx(Box::new(File::create("output.vbq")?)) - /// .input_stdin() - /// .run()?; - /// # Ok::<(), binseq::Error>(()) - /// ``` #[must_use] pub fn input_stdin(mut self) -> Self { self.input = Some(FastxInput::Stdin); self } - /// Read from paired FASTX files (R1, R2) - /// - /// This automatically sets the writer to paired mode. - /// - /// # Example - /// - /// ```rust,no_run - /// # use binseq::write::{BinseqWriterBuilder, Format}; - /// # use std::fs::File; - /// BinseqWriterBuilder::new(Format::Vbq) - /// .encode_fastx(Box::new(File::create("output.vbq")?)) - /// .input_paired("R1.fastq", "R2.fastq") - /// .run()?; - /// # Ok::<(), binseq::Error>(()) - /// ``` + /// Read from paired FASTX files (R1, R2); sets the writer to paired mode #[must_use] pub fn input_paired>(mut self, r1: P, r2: P) -> Self { self.input = Some(FastxInput::Paired( @@ -137,52 +96,14 @@ impl FastxEncoderBuilder { self } - /// Set the number of threads for parallel processing - /// - /// If not set or set to 0, uses all available CPU cores. - /// - /// # Example - /// - /// ```rust,no_run - /// # use binseq::write::{BinseqWriterBuilder, Format}; - /// # use std::fs::File; - /// BinseqWriterBuilder::new(Format::Vbq) - /// .encode_fastx(Box::new(File::create("output.vbq")?)) - /// .input("input.fastq") - /// .threads(8) - /// .run()?; - /// # Ok::<(), binseq::Error>(()) - /// ``` + /// Set the number of threads for parallel processing (0: all available cores) #[must_use] pub fn threads(mut self, n: usize) -> Self { self.threads = n; self } - /// Execute the FASTX encoding - /// - /// This consumes the builder and returns a `BinseqWriter` that has been - /// populated with all records from the input FASTX file(s). - /// - /// # Errors - /// - /// Returns an error if: - /// - The input files cannot be read - /// - The FASTX format is invalid - /// - The writer configuration is incompatible with the input - /// - For BQ format with stdin input (cannot detect sequence length) - /// - /// # Example - /// - /// ```rust,no_run - /// # use binseq::write::{BinseqWriterBuilder, Format}; - /// # use std::fs::File; - /// let writer = BinseqWriterBuilder::new(Format::Vbq) - /// .encode_fastx(Box::new(File::create("output.vbq")?)) - /// .input("input.fastq") - /// .run()?; - /// # Ok::<(), binseq::Error>(()) - /// ``` + /// Execute the FASTX encoding, consuming the builder pub fn run(mut self) -> Result<()> { let (r1, r2) = match self.input { Some(FastxInput::Single(path)) => { @@ -265,10 +186,7 @@ fn detect_seq_len( Ok((slen, xlen)) } -/// Parallel encoder for FASTX records to BINSEQ format -/// -/// This struct implements the `ParallelProcessor` and `PairedParallelProcessor` -/// traits from `paraseq` to enable efficient parallel encoding of FASTX files. +/// Parallel encoder implementing `paraseq`'s processor traits #[derive(Clone)] struct Encoder { /// Global writer (shared across threads) diff --git a/src/vbq/header.rs b/src/vbq/header.rs index bec355a..3777299 100644 --- a/src/vbq/header.rs +++ b/src/vbq/header.rs @@ -1,16 +1,7 @@ -//! # File and Block Header Definitions +//! VBQ file and block header definitions. //! -//! This module defines the header structures used in the VBQ file format. -//! -//! The VBQ format consists of two primary header types: -//! -//! 1. `FileHeader` - The file header that appears at the beginning of a VBQ file, -//! containing information about the overall file format and configuration. -//! -//! 2. `BlockHeader` - Headers that appear before each block of records, containing -//! information specific to that block like its size and number of records. -//! -//! Both headers are fixed-size and include magic numbers to validate file integrity. +//! `FileHeader` opens the file; a `BlockHeader` precedes each record block. +//! Both are fixed-size (32 bytes) and carry magic numbers for validation. use std::io::{Read, Write}; @@ -21,52 +12,30 @@ use crate::{ utils::{read_u32_le, read_u64_le}, }; -/// Magic number for file identification: "VSEQ" in ASCII (0x51455356) -/// -/// This constant is used in the file header to identify VBQ formatted files. -#[allow(clippy::unreadable_literal)] -const MAGIC: u32 = 0x51455356; - -/// The magic bytes as they appear at the start of a VBQ file on disk. -/// -/// Used to identify VBQ files by content rather than by file extension. -pub const FILE_MAGIC: [u8; 4] = MAGIC.to_le_bytes(); +/// Magic bytes at the start of a VBQ file on disk +pub const FILE_MAGIC: [u8; 4] = *b"VSEQ"; +const MAGIC: u32 = u32::from_le_bytes(FILE_MAGIC); -/// Magic number for block identification: "BLOCKSEQ" in ASCII (0x5145534B434F4C42) -/// -/// This constant is used in block headers to validate block integrity. +/// Block magic number: "BLOCKSEQ" in ASCII (0x5145534B434F4C42) #[allow(clippy::unreadable_literal)] const BLOCK_MAGIC: u64 = 0x5145534B434F4C42; /// Current format version number -/// -/// This should be incremented when making backwards-incompatible changes to the format. const FORMAT: u8 = 1; -/// Size of the file header in bytes (32 bytes) -/// -/// The file header has a fixed size to simplify parsing. +/// Size of the file header in bytes pub const SIZE_HEADER: usize = 32; -/// Size of the block header in bytes (32 bytes) -/// -/// Each block header has a fixed size to simplify block navigation. +/// Size of each block header in bytes pub const SIZE_BLOCK_HEADER: usize = 32; -/// Default block size in bytes: 128KB -/// -/// This defines the default virtual size of each record block. -/// A larger block size can improve compression ratio but reduces random access granularity. +/// Default virtual block size in bytes (128KB) pub const BLOCK_SIZE: u64 = 128 * 1024; -/// Reserved bytes for future use in the file header -/// -/// These bytes are set to a placeholder value (42) and reserved for future extensions. +/// Reserved bytes in the file header (placeholder value 42) pub const RESERVED_BYTES: [u8; 13] = [42; 13]; -/// Reserved bytes for future use in block headers (12 bytes) -/// -/// These bytes are set to a placeholder value (42) and reserved for future extensions. +/// Reserved bytes in block headers (placeholder value 42) pub const RESERVED_BYTES_BLOCK: [u8; 12] = [42; 12]; #[derive(Default, Debug, Clone, Copy)] @@ -134,88 +103,41 @@ impl FileHeaderBuilder { } } -/// File header for VBQ files -/// -/// This structure represents the 32-byte header that appears at the beginning of every -/// VBQ file. It contains configuration information about the file format, including -/// whether quality scores are included, whether blocks are compressed, and whether -/// records contain paired sequences. -/// -/// # Fields -/// -/// * `magic` - Magic number to validate file format ("VSEQ", 4 bytes) -/// * `format` - Version number of the file format (1 byte) -/// * `block` - Size of each block in bytes (8 bytes) -/// * `qual` - Whether quality scores are included (1 byte boolean) -/// * `compressed` - Whether blocks are ZSTD compressed (1 byte boolean) -/// * `paired` - Whether records contain paired sequences (1 byte boolean) -/// * `reserved` - Reserved bytes for future extensions (16 bytes) +/// 32-byte header at the start of every VBQ file. #[derive(Clone, Copy, Debug, PartialEq)] pub struct FileHeader { - /// Magic number to identify the file format ("VSEQ") - /// - /// Always set to 0x51455356 (4 bytes) + /// Magic number "VSEQ" (4 bytes) pub magic: u32, - /// Version of the file format - /// - /// Currently set to 1 (1 byte) + /// Format version (1 byte) pub format: u8, - /// Block size in bytes - /// - /// This is the virtual (uncompressed) size of each record block (8 bytes) + /// Virtual (uncompressed) block size in bytes (8 bytes) pub block: u64, - /// Whether quality scores are included with sequences - /// - /// If true, quality scores are stored for each nucleotide (1 byte) + /// Whether quality scores are included (1 byte) pub qual: bool, - /// Whether internal blocks are compressed with ZSTD - /// - /// If true, blocks are compressed individually (1 byte) + /// Whether blocks are ZSTD compressed (1 byte) pub compressed: bool, - /// Whether records contain paired sequences - /// - /// If true, each record has both primary and extended sequences (1 byte) + /// Whether records contain paired sequences (1 byte) pub paired: bool, - /// The bitsize of the sequence data (1 byte) - /// - /// Specifies the number of bits per nucleotide: - /// - 2-bit: Standard encoding (A=00, C=01, G=10, T=11) - /// - 4-bit: Extended encoding supporting ambiguous nucleotides + /// Bits per nucleotide: 2-bit standard or 4-bit ambiguous (1 byte) pub bits: BitSize, - /// Whether sequence headers are included with sequences (1 byte) - /// - /// When true, each record includes length-prefixed UTF-8 header strings - /// for both primary and extended (paired) sequences + /// Whether length-prefixed sequence headers are included (1 byte) pub headers: bool, - /// Whether flags are included with sequences (1 byte) - /// - /// When true, each record includes length-prefixed UTF-8 flag strings - /// for both primary and extended (paired) sequences + /// Whether per-record flags are included (1 byte) pub flags: bool, - /// Reserved bytes for future format extensions - /// - /// Currently filled with placeholder values (13 bytes) + /// Reserved bytes for future extensions (13 bytes) pub reserved: [u8; 13], } impl Default for FileHeader { - /// Creates a default header with default block size and all features disabled - /// - /// The default header: - /// - Uses the default block size (128KB) - /// - Does not include quality scores - /// - Does not use compression - /// - Does not support paired sequences - /// - Does not include sequence headers - /// - Uses 2-bit nucleotide encoding + /// Default block size (128KB), 2-bit encoding, all features disabled. fn default() -> Self { Self { magic: MAGIC, @@ -232,24 +154,7 @@ impl Default for FileHeader { } } impl FileHeader { - /// Creates a header from a 32-byte buffer - /// - /// This function parses a raw byte buffer into a `FileHeader` structure, - /// validating the magic number and format version. - /// - /// # Parameters - /// - /// * `buffer` - A 32-byte array containing the header data - /// - /// # Returns - /// - /// * `Result` - A valid header if parsing was successful - /// - /// # Errors - /// - /// * `HeaderError::InvalidMagicNumber` - If the magic number doesn't match "VSEQ" - /// * `HeaderError::InvalidFormatVersion` - If the format version is unsupported - /// * `HeaderError::InvalidReservedBytes` - If the reserved bytes section is invalid + /// Parses a header from a 32-byte buffer, validating magic and version. pub fn from_bytes(buffer: &[u8; SIZE_HEADER]) -> Result { let magic = read_u32_le(&buffer[0..4]); if magic != MAGIC { @@ -290,22 +195,7 @@ impl FileHeader { }) } - /// Writes the header to a writer - /// - /// This function serializes the header structure into a 32-byte buffer and writes - /// it to the provided writer. - /// - /// # Parameters - /// - /// * `writer` - Any type that implements the `Write` trait - /// - /// # Returns - /// - /// * `Result<()>` - Success if the header was written - /// - /// # Errors - /// - /// * IO errors if writing to the writer fails + /// Serializes the header as 32 bytes to a writer. pub fn write_bytes(&self, writer: &mut W) -> Result<()> { let mut buffer = [0u8; SIZE_HEADER]; buffer[0..4].copy_from_slice(&self.magic.to_le_bytes()); @@ -322,23 +212,7 @@ impl FileHeader { Ok(()) } - /// Reads a header from a reader - /// - /// This function reads 32 bytes from the provided reader and parses them into - /// a `FileHeader` structure. - /// - /// # Parameters - /// - /// * `reader` - Any type that implements the `Read` trait - /// - /// # Returns - /// - /// * `Result` - A valid header if reading and parsing was successful - /// - /// # Errors - /// - /// * IO errors if reading from the reader fails - /// * Header validation errors from `from_bytes()` + /// Reads and parses a 32-byte header from a reader. pub fn from_reader(reader: &mut R) -> Result { let mut buffer = [0u8; SIZE_HEADER]; reader.read_exact(&mut buffer)?; @@ -351,56 +225,23 @@ impl FileHeader { } } -/// Block header for VBQ block data -/// -/// Each block in a VBQ file is preceded by a 32-byte block header that contains -/// information about the block including its size and the number of records it contains. -/// -/// # Fields -/// -/// * `magic` - Magic number to validate block integrity ("BLOCKSEQ", 8 bytes) -/// * `size` - Actual size of the block in bytes (8 bytes) -/// * `records` - Number of records in the block (4 bytes) -/// * `reserved` - Reserved bytes for future extensions (12 bytes) +/// 32-byte header preceding each block in a VBQ file. #[derive(Clone, Copy, Debug)] pub struct BlockHeader { - /// Magic number to identify the block ("BLOCKSEQ") - /// - /// Always set to 0x5145534B434F4C42 (8 bytes) + /// Magic number "BLOCKSEQ" (8 bytes) pub magic: u64, - /// Actual size of the block in bytes - /// - /// This can differ from the virtual block size in the file header - /// when compression is enabled (8 bytes) + /// Actual on-disk block size in bytes; differs from the virtual size when compressed (8 bytes) pub size: u64, - /// Number of records stored in this block - /// - /// Used to iterate through records efficiently (4 bytes) + /// Number of records in this block (4 bytes) pub records: u32, - /// Reserved bytes for future extensions - /// - /// Currently filled with placeholder values (12 bytes) + /// Reserved bytes for future extensions (12 bytes) pub reserved: [u8; 12], } impl BlockHeader { - /// Creates a new block header - /// - /// # Parameters - /// - /// * `size` - The actual size of the block in bytes (can be compressed size) - /// * `records` - The number of records contained in the block - /// - /// # Example - /// - /// ```rust - /// use binseq::vbq::BlockHeader; - /// - /// // Create a block header for a block with 1024 bytes and 100 records - /// let header = BlockHeader::new(1024, 100); - /// ``` + /// Creates a new block header with the given on-disk size and record count. #[must_use] pub fn new(size: u64, records: u32) -> Self { Self { @@ -426,22 +267,7 @@ impl BlockHeader { self.size == 0 && self.records == 0 } - /// Writes the block header to a writer - /// - /// This function serializes the block header structure into a 32-byte buffer and writes - /// it to the provided writer. - /// - /// # Parameters - /// - /// * `writer` - Any type that implements the `Write` trait - /// - /// # Returns - /// - /// * `Result<()>` - Success if the header was written - /// - /// # Errors - /// - /// * IO errors if writing to the writer fails + /// Serializes the block header as 32 bytes to a writer. pub fn write_bytes(&self, writer: &mut W) -> Result<()> { let mut buffer = [0u8; SIZE_BLOCK_HEADER]; buffer[0..8].copy_from_slice(&self.magic.to_le_bytes()); @@ -452,22 +278,7 @@ impl BlockHeader { Ok(()) } - /// Creates a block header from a 32-byte buffer - /// - /// This function parses a raw byte buffer into a `BlockHeader` structure, - /// validating the magic number. - /// - /// # Parameters - /// - /// * `buffer` - A 32-byte array containing the block header data - /// - /// # Returns - /// - /// * `Result` - A valid block header if parsing was successful - /// - /// # Errors - /// - /// * `ReadError::InvalidBlockMagicNumber` - If the magic number doesn't match "BLOCKSEQ" + /// Parses a block header from a 32-byte buffer, validating the magic number. pub fn from_bytes(buffer: &[u8; SIZE_BLOCK_HEADER]) -> Result { let magic = read_u64_le(&buffer[0..8]); if magic != BLOCK_MAGIC { diff --git a/src/vbq/index.rs b/src/vbq/index.rs index c530dca..4b048af 100644 --- a/src/vbq/index.rs +++ b/src/vbq/index.rs @@ -1,36 +1,13 @@ -//! # VBQ Index Format +//! Embedded index for VBQ files (v0.7.0+). //! -//! This module implements the embedded index format for VBQ files. -//! -//! ## Format Changes (v0.7.0+) -//! -//! **BREAKING CHANGE**: The VBQ index is now embedded at the end of VBQ files, -//! improving portability and eliminating the need to manage auxiliary files. -//! -//! ## Embedded Index Structure -//! -//! The index is located at the end of the VBQ file with this layout: +//! The index sits at the end of the file: //! //! ```text //! [VBQ Data Blocks][Compressed Index][Index Size (u64)][INDEX_END_MAGIC (u64)] //! ``` //! -//! Where: -//! - **Compressed Index**: ZSTD-compressed index data (`IndexHeader` + `BlockRanges`) -//! - **Index Size**: 8 bytes indicating size of compressed index data -//! - **`INDEX_END_MAGIC`**: 8 bytes (`0x444E455845444E49` = "INDEXEND") -//! -//! ## Index Contents -//! -//! The compressed index contains: -//! 1. **`IndexHeader`** (32 bytes): Metadata about the indexed file -//! 2. **`BlockRange` entries** (32 bytes each): One per data block -//! -//! ## Key Changes from v0.6.x -//! -//! - Index is now embedded in VBQ files -//! - Cumulative record counts changed from `u32` to `u64` -//! - Support for files with more than 4 billion records +//! The compressed section is ZSTD-compressed and holds an `IndexHeader` +//! (32 bytes) followed by one 32-byte `BlockRange` per data block. use std::io::{Cursor, Read, Write}; @@ -52,97 +29,28 @@ pub const INDEX_END_MAGIC: u64 = 0x444E455845444E49; /// Index Block Reservation pub const INDEX_RESERVATION: [u8; 4] = [42; 4]; -/// Descriptor of the dimensions of a block in a VBQ file -/// -/// A `BlockRange` contains metadata about a single block within a VBQ file, -/// including its position, size, and record count. This information enables -/// efficient random access to blocks without scanning the entire file. -/// -/// Block ranges are stored in a `BlockIndex` to form a complete index of a VBQ file. -/// Each range is serialized to a fixed-size 32-byte structure when stored in the embedded index. -/// -/// ## Format Changes (v0.7.0+) +/// Position, size, and record counts of a single block; 32 bytes serialized. /// -/// - `cumulative_records` field changed from `u32` to `u64` -/// - Supports files with more than 4 billion records -/// - Reserved bytes reduced from 8 to 4 bytes -/// -/// # Examples -/// -/// ```rust -/// use binseq::vbq::BlockRange; -/// -/// // Create a new block range -/// let range = BlockRange::new( -/// 1024, // Starting offset in the file (bytes) -/// 8192, // Length of the block (bytes) -/// 1000, // Number of records in this block -/// 5000 // Cumulative number of records up to this block (now u64) -/// ); -/// -/// // Use the range information -/// println!("Block starts at byte {}", range.start_offset); -/// println!("Block contains {} records", range.block_records); -/// ``` +/// Stored in a `BlockIndex` to enable random access without scanning the file. #[derive(Debug, Clone, Copy)] pub struct BlockRange { - /// File offset where the block starts (in bytes, including headers) - /// - /// This is the absolute byte position in the file where this block begins, - /// including the file header and block header. - /// - /// (8 bytes in serialized form) + /// Absolute file offset where the block (header included) starts (8 bytes) pub start_offset: u64, - /// Length of the block data in bytes - /// - /// This is the size of the block data, not including the block header. - /// For compressed blocks, this is the compressed size. - /// - /// (8 bytes in serialized form) + /// Block data length in bytes, excluding the block header; compressed size if compressed (8 bytes) pub len: u64, - /// Number of records contained in this block - /// - /// (4 bytes in serialized form) + /// Number of records in this block (4 bytes) pub block_records: u32, - /// Cumulative number of records up to this block - /// - /// This allows efficient determination of which block contains a specific record - /// by index without scanning through all previous blocks. - /// - /// **BREAKING CHANGE (v0.7.0+)**: Changed from u32 to u64 to support files - /// with more than 4 billion records. - /// - /// (8 bytes in serialized form) + /// Cumulative number of records before this block (8 bytes) pub cumulative_records: u64, - /// Reserved bytes for future extensions + /// Reserved bytes for future extensions (4 bytes) pub reservation: [u8; 4], } impl BlockRange { - /// Creates a new `BlockRange` with the specified parameters - /// - /// # Parameters - /// - /// * `start_offset` - The byte offset in the file where this block starts - /// * `len` - The length of the block data in bytes - /// * `block_records` - The number of records contained in this block - /// * `cumulative_records` - The total number of records up to and including this block - /// - /// # Returns - /// - /// A new `BlockRange` instance with the specified parameters - /// - /// # Examples - /// - /// ```rust - /// use binseq::vbq::BlockRange; - /// - /// // Create a new block range for a block starting at byte 1024 - /// let range = BlockRange::new(1024, 8192, 1000, 5000); - /// ``` + /// Creates a new `BlockRange`. #[must_use] pub fn new(start_offset: u64, len: u64, block_records: u32, cumulative_records: u64) -> Self { Self { @@ -154,24 +62,7 @@ impl BlockRange { } } - /// Serializes the block range to a binary format and writes it to the provided writer - /// - /// This method serializes the `BlockRange` to a fixed-size 32-byte structure and - /// writes it to the provided writer. The serialized format is: - /// - Bytes 0-7: `start_offset` (u64, little endian) - /// - Bytes 8-15: len (u64, little endian) - /// - Bytes 16-19: `block_records` (u32, little endian) - /// - Bytes 20-23: `cumulative_records` (u32, little endian) - /// - Bytes 24-31: reservation (8 bytes) - /// - /// # Parameters - /// - /// * `writer` - The destination to write the serialized block range to - /// - /// # Returns - /// - /// * `Ok(())` - If the block range was successfully written - /// * `Err(_)` - If an error occurred during writing + /// Serializes the block range as 32 bytes (little-endian fields) to a writer. pub fn write_bytes(&self, writer: &mut W) -> Result<()> { let mut buf = [0; SIZE_BLOCK_RANGE]; buf[0..8].copy_from_slice(&self.start_offset.to_le_bytes()); @@ -183,16 +74,7 @@ impl BlockRange { Ok(()) } - /// Deserializes a `BlockRange` from a slice of bytes - /// - /// # Format - /// - /// The buffer is expected to contain: - /// - Bytes 0-7: `start_offset` (u64, little endian) - /// - Bytes 8-15: len (u64, little endian) - /// - Bytes 16-19: `block_records` (u32, little endian) - /// - Bytes 20-27: `cumulative_records` (u64, little endian) - /// - Bytes 28-31: reservation (ignored, default value used) + /// Deserializes a `BlockRange` from bytes (layout mirrors `write_bytes`). /// /// # Panics /// @@ -209,55 +91,19 @@ impl BlockRange { } } -/// Header for a VBQ index file -/// -/// The `IndexHeader` contains metadata about an index file, including a magic number -/// for validation and the size of the indexed file. This allows verifying that an index -/// file matches its corresponding VBQ file. -/// -/// The header has a fixed size of 32 bytes to ensure compatibility across versions. +/// 32-byte header of the embedded index: `INDEX_MAGIC` (8 bytes), +/// indexed file size (8 bytes), and 16 reserved bytes. #[derive(Debug, Clone, Copy)] pub struct IndexHeader { /// Total size of the indexed VBQ file in bytes - /// - /// The serialized form also carries the `INDEX_MAGIC` magic number (8 bytes) - /// and 16 reserved bytes. bytes: u64, } impl IndexHeader { - /// Creates a new index header for a VBQ file of the specified size - /// - /// # Parameters - /// - /// * `bytes` - The total size of the VBQ file being indexed, in bytes - /// - /// # Returns - /// - /// A new `IndexHeader` instance with the appropriate magic number and size + /// Creates an index header for a VBQ file of the given size in bytes. pub fn new(bytes: u64) -> Self { Self { bytes } } - /// Reads an index header from the provided reader - /// - /// This method reads 32 bytes from the provided reader and deserializes them - /// into an `IndexHeader`. It validates the magic number to ensure that the file - /// is indeed a VBQ index file. - /// - /// # Parameters - /// - /// * `reader` - The source from which to read the header - /// - /// # Returns - /// - /// * `Ok(Self)` - If the header was successfully read and has a valid magic number - /// * `Err(_)` - If an error occurred during reading or the magic number is invalid - /// - /// # Format - /// - /// The header is expected to be 32 bytes with the following structure: - /// - Bytes 0-7: magic number (u64, little endian, must be `INDEX_MAGIC`) - /// - Bytes 8-15: file size in bytes (u64, little endian) - /// - Bytes 16-31: reserved for future extensions + /// Parses an index header from bytes, validating the magic number. pub fn from_bytes(buffer: &[u8]) -> Result { let magic = read_u64_le(&buffer[0..8]); if magic != INDEX_MAGIC { @@ -268,26 +114,7 @@ impl IndexHeader { }) } - /// Serializes the index header to a binary format and writes it to the provided writer - /// - /// This method serializes the `IndexHeader` to a fixed-size 32-byte structure and - /// writes it to the provided writer. This is typically used when saving an index to a file. - /// - /// # Parameters - /// - /// * `writer` - The destination to write the serialized header to - /// - /// # Returns - /// - /// * `Ok(())` - If the header was successfully written - /// * `Err(_)` - If an error occurred during writing - /// - /// # Format - /// - /// The header is serialized as: - /// - Bytes 0-7: magic number (u64, little endian) - /// - Bytes 8-15: file size in bytes (u64, little endian) - /// - Bytes 16-31: reserved for future extensions + /// Serializes the index header as 32 bytes to a writer. pub fn write_bytes(self, writer: &mut W) -> Result<()> { let mut buffer = [42; INDEX_HEADER_SIZE]; buffer[0..8].copy_from_slice(&INDEX_MAGIC.to_le_bytes()); @@ -297,16 +124,10 @@ impl IndexHeader { } } -/// Complete index for a VBQ file -/// -/// A `BlockIndex` contains metadata about a VBQ file and all of its blocks, -/// enabling efficient random access and parallel processing. It consists of an -/// `IndexHeader` and a collection of `BlockRange` entries, one for each block in -/// the file. +/// Complete embedded index of a VBQ file: an `IndexHeader` plus one +/// `BlockRange` per block. /// -/// The index is embedded at the end of VBQ files and can be loaded using -/// `MmapReader::load_index()`. Once loaded, it provides information about block -/// locations, sizes, and record counts. +/// Loaded from the end of a VBQ file with `MmapReader::load_index()`. /// /// # Examples /// @@ -314,7 +135,6 @@ impl IndexHeader { /// use binseq::vbq::MmapReader; /// use std::path::Path; /// -/// // Load the embedded index from a VBQ file /// let reader = MmapReader::new(Path::new("example.vbq")).unwrap(); /// let index = reader.load_index().unwrap(); /// println!("File contains {} blocks", index.n_blocks()); @@ -324,19 +144,11 @@ pub struct BlockIndex { /// Header containing metadata about the indexed file pub(crate) header: IndexHeader, - /// Collection of block ranges, one for each block in the file + /// Block ranges, one per block in the file pub(crate) ranges: Vec, } impl BlockIndex { - /// Creates a new empty block index with the specified header - /// - /// # Parameters - /// - /// * `header` - The index header containing metadata about the indexed file - /// - /// # Returns - /// - /// A new empty `BlockIndex` instance + /// Creates a new empty block index with the given header. #[must_use] pub fn new(header: IndexHeader) -> Self { Self { @@ -344,22 +156,7 @@ impl BlockIndex { ranges: Vec::default(), } } - /// Returns the number of blocks in the indexed file - /// - /// # Returns - /// - /// The number of blocks in the VBQ file described by this index - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{BlockIndex, MmapReader}; - /// use std::path::Path; - /// - /// let reader = MmapReader::new(Path::new("example.vbq")).unwrap(); - /// let index = reader.load_index().unwrap(); - /// println!("The file contains {} blocks", index.n_blocks()); - /// ``` + /// Returns the number of blocks in the indexed file. #[must_use] pub fn n_blocks(&self) -> usize { self.ranges.len() @@ -374,20 +171,7 @@ impl BlockIndex { Ok(()) } - /// Write the collection of `BlockRange` to an output handle - /// Writes all block ranges to the provided writer - /// - /// This method is used internally to write the block ranges to the embedded index. - /// It can also be used to serialize an index to any destination that implements `Write`. - /// - /// # Parameters - /// - /// * `writer` - The destination to write the block ranges to - /// - /// # Returns - /// - /// * `Ok(())` - If all block ranges were successfully written - /// * `Err(_)` - If an error occurred during writing + /// Writes all non-empty block ranges to a writer. pub fn write_range(&self, writer: &mut W) -> Result<()> { self.ranges .iter() @@ -395,14 +179,7 @@ impl BlockIndex { .try_for_each(|range| -> Result<()> { range.write_bytes(writer) }) } - /// Adds a block range to the index - /// - /// This method is used internally during index creation to add information - /// about each block in the file. Blocks are typically added in order. - /// - /// # Parameters - /// - /// * `range` - The block range to add to the index + /// Adds a block range to the index. fn add_range(&mut self, range: BlockRange) { self.ranges.push(range); } @@ -428,31 +205,7 @@ impl BlockIndex { Ok(ranges) } - /// Get a reference to the internal ranges - /// Returns a reference to the collection of block ranges - /// - /// This provides access to the metadata for all blocks in the indexed file, - /// which can be used for operations like parallel processing or random access. - /// - /// # Returns - /// - /// A slice containing all `BlockRange` entries in this index - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// use std::path::Path; - /// - /// let reader = MmapReader::new(Path::new("example.vbq")).unwrap(); - /// let index = reader.load_index().unwrap(); - /// - /// // Examine the ranges to determine which blocks to process - /// for (i, range) in index.ranges().iter().enumerate() { - /// println!("Block {}: {} records at offset {}", - /// i, range.block_records, range.start_offset); - /// } - /// ``` + /// Returns the block ranges in this index. #[must_use] pub fn ranges(&self) -> &[BlockRange] { &self.ranges diff --git a/src/vbq/mod.rs b/src/vbq/mod.rs index 12cc4bf..671e6c4 100644 --- a/src/vbq/mod.rs +++ b/src/vbq/mod.rs @@ -1,32 +1,16 @@ //! # VBQ Format //! -//! VBQ is a high-performance binary format for variable-length nucleotide sequences -//! that optimizes both storage efficiency and parallel processing capabilities. +//! VBQ is a binary format for variable-length nucleotide sequences with optional +//! quality scores, headers, and per-block ZSTD compression. //! -//! For more information on the format, please refer to our [preprint](https://www.biorxiv.org/content/10.1101/2025.04.08.647863v1). -//! -//! ## Overview -//! -//! VBQ extends the core principles of BINSEQ to accommodate: -//! -//! * **Variable-length sequences**: Unlike BINSEQ which requires fixed-length reads, VBQ can store -//! sequences of any length, making it suitable for technologies like PacBio and Oxford Nanopore. -//! -//! * **Quality scores**: Optional storage of quality scores alongside nucleotide data when needed. -//! -//! * **Sequence headers**: Optional storage of sequence identifiers/headers with each record. -//! -//! * **Block-based organization**: Data is organized into fixed-size independent record blocks -//! for efficient parallel processing. -//! -//! * **Compression**: Optional ZSTD compression of individual blocks balances storage -//! efficiency with processing speed. +//! **VBQ is superseded by [`cbq`](crate::cbq)**, which improves on its compression +//! and read throughput. VBQ remains fully supported for existing files, but new +//! projects should use CBQ. //! -//! * **Paired-end support**: Native support for paired sequences without needing multiple files. -//! -//! * **Multi-bit encoding**: Support for 2-bit and 4-bit nucleotide encodings. -//! -//! * **Embedded index**: Self-contained files with embedded index data for efficient random access. +//! Records are row-based: each record stores its fields contiguously +//! (2-bit or 4-bit encoded sequences), organized into independently +//! compressed blocks with an embedded index for random access. +//! For more information on the format, please refer to our [preprint](https://www.biorxiv.org/content/10.1101/2025.04.08.647863v1). //! //! ## File Structure //! @@ -63,28 +47,8 @@ //! * Extended sequence data (optional, for paired-end) //! * Primary quality scores (optional, if `qual` flag set) //! * Extended quality scores (optional, if paired and `qual` flag set) -//! * Primary header length (8 bytes, if `headers` flag set) -//! * Primary header data (UTF-8 string, if `headers` flag set) -//! * Extended header length (8 bytes, if paired and `headers` flag set) -//! * Extended header data (UTF-8 string, if paired and `headers` flag set) -//! -//! ## Recent Format Changes (v0.7.0+) -//! -//! * **Embedded Index**: Index data is now stored within the VBQ file itself, eliminating -//! improving portability. -//! * **Headers Support**: Optional sequence identifiers can be stored with each record. -//! * **Extended Capacity**: u64 indexing supports files with more than 4 billion records. -//! * **Multi-bit Encoding**: Support for both 2-bit and 4-bit nucleotide encodings. -//! -//! ## Performance Characteristics -//! -//! VBQ is designed for high-throughput parallel processing: -//! -//! * Independent blocks enable true parallel processing without synchronization -//! * Memory-mapped access provides efficient I/O -//! * Embedded index enables fast random access without auxiliary files -//! * Multi-bit encoding (2-bit/4-bit) optimizes storage for different use cases -//! * Optional ZSTD compression reduces file size with minimal performance impact +//! * Primary header length + data (8 bytes + UTF-8, if `headers` flag set) +//! * Extended header length + data (8 bytes + UTF-8, if paired and `headers` flag set) //! //! ## Usage Example //! diff --git a/src/vbq/reader.rs b/src/vbq/reader.rs index a4b2205..34befcc 100644 --- a/src/vbq/reader.rs +++ b/src/vbq/reader.rs @@ -1,53 +1,9 @@ -//! Reader implementation for VBQ files +//! Reader implementation for VBQ files. //! -//! This module provides functionality for reading sequence data from VBQ files, -//! including support for compressed blocks, quality scores, paired-end reads, and sequence headers. -//! -//! ## Format Changes (v0.7.0+) -//! -//! - **Embedded Index**: Readers now load the index from within VBQ files -//! - **Headers Support**: Optional sequence headers/identifiers can be read from each record -//! - **Multi-bit Encoding**: Support for reading 2-bit and 4-bit nucleotide encodings -//! - **Extended Capacity**: u64 indexing supports files with more than 4 billion records -//! -//! ## Index Loading -//! -//! The reader automatically loads the embedded index from the end of VBQ files: -//! 1. Seeks to the end of the file to read the index trailer -//! 2. Validates the `INDEX_END_MAGIC` marker -//! 3. Reads the index size and decompresses the embedded index -//! 4. Uses the index for efficient random access and parallel processing -//! -//! ## Memory-Mapped Reading -//! -//! The `MmapReader` provides efficient access to large files through memory mapping: -//! - Zero-copy access to file data -//! - Efficient random access using the embedded index -//! - Support for parallel processing across record ranges -//! -//! ## Example -//! -//! ```rust,no_run -//! use binseq::vbq::MmapReader; -//! use binseq::BinseqRecord; -//! -//! // Open a VBQ file (index is automatically loaded) -//! let mut reader = MmapReader::new("example.vbq").unwrap(); -//! let mut block = reader.new_block(); -//! -//! // Read records with headers and quality scores -//! while reader.read_block_into(&mut block).unwrap() { -//! for record in block.iter() { -//! let seq = record.sseq(); -//! let header = record.sheader(); -//! println!("Header: {}", std::str::from_utf8(header).unwrap()); -//! println!("Sequence: {}", std::str::from_utf8(seq).unwrap()); -//! if !record.squal().is_empty() { -//! println!("Quality: {}", std::str::from_utf8(record.squal()).unwrap()); -//! } -//! } -//! } -//! ``` +//! `MmapReader` memory-maps a VBQ file for zero-copy sequential or parallel +//! reading, handling compressed blocks, quality scores, paired-end reads, and +//! sequence headers. The embedded index at the end of the file drives random +//! access and parallel processing. use std::fs::File; use std::ops::Range; @@ -70,18 +26,6 @@ use crate::{ error::{ReadError, Result}, }; -/// Calculates the number of 64-bit words needed to store a nucleotide sequence of the given length -/// -/// Nucleotides are packed into 64-bit words with 2 bits per nucleotide (32 nucleotides per word). -/// This function calculates how many 64-bit words are needed to encode a sequence of a given length. -/// -/// # Parameters -/// -/// * `len` - Length of the nucleotide sequence in basepairs -/// -/// # Returns -/// -/// The number of 64-bit words required to encode the sequence /// Number of bases packed into each u64 word for the given bitsize fn bases_per_word(bitsize: BitSize) -> usize { match bitsize { @@ -138,40 +82,18 @@ struct RecordMetadata { has_quality: bool, } -/// A container for a block of VBQ records -/// -/// The `RecordBlock` struct represents a single block of records read from a VBQ file. -/// It stores the raw data for multiple records in vectors, allowing efficient iteration -/// over the records without copying memory for each record. -/// -/// ## Format Support (v0.7.0+) +/// A reusable container for one block of VBQ records. /// -/// - Supports reading records with optional sequence headers -/// - Handles both 2-bit and 4-bit nucleotide encodings -/// - Supports quality scores and paired sequences -/// - Compatible with both compressed and uncompressed blocks -/// -/// The `RecordBlock` is reused when reading blocks sequentially from a file, with its -/// contents being cleared and replaced with each new block that is read. -/// -/// # Examples -/// -/// ```rust,no_run -/// use binseq::vbq::MmapReader; -/// -/// let reader = MmapReader::new("example.vbq").unwrap(); -/// let mut block = reader.new_block(); // Create a block with appropriate size -/// ``` +/// Cleared and refilled on each `MmapReader::read_block_into` call, so records +/// can be iterated without per-record allocation. pub struct RecordBlock { /// Bitsize of the records in the block bitsize: BitSize, - /// Index of the first record in the block - /// This allows records to maintain their global position in the file + /// Global index of the first record in the block index: usize, /// Reusable buffer for temporary storage during decompression - /// Using a reusable buffer reduces memory allocations rbuf: Vec, /// Sequence data (u64s) - small copy during parsing @@ -180,8 +102,7 @@ pub struct RecordBlock { /// Record metadata stored as compact spans records: Vec, - /// Maximum size of the block in bytes - /// This is derived from the file header's block size field + /// Maximum block size in bytes (from the file header) block_size: usize, /// Reusable zstd decompression context @@ -197,20 +118,10 @@ pub struct RecordBlock { default_quality_score: u8, } impl RecordBlock { - /// Creates a new empty `RecordBlock` with the specified block size - /// - /// The block size should match the one specified in the VBQ file header - /// for proper operation. This is typically handled automatically when using - /// `MmapReader::new_block()`. + /// Creates a new empty `RecordBlock`. /// - /// # Parameters - /// - /// * `bitsize` - Bitsize of the records in the block - /// * `block_size` - Maximum size of the block in bytes - /// - /// # Returns - /// - /// A new empty `RecordBlock` instance + /// The block size must match the file header's; use `MmapReader::new_block()` + /// to get this automatically. #[must_use] pub fn new(bitsize: BitSize, block_size: usize) -> Self { Self { @@ -227,73 +138,31 @@ impl RecordBlock { } } - /// Sets the default quality score for the block - /// - /// # Parameters - /// - /// * `score` - Default quality score for the block + /// Sets the default quality score for the block. pub fn set_default_quality_score(&mut self, score: u8) { self.default_quality_score = score; self.qbuf.clear(); } - /// Returns the number of records in this block - /// - /// # Returns - /// - /// The number of records currently stored in this block + /// Returns the number of records in this block. #[must_use] pub fn n_records(&self) -> usize { self.records.len() } - /// Returns an iterator over the records in this block - /// - /// The iterator yields `RefRecord` instances that provide access to the record data - /// without copying the underlying data. - /// - /// # Returns - /// - /// An iterator over the records in this block - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// use binseq::BinseqRecord; - /// - /// let mut reader = MmapReader::new("example.vbq").unwrap(); - /// let mut block = reader.new_block(); - /// reader.read_block_into(&mut block).unwrap(); - /// - /// // Iterate over records in the block - /// for record in block.iter() { - /// println!("Record {}", record.index()); - /// } - /// ``` + /// Returns an iterator of zero-copy `RefRecord`s over the block. #[must_use] #[allow(clippy::iter_without_into_iter)] pub fn iter(&self) -> RecordBlockIter<'_> { RecordBlockIter::new(self) } - /// Updates the starting index of the block - /// - /// This is used internally to keep track of the global position of records - /// within the file, allowing each record to maintain its original index. - /// - /// # Parameters - /// - /// * `index` - The index of the first record in the block + /// Sets the global index of the first record in the block. fn update_index(&mut self, index: usize) { self.index = index; } - /// Clears all data from the block - /// - /// This method resets the block to an empty state, clearing all vectors and resetting - /// the index to 0. This is typically used when reusing a block for reading a new block - /// from a file. + /// Resets the block to an empty state for reuse. pub fn clear(&mut self) { self.index = 0; self.records.clear(); @@ -303,19 +172,7 @@ impl RecordBlock { // Note: We keep qbuf allocated for reuse } - /// Ingest the bytes from a block into the record block - /// - /// This method takes a slice of bytes and processes it to extract - /// the records from the block. It is used when reading a block from - /// a file into a record block. - /// - /// This is a private method used primarily for parallel processing. - /// - /// # Parameters - /// - /// * `bytes` - A slice of bytes containing the block data - /// * `has_quality` - A boolean indicating whether the block contains quality scores - /// * `has_header` - A boolean indicating whether the block contains headers + /// Ingests uncompressed block bytes and parses the records. fn ingest_bytes( &mut self, bytes: &[u8], @@ -480,10 +337,8 @@ impl RecordBlock { /// Decodes all sequences in the block at once. /// - /// Note: - /// Each record's sequence is padded internally to the nearest u64. - /// Because of this the global decoding will include nucleotides that are not present in the original data. - /// We track the non-contiguous regions of the sequence separately. + /// Each record's sequence is padded to the nearest u64, so the decoded + /// buffer contains padding nucleotides; per-record spans track the real regions. pub fn decode_all(&mut self) -> Result<()> { if self.sequences.is_empty() { return Ok(()); @@ -718,37 +573,11 @@ impl BinseqRecord for RefRecord<'_> { } } -/// Memory-mapped reader for VBQ files -/// -/// [`MmapReader`] provides efficient, memory-mapped access to VBQ files. It allows -/// sequential reading of record blocks and supports parallel processing of records. -/// -/// ## Format Support (v0.7.0+) -/// -/// - **Embedded Index**: Automatically loads index from within VBQ files -/// - **Headers Support**: Reads optional sequence headers/identifiers from records -/// - **Multi-bit Encoding**: Supports both 2-bit and 4-bit nucleotide encodings -/// - **Extended Capacity**: u64 indexing supports files with more than 4 billion records +/// Memory-mapped reader for VBQ files. /// -/// Memory mapping allows the operating system to lazily load file contents as needed, -/// which can be more efficient than standard file I/O, especially for large files. -/// -/// The [`MmapReader`] is designed to be used in a multi-threaded environment, and it -/// is built around [`RecordBlock`]s which are the units of data in a VBQ file. -/// Each one would be held by a separate thread and would load data from the shared -/// [`MmapReader`] through the [`MmapReader::read_block_into`] method. However, they can -/// also be used in a single-threaded environment for sequential processing. -/// -/// Each [`RecordBlock`] contains a [`BlockHeader`] and is used to access [`RefRecord`]s -/// which implement the [`BinseqRecord`] trait. -/// -/// ## Index Loading -/// -/// The reader automatically loads the embedded index by: -/// 1. Reading the index trailer from the end of the file -/// 2. Validating the `INDEX_END_MAGIC` marker -/// 3. Decompressing the embedded index data -/// 4. Using the index for efficient random access and parallel processing +/// Reads [`RecordBlock`]s sequentially via [`MmapReader::read_block_into`], or +/// processes records in parallel via the [`ParallelReader`] trait. Blocks yield +/// [`RefRecord`]s implementing the [`BinseqRecord`] trait. /// /// # Examples /// @@ -756,26 +585,16 @@ impl BinseqRecord for RefRecord<'_> { /// use binseq::vbq::MmapReader; /// use binseq::{BinseqRecord, Result}; /// -/// #[allow(deprecated)] /// fn main() -> Result<()> { -/// let path = "./data/subset.vbq"; -/// let mut reader = MmapReader::new(path)?; // Index loaded automatically -/// -/// // Create buffers for sequence data and headers +/// let mut reader = MmapReader::new("./data/subset.vbq")?; /// let mut seq_buffer = Vec::new(); /// let mut block = reader.new_block(); /// -/// // Read blocks sequentially /// while reader.read_block_into(&mut block)? { -/// println!("Read a block with {} records", block.n_records()); /// for record in block.iter() { -/// // Decode sequence and header /// record.decode_s(&mut seq_buffer)?; -/// let header = record.sheader(); -/// -/// println!("Header: {}", std::str::from_utf8(&header).unwrap_or("")); +/// println!("Header: {}", std::str::from_utf8(record.sheader()).unwrap_or("")); /// println!("Sequence: {}", std::str::from_utf8(&seq_buffer).unwrap_or("")); -/// /// seq_buffer.clear(); /// } /// } @@ -802,37 +621,7 @@ pub struct MmapReader { default_quality_score: u8, } impl MmapReader { - /// Creates a new `MmapReader` for a VBQ file - /// - /// This method opens the specified file, memory-maps its contents, reads the - /// VBQ header information, and loads the embedded index. The reader is positioned - /// at the beginning of the first record block after the header. - /// - /// ## Index Loading (v0.7.0+) - /// - /// The embedded index is automatically loaded from the end of the file. - /// - /// # Parameters - /// - /// * `path` - Path to the VBQ file to open - /// - /// # Returns - /// - /// A new `MmapReader` instance if successful - /// - /// # Errors - /// - /// * `ReadError::InvalidFileType` if the path doesn't point to a regular file - /// * I/O errors if the file can't be opened or memory-mapped - /// * Header validation errors if the file doesn't contain a valid VBQ header - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// - /// let reader = MmapReader::new("path/to/file.vbq").unwrap(); - /// ``` + /// Opens and memory-maps a VBQ file, positioned at the first record block. pub fn new>(path: P) -> Result { // Verify it's a regular file before attempting to map let file = File::open(&path)?; @@ -864,23 +653,7 @@ impl MmapReader { self.default_quality_score = score; } - /// Creates a new empty record block with the appropriate size for this file - /// - /// This creates a `RecordBlock` with a block size matching the one specified in the - /// file's header, ensuring it will be able to hold a full block of records. - /// - /// # Returns - /// - /// A new empty `RecordBlock` instance sized appropriately for this file - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// - /// let reader = MmapReader::new("example.vbq").unwrap(); - /// let mut block = reader.new_block(); - /// ``` + /// Creates a new empty `RecordBlock` sized to this file's block size. #[must_use] pub fn new_block(&self) -> RecordBlock { let mut block = RecordBlock::new(self.header.bits, self.header.block as usize); @@ -888,33 +661,12 @@ impl MmapReader { block } - /// Sets whether to decode sequences at once in each block - /// - /// # Arguments - /// - /// * `decode_block` - Whether to decode sequences at once in each block - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// - /// let mut reader = MmapReader::new("example.vbq").unwrap(); - /// reader.set_decode_block(false); - /// ``` + /// Sets whether each block's sequences are batch-decoded during parallel processing. pub fn set_decode_block(&mut self, decode_block: bool) { self.decode_block = decode_block; } - /// Returns a copy of the file's header information - /// - /// The header contains information about the file format, including whether - /// quality scores are included, whether blocks are compressed, and whether - /// records are paired. - /// - /// # Returns - /// - /// A copy of the file's `FileHeader` + /// Returns a copy of the file's header. #[must_use] pub fn header(&self) -> FileHeader { self.header @@ -926,52 +678,9 @@ impl MmapReader { self.header.is_paired() } - /// Fills an existing `RecordBlock` with the next block of records from the file + /// Reads the next block of records into `block`, decompressing if needed. /// - /// This method reads the next block of records from the current position in the file - /// and populates the provided `RecordBlock` with the data. The block is cleared and reused - /// to avoid unnecessary memory allocations. This is the primary method for sequential - /// reading of VBQ files. - /// - /// The method automatically handles decompression if the file was written with - /// compression enabled and updates the total record count as it progresses through the file. - /// - /// # Parameters - /// - /// * `block` - A mutable reference to a `RecordBlock` to be filled with data - /// - /// # Returns - /// - /// * `Ok(true)` - If a block was successfully read - /// * `Ok(false)` - If the end of the file was reached (no more blocks) - /// * `Err(_)` - If an error occurred during reading - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// use binseq::BinseqRecord; - /// use std::io::Write; - /// - /// let mut reader = MmapReader::new("example.vbq").unwrap(); - /// let mut block = reader.new_block(); - /// let mut sequence_buffer = Vec::new(); - /// - /// // Read blocks until the end of file - /// while reader.read_block_into(&mut block).unwrap() { - /// println!("Read block with {} records", block.n_records()); - /// - /// // Process each record - /// for record in block.iter() { - /// // Decode the nucleotide sequence - /// record.decode_s(&mut sequence_buffer).unwrap(); - /// - /// // Do something with the sequence - /// println!("Record {}: length {}", record.index(), sequence_buffer.len()); - /// sequence_buffer.clear(); - /// } - /// } - /// ``` + /// Returns `Ok(false)` when the end of the data blocks is reached. pub fn read_block_into(&mut self, block: &mut RecordBlock) -> Result { // Clear the block block.clear(); @@ -1035,34 +744,7 @@ impl MmapReader { Ok(true) } - /// Loads the embedded block index from this VBQ file - /// - /// The block index provides metadata about each block in the file, enabling - /// random access to blocks and parallel processing. This method reads the - /// embedded index from the end of the VBQ file. - /// - /// # Returns - /// - /// The loaded `BlockIndex` if successful - /// - /// # Errors - /// - /// * File I/O errors when reading the index - /// * Parsing errors if the VBQ file has invalid format or missing index - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::MmapReader; - /// - /// let reader = MmapReader::new("example.vbq").unwrap(); - /// - /// // Load the embedded index - /// let index = reader.load_index().unwrap(); - /// - /// // Use the index to get information about the file - /// println!("Number of blocks: {}", index.n_blocks()); - /// ``` + /// Loads the embedded block index from the end of the file. pub fn load_index(&self) -> Result { let start_pos_magic = self.mmap.len() - 8; let start_pos_index_size = start_pos_magic - 8; @@ -1093,103 +775,10 @@ impl MmapReader { } impl ParallelReader for MmapReader { - /// Processes all records in the file in parallel using multiple threads - /// - /// This method provides efficient parallel processing of VBQ files by distributing - /// blocks across multiple worker threads. The file's block structure is leveraged to divide - /// the work evenly without requiring thread synchronization during processing, which leads - /// to near-linear scaling with the number of threads. - /// - /// The method automatically loads or creates an index file to identify block boundaries, - /// then distributes the blocks among the requested number of threads. Each thread processes - /// its assigned blocks sequentially, but multiple blocks are processed in parallel across - /// threads. - /// - /// # Type Parameters - /// - /// * `P` - A type that implements the `ParallelProcessor` trait, which defines how records are processed - /// - /// # Parameters - /// - /// * `self` - Consumes the reader, as it will be used across multiple threads - /// * `processor` - An instance of a type implementing `ParallelProcessor` that will be cloned for each thread - /// * `num_threads` - Number of worker threads to use for processing - /// - /// # Returns - /// - /// * `Ok(())` - If all records were successfully processed - /// * `Err(_)` - If an error occurs during processing - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{MmapReader, RefRecord}; - /// use binseq::{ParallelProcessor, ParallelReader, BinseqRecord, Result}; - /// use std::sync::atomic::{AtomicUsize, Ordering}; - /// use std::sync::Arc; - /// - /// // Create a simple processor that counts records - /// struct RecordCounter { - /// count: Arc, - /// thread_id: usize, - /// } - /// - /// impl RecordCounter { - /// fn new() -> Self { - /// Self { - /// count: Arc::new(AtomicUsize::new(0)), - /// thread_id: 0, - /// } - /// } + /// Processes all records in parallel by distributing blocks across threads. /// - /// fn total_count(&self) -> usize { - /// self.count.load(Ordering::Relaxed) - /// } - /// } - /// - /// impl Clone for RecordCounter { - /// fn clone(&self) -> Self { - /// Self { - /// count: Arc::clone(&self.count), - /// thread_id: 0, - /// } - /// } - /// } - /// - /// impl ParallelProcessor for RecordCounter { - /// fn process_record(&mut self, _record: R) -> Result<()> { - /// self.count.fetch_add(1, Ordering::Relaxed); - /// Ok(()) - /// } - /// - /// fn on_batch_complete(&mut self) -> Result<()> { - /// // Optional: perform actions after each block is processed - /// Ok(()) - /// } - /// - /// fn set_tid(&mut self, tid: usize) { - /// self.thread_id = tid; - /// } - /// } - /// - /// // Use the processor with a VBQ file - /// let reader = MmapReader::new("example.vbq").unwrap(); - /// let counter = RecordCounter::new(); - /// - /// // Process the file with 4 threads - /// reader.process_parallel(counter.clone(), 4).unwrap(); - /// - /// // Get the total number of records processed - /// println!("Total records: {}", counter.total_count()); - /// ``` - /// - /// # Notes - /// - /// * The `ParallelProcessor` instance is cloned for each worker thread, so any shared state - /// should be wrapped in thread-safe containers like `Arc`. - /// * The `set_tid` method is called with a unique thread ID before processing begins, which - /// can be used to distinguish between worker threads. - /// * This method consumes the reader (takes ownership), as it's distributed across threads. + /// The processor is cloned per thread; share state via thread-safe + /// containers like `Arc`. fn process_parallel( self, processor: P, @@ -1199,27 +788,7 @@ impl ParallelReader for MmapReader { self.process_parallel_range(processor, num_threads, 0..num_records) } - /// Process records in parallel within a specified range - /// - /// This method allows parallel processing of a subset of records within the file, - /// defined by a start and end index. The method maps the record range to the - /// appropriate blocks and processes only the records within the specified range. - /// - /// # Arguments - /// - /// * `processor` - The processor to use for each record - /// * `num_threads` - The number of threads to spawn - /// * `start` - The starting record index (inclusive) - /// * `end` - The ending record index (exclusive) - /// - /// # Type Parameters - /// - /// * `P` - A type that implements `ParallelProcessor` and can be cloned - /// - /// # Returns - /// - /// * `Ok(())` - If all records were processed successfully - /// * `Err(Error)` - If an error occurred during processing + /// Processes only the records within `range` in parallel. fn process_parallel_range( self, processor: P, diff --git a/src/vbq/writer.rs b/src/vbq/writer.rs index e5641be..9e245a3 100644 --- a/src/vbq/writer.rs +++ b/src/vbq/writer.rs @@ -1,66 +1,13 @@ -//! Writer implementation for VBQ files +//! Writer implementation for VBQ files. //! -//! This module provides functionality for writing sequence data to VBQ files, -//! including support for compression, quality scores, paired-end reads, and sequence headers. -//! -//! The VBQ writer implements a block-based approach where records are packed -//! into fixed-size blocks. Each block has a header containing metadata about the -//! records it contains. Blocks may be optionally compressed using zstd compression. -//! -//! ## Format Changes (v0.7.0+) -//! -//! - **Embedded Index**: Writers now automatically embed an index at the end of the file -//! - **Headers Support**: Optional sequence headers/identifiers can be written with each record -//! - **Multi-bit Encoding**: Support for 2-bit and 4-bit nucleotide encodings -//! - **Extended Capacity**: u64 indexing supports more than 4 billion records -//! -//! ## File Structure Written +//! Records are packed into fixed-size blocks, optionally zstd-compressed, +//! producing: //! //! ```text //! [File Header][Data Blocks][Compressed Index][Index Size][Index End Magic] //! ``` //! -//! The writer automatically: -//! 1. Writes the file header -//! 2. Writes data blocks as records are added -//! 3. Builds an index during writing -//! 4. On `finish()`, compresses and embeds the index at the end of the file -//! -//! # Example -//! -//! ```rust,no_run -//! use binseq::vbq::{WriterBuilder, FileHeaderBuilder}; -//! use binseq::SequencingRecordBuilder; -//! use std::fs::File; -//! -//! // Create a VBQ file writer with headers and compression -//! let file = File::create("example.vbq").unwrap(); -//! let header = FileHeaderBuilder::new() -//! .block(128 * 1024) -//! .qual(true) -//! .compressed(true) -//! .headers(true) -//! .flags(true) -//! .build(); -//! -//! let mut writer = WriterBuilder::default() -//! .header(header) -//! .build(file) -//! .unwrap(); -//! -//! // Write a nucleotide sequence with quality scores and header -//! let record = SequencingRecordBuilder::default() -//! .s_seq(b"ACGTACGTACGT") -//! .s_qual(b"IIIIIIIIIIII") -//! .s_header(b"sequence_001") -//! .flag(0) -//! .build() -//! .unwrap(); -//! writer.push(record).unwrap(); -//! -//! // Must call finish() to write the embedded index -//! writer.finish().unwrap(); -//! ``` +//! Call `finish()` when done so the embedded index is written. use std::io::Write; @@ -75,33 +22,7 @@ use crate::vbq::header::{SIZE_BLOCK_HEADER, SIZE_HEADER}; use crate::vbq::index::{INDEX_END_MAGIC, IndexHeader}; use crate::vbq::{BlockIndex, BlockRange}; -/// A builder for creating configured `Writer` instances -/// -/// This builder provides a fluent interface for configuring and creating a -/// `Writer` with customized settings. It allows specifying the file header, -/// encoding policy, and whether to operate in headless mode. -/// -/// # Examples -/// -/// ```rust,no_run -/// use binseq::vbq::{WriterBuilder, FileHeaderBuilder}; -/// use binseq::Policy; -/// use std::fs::File; -/// -/// // Create a writer with custom settings -/// let file = File::create("example.vbq").unwrap(); -/// let mut writer = WriterBuilder::default() -/// .header(FileHeaderBuilder::new() -/// .block(65536) -/// .qual(true) -/// .compressed(true) -/// .build()) -/// .policy(Policy::IgnoreSequence) -/// .build(file) -/// .unwrap(); -/// -/// // Use the writer... -/// ``` +/// Builder for configured [`Writer`] instances. #[derive(Default)] pub struct WriterBuilder { /// Header of the file @@ -112,122 +33,29 @@ pub struct WriterBuilder { headless: Option, } impl WriterBuilder { - /// Sets the header for the VBQ file - /// - /// The header defines the file format parameters such as block size, whether - /// the file contains quality scores, paired-end reads, and compression settings. - /// - /// # Parameters - /// - /// * `header` - The `FileHeader` to use for the file - /// - /// # Returns - /// - /// The builder with the header configured - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder, FileHeaderBuilder}; - /// - /// // Create a header with 64KB blocks and quality scores - /// let header = FileHeaderBuilder::new() - /// .block(65536) - /// .qual(true) - /// .paired(true) - /// .compressed(true) - /// .build(); - /// - /// let builder = WriterBuilder::default().header(header); - /// ``` + /// Sets the file header (block size, quality, pairing, compression, etc.). #[must_use] pub fn header(mut self, header: FileHeader) -> Self { self.header = Some(header); self } - /// Sets the encoding policy for nucleotide sequences - /// - /// The policy determines how sequences are encoded into the binary format. - /// Different policies offer trade-offs between compression ratio and compatibility - /// with different types of sequence data. - /// - /// # Parameters - /// - /// * `policy` - The encoding policy to use - /// - /// # Returns - /// - /// The builder with the encoding policy configured - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder}; - /// use binseq::Policy; - /// - /// let builder = WriterBuilder::default().policy(Policy::IgnoreSequence); - /// ``` + /// Sets the encoding policy for invalid nucleotides. #[must_use] pub fn policy(mut self, policy: Policy) -> Self { self.policy = Some(policy); self } - /// Sets whether to operate in headless mode - /// - /// In headless mode, the writer does not write a file header. This is useful - /// when creating part of a file that will be merged with other parts later, - /// such as in parallel writing scenarios. - /// - /// # Parameters - /// - /// * `headless` - Whether to operate in headless mode - /// - /// # Returns - /// - /// The builder with the headless mode configured - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::WriterBuilder; - /// - /// // Create a headless writer for parallel writing - /// let builder = WriterBuilder::default().headless(true); - /// ``` + /// Sets headless mode: skips writing the file header, for parts that + /// will be merged into another file later (e.g. parallel writing). #[must_use] pub fn headless(mut self, headless: bool) -> Self { self.headless = Some(headless); self } - /// Builds a `Writer` with the configured settings - /// - /// This finalizes the builder and creates a new `Writer` instance using - /// the provided writer and the configured settings. If any settings were not - /// explicitly set, default values will be used. - /// - /// # Parameters - /// - /// * `inner` - The underlying writer where data will be written - /// - /// # Returns - /// - /// * `Ok(Writer)` - A configured `Writer` ready for use - /// * `Err(_)` - If an error occurred while initializing the writer - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::WriterBuilder; - /// use std::fs::File; - /// - /// let file = File::create("example.vbq").unwrap(); - /// let mut writer = WriterBuilder::default() - /// .build(file) - /// .unwrap(); - /// ``` + /// Builds a `Writer` around `inner`, using defaults for unset options. pub fn build(self, inner: W) -> Result> { Writer::new( inner, @@ -238,54 +66,11 @@ impl WriterBuilder { } } -/// Writer for VBQ format files -/// -/// The `Writer` handles writing nucleotide sequence data to VBQ files in a -/// block-based format. It manages the file structure, compression settings, and ensures -/// data is properly encoded and organized. -/// -/// ## File Structure -/// -/// A VBQ file consists of: -/// 1. A file header that defines parameters like block size and compression settings -/// 2. A series of blocks, each with: -/// - A block header with metadata (e.g., record count) -/// - A collection of encoded records +/// Block-based writer for VBQ files. /// -/// Each block is filled with records until either the block is full or no more complete -/// records can fit. The writer automatically handles block boundaries and creates new -/// blocks as needed. -/// -/// ## Usage -/// -/// The writer supports multiple formats: -/// - Single-end sequences with or without quality scores -/// - Paired-end sequences with or without quality scores -/// -/// It's recommended to use the `WriterBuilder` to create and configure a writer -/// instance with the appropriate settings. -/// -/// ```rust,no_run -/// use binseq::vbq::{WriterBuilder, FileHeader}; -/// use binseq::SequencingRecordBuilder; -/// use std::fs::File; -/// -/// // Create a writer for single-end reads -/// let file = File::create("example.vbq").unwrap(); -/// let mut writer = WriterBuilder::default() -/// .header(FileHeader::default()) -/// .build(file) -/// .unwrap(); -/// -/// // Write a sequence -/// let record = SequencingRecordBuilder::default() -/// .s_seq(b"ACGTACGTACGT") -/// .build() -/// .unwrap(); -/// writer.push(record).unwrap(); -/// -/// // Writer automatically flushes when dropped -/// ``` +/// Fills fixed-size blocks with encoded records, starting a new block when a +/// record would overflow the current one. Create via [`WriterBuilder`]; call +/// [`Writer::finish`] (also invoked on drop) to flush and embed the index. #[derive(Clone)] pub struct Writer { /// Inner Writer @@ -336,49 +121,14 @@ impl Writer { Ok(wtr) } - /// Initializes the writer by writing the file header - /// - /// This method is called automatically during creation unless headless mode is enabled. - /// It writes the `FileHeader` to the underlying writer. - /// - /// # Returns - /// - /// * `Ok(())` - If the header was successfully written - /// * `Err(_)` - If an error occurred during writing + /// Writes the file header; called on creation unless headless. fn init(&mut self) -> Result<()> { self.header.write_bytes(&mut self.inner)?; self.bytes_written += SIZE_HEADER; Ok(()) } - /// Checks if the writer is configured for paired-end reads - /// - /// This method returns whether the writer expects paired-end reads based on the - /// header settings. If true, you should use `write_paired_nucleotides` instead of - /// `write_nucleotides` to write sequences. - /// - /// # Returns - /// - /// `true` if the writer is configured for paired-end reads, `false` otherwise - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder, FileHeader}; - /// use std::fs::File; - /// - /// // Create a header for paired-end reads - /// let mut header = FileHeader::default(); - /// header.paired = true; - /// - /// let file = File::create("paired_reads.vbq").unwrap(); - /// let writer = WriterBuilder::default() - /// .header(header) - /// .build(file) - /// .unwrap(); - /// - /// assert!(writer.is_paired()); - /// ``` + /// Returns whether the writer is configured for paired-end reads. pub fn is_paired(&self) -> bool { self.header.paired } @@ -393,35 +143,7 @@ impl Writer { self.encoder.policy() } - /// Checks if the writer is configured for quality scores - /// - /// This method returns whether the writer expects quality scores based on the - /// header settings. If true, you should use methods that include quality scores - /// (`write_nucleotides_with_quality` or `write_paired_nucleotides_with_quality`) - /// to write sequences. - /// - /// # Returns - /// - /// `true` if the writer is configured for quality scores, `false` otherwise - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder, FileHeader}; - /// use std::fs::File; - /// - /// // Create a header for sequences with quality scores - /// let mut header = FileHeader::default(); - /// header.qual = true; - /// - /// let file = File::create("reads_with_quality.vbq").unwrap(); - /// let writer = WriterBuilder::default() - /// .header(header) - /// .build(file) - /// .unwrap(); - /// - /// assert!(writer.has_quality()); - /// ``` + /// Returns whether the writer is configured for quality scores. pub fn has_quality(&self) -> bool { self.header.qual } @@ -464,21 +186,10 @@ impl Writer { ) } - /// Writes a record using the unified [`SequencingRecord`] API + /// Writes a [`SequencingRecord`], the unified API shared with the BQ and CBQ writers. /// - /// This method provides a consistent interface with BQ and CBQ writers. - /// It automatically routes to the single or paired write path based on - /// whether the record contains paired data. - /// - /// # Arguments - /// - /// * `record` - A [`SequencingRecord`] containing the sequence data to write - /// - /// # Returns - /// - /// * `Ok(true)` if the record was written successfully - /// * `Ok(false)` if the record was skipped due to invalid nucleotides - /// * `Err(_)` if writing failed + /// Returns `Ok(false)` if the record was skipped by the encoding policy + /// (e.g. invalid nucleotides). /// /// # Examples /// @@ -487,11 +198,7 @@ impl Writer { /// use binseq::SequencingRecordBuilder; /// use std::fs::File; /// - /// let header = FileHeaderBuilder::new() - /// .qual(true) - /// .headers(true) - /// .build(); - /// + /// let header = FileHeaderBuilder::new().qual(true).headers(true).build(); /// let mut writer = WriterBuilder::default() /// .header(header) /// .build(File::create("example.vbq").unwrap()) @@ -578,42 +285,10 @@ impl Writer { Ok(true) } - /// Finishes writing and flushes all data to the underlying writer - /// - /// This method should be called when you're done writing to ensure all data - /// is properly flushed to the underlying writer. It's automatically called - /// when the writer is dropped, but calling it explicitly allows you to handle - /// any errors that might occur during flushing. - /// - /// # Returns - /// - /// * `Ok(())` - If all data was successfully flushed - /// * `Err(_)` - If an error occurred during flushing - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder, FileHeader}; - /// use binseq::SequencingRecordBuilder; - /// use std::fs::File; + /// Flushes remaining data and writes the embedded index. /// - /// let file = File::create("example.vbq").unwrap(); - /// let mut writer = WriterBuilder::default() - /// .build(file) - /// .unwrap(); - /// - /// // Write some sequences... - /// let record = SequencingRecordBuilder::default() - /// .s_seq(b"ACGTACGTACGT") - /// .build() - /// .unwrap(); - /// writer.push(record).unwrap(); - /// - /// // Manually finish and check for errors - /// if let Err(e) = writer.finish() { - /// eprintln!("Error flushing data: {}", e); - /// } - /// ``` + /// Called automatically on drop, but calling it explicitly lets you handle + /// errors. Idempotent: the index is only written once. pub fn finish(&mut self) -> Result<()> { // Flush any remaining data in the current block self.flush_block()?; @@ -638,57 +313,11 @@ impl Writer { &mut self.cblock } - /// Ingests data from another `Writer` that uses a `Vec` as its inner writer - /// - /// This method is particularly useful for parallel processing, where multiple writers - /// might be writing to memory buffers and need to be combined into a single file. It - /// transfers all complete blocks and any partial blocks from the other writer into this one. - /// - /// The method clears the other writer's buffer after ingestion, allowing it to be reused. - /// - /// # Parameters - /// - /// * `other` - Another `Writer` whose inner writer is a `Vec` - /// - /// # Returns + /// Transfers all blocks (complete and partial) from an in-memory writer + /// into this one, clearing the other for reuse. Used to combine per-thread + /// buffers in parallel writing. /// - /// * `Ok(())` - If ingestion was successful - /// * `Err(_)` - If an error occurred during ingestion or if the headers are incompatible - /// - /// # Errors - /// - /// Returns an error if: - /// - The headers of the two writers are not compatible (`WriteError::IncompatibleHeaders`) - /// - An I/O error occurred during data transfer - /// - /// # Examples - /// - /// ```rust,no_run - /// use binseq::vbq::{WriterBuilder, FileHeader}; - /// use binseq::SequencingRecordBuilder; - /// use std::fs::File; - /// - /// // Create a file writer - /// let file = File::create("combined.vbq").unwrap(); - /// let mut file_writer = WriterBuilder::default() - /// .build(file) - /// .unwrap(); - /// - /// // Create a memory writer - /// let mut mem_writer = WriterBuilder::default() - /// .build(Vec::new()) - /// .unwrap(); - /// - /// // Write some data to the memory writer - /// let record = SequencingRecordBuilder::default() - /// .s_seq(b"ACGTACGT") - /// .build() - /// .unwrap(); - /// mem_writer.push(record).unwrap(); - /// - /// // Ingest data from memory writer into file writer - /// file_writer.ingest(&mut mem_writer).unwrap(); - /// ``` + /// Errors if the two writers' headers are incompatible. pub fn ingest(&mut self, other: &mut Writer>) -> Result<()> { if self.header != other.header { return Err(WriteError::IncompatibleHeaders(self.header, other.header).into()); diff --git a/src/write.rs b/src/write.rs index f4b072f..ed08674 100644 --- a/src/write.rs +++ b/src/write.rs @@ -1,7 +1,7 @@ //! Unified writer interface for BINSEQ formats //! -//! This module provides a unified `BinseqWriter` enum that abstracts over the three -//! BINSEQ format writers (BQ, VBQ, CBQ), allowing format-agnostic writing of sequence data. +//! [`BinseqWriter`] abstracts over the three format writers (CBQ, BQ, VBQ), +//! allowing format-agnostic writing of sequence data. //! //! # Example //! @@ -9,8 +9,8 @@ //! use binseq::{write::{BinseqWriter, BinseqWriterBuilder, Format}, SequencingRecordBuilder}; //! use std::io::Cursor; //! -//! // Create a VBQ writer with quality scores and headers -//! let mut writer = BinseqWriterBuilder::new(Format::Vbq) +//! // Create a CBQ writer with quality scores and headers +//! let mut writer = BinseqWriterBuilder::new(Format::Cbq) //! .paired(false) //! .quality(true) //! .headers(true) @@ -31,7 +31,7 @@ //! //! # Parallel Writing //! -//! For parallel writing scenarios, use `headless(true)` for thread-local writers +//! For parallel writing, use `headless(true)` for thread-local writers //! and `ingest()` to merge them into a global writer: //! //! ```rust,no_run @@ -39,9 +39,9 @@ //! use std::fs::File; //! //! // Global writer (writes header) -//! let mut global = BinseqWriterBuilder::new(Format::Vbq) +//! let mut global = BinseqWriterBuilder::new(Format::Cbq) //! .paired(false) -//! .build(File::create("output.vbq").unwrap()) +//! .build(File::create("output.cbq").unwrap()) //! .unwrap(); //! //! // Thread-local writer (headless, Vec buffer) @@ -305,15 +305,9 @@ impl BinseqWriterBuilder { /// Encode FASTX file(s) to BINSEQ format /// - /// This method returns a [`FastxEncoderBuilder`](crate::utils::FastxEncoderBuilder) that allows you to configure - /// the input source and threading options before executing the encoding. - /// - /// This is an alternative to [`build`](Self::build) that directly processes - /// FASTX files using parallel processing. - /// - /// # Availability - /// - /// This method is only available when the `paraseq` feature is enabled. + /// Returns a [`FastxEncoderBuilder`](crate::utils::FastxEncoderBuilder) for configuring + /// the input source and threading before running the (parallel) encoding. + /// Requires the `paraseq` feature. /// /// # Example /// @@ -349,11 +343,7 @@ impl BinseqWriterBuilder { /// Build the writer /// - /// # Errors - /// - /// Returns an error if: - /// - Format is BQ and `slen` is not set - /// - Format is BQ, `paired` is true, but `xlen` is not set + /// Errors for BQ if `slen` (or `xlen` when paired) is not set. pub fn build(self, writer: W) -> Result> { match self.format { Format::Bq => self.build_bq(writer), @@ -451,8 +441,7 @@ impl BinseqWriterBuilder { /// Unified writer for BINSEQ formats /// -/// This enum wraps the three format-specific writers (BQ, VBQ, CBQ) and provides -/// a unified interface for writing sequence data. +/// Wraps the three format-specific writers behind one interface. // `cbq::ColumnarBlockWriter` is intrinsically larger than the other variants (it holds a // reusable `ColumnarBlock` encode buffer). Boxing it would shrink this enum but is a breaking // change to the variant's public field type, so it's left as-is rather than churn downstream @@ -470,14 +459,9 @@ pub enum BinseqWriter { impl BinseqWriter { /// Push a record to the writer /// - /// Returns `Ok(true)` if the record was written successfully, or `Ok(false)` - /// if the record was skipped due to invalid nucleotides (based on the configured - /// policy). CBQ always returns `Ok(true)` as it handles N's explicitly. - /// - /// # Errors - /// - /// Returns an error if there's an I/O error or if the record doesn't match - /// the writer's configuration (e.g., paired record to unpaired writer). + /// Returns `Ok(false)` if the record was skipped due to invalid nucleotides + /// (per the configured [`Policy`]). CBQ stores `N`s natively and always + /// returns `Ok(true)`. pub fn push(&mut self, record: SequencingRecord) -> Result { match self { Self::Bq(w) => w.push(record), @@ -488,12 +472,7 @@ impl BinseqWriter { /// Finish writing and flush any remaining data /// - /// For VBQ and CBQ formats, this writes the embedded index. For BQ, this - /// is equivalent to `flush()`. - /// - /// # Errors - /// - /// Returns an error if there's an I/O error writing the final data. + /// For CBQ and VBQ this writes the embedded index; for BQ it is equivalent to `flush()`. pub fn finish(&mut self) -> Result<()> { match self { Self::Bq(w) => w.flush(), @@ -560,15 +539,8 @@ impl Clone for BinseqWriter { impl BinseqWriter { /// Ingest records from a headless `Vec` writer into this writer /// - /// This is used in parallel writing scenarios where thread-local writers - /// buffer to `Vec` and then get merged into a global writer. - /// - /// # Errors - /// - /// Returns an error if: - /// - The source and destination writers have different formats - /// - The source and destination writers have incompatible headers - /// - There's an I/O error during ingestion + /// Used in parallel writing to merge thread-local buffers into a global writer. + /// Errors if the writers have incompatible formats or headers. pub fn ingest(&mut self, other: &mut BinseqWriter>) -> Result<()> { match (self, other) { (Self::Bq(dst), BinseqWriter::Bq(src)) => dst.ingest(src), @@ -578,19 +550,11 @@ impl BinseqWriter { } } - /// Ingest *completed* records from a headless `Vec` writer into this writer - /// - /// This is used in parallel writing scenarios where thread-local writers - /// buffer to `Vec` and then get merged into a global writer. - /// - /// Currently only different for CBQ + /// Ingest only *completed* blocks from a headless `Vec` writer /// - /// # Errors - /// - /// Returns an error if: - /// - The source and destination writers have different formats - /// - The source and destination writers have incompatible headers - /// - There's an I/O error during ingestion + /// Like [`ingest`](Self::ingest), but for CBQ it drains only already-compressed + /// blocks, leaving the in-progress block accumulating on the worker thread. + /// Identical to `ingest` for the other formats. pub fn ingest_completed(&mut self, other: &mut BinseqWriter>) -> Result<()> { match (self, other) { (Self::Cbq(dst), BinseqWriter::Cbq(src)) => dst.ingest_completed(src), @@ -600,14 +564,9 @@ impl BinseqWriter { } impl BinseqWriter { - /// Create a new headless writer with the same configuration, using a `Vec` buffer - /// - /// This is useful for parallel writing scenarios where each thread has its own - /// buffer that gets merged into a global writer via `ingest()`. - /// - /// # Errors + /// Create a headless writer with the same configuration, buffering to `Vec` /// - /// Returns an error if the writer cannot be created. + /// Used for per-thread writers that are merged into a global writer via `ingest()`. pub fn new_headless_buffer(&self) -> Result>> { match self { Self::Bq(w) => {