Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 5 additions & 4 deletions src/commands/clip.rs
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@ use std::sync::atomic::{AtomicU64, Ordering};
use super::command::Command;
use super::common::{
BamIoOptions, CompressionOptions, QueueMemoryOptions, SchedulerOptions, ThreadingOptions,
build_pipeline_config,
build_pipeline_config, parse_bool,
};

/// Clips reads in a BAM file to remove overlaps
Expand Down Expand Up @@ -100,7 +100,7 @@ pub struct Clip {
pub sort_order: Option<String>,

/// Clip overlapping read pairs
#[arg(long = "clip-overlapping-reads", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "clip-overlapping-reads", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub clip_overlapping_reads: bool,

/// Clip reads that extend past their mate's start position
Expand All @@ -111,6 +111,7 @@ pub struct Clip {
num_args = 0..=1,
default_missing_value = "true",
action = clap::ArgAction::Set,
value_parser = parse_bool,
)]
pub clip_extending_past_mate: bool,

Expand All @@ -131,11 +132,11 @@ pub struct Clip {
pub read_two_three_prime: usize,

/// Upgrade existing clipping to the specified clipping mode
#[arg(short = 'H', long = "upgrade-clipping", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'H', long = "upgrade-clipping", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub upgrade_clipping: bool,

/// Automatically clip extended attributes that match read length
#[arg(short = 'a', long = "auto-clip-attributes", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'a', long = "auto-clip-attributes", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub auto_clip_attributes: bool,

/// Output file for clipping metrics
Expand Down
94 changes: 88 additions & 6 deletions src/commands/common.rs
Original file line number Diff line number Diff line change
Expand Up @@ -104,11 +104,11 @@ pub struct ConsensusCallingOptions {
pub min_input_base_quality: u8,

/// Produce per-base tags (cd, ce) in addition to per-read tags
#[arg(short = 'B', long = "output-per-base-tags", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'B', long = "output-per-base-tags", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub output_per_base_tags: bool,

/// Quality-trim reads before consensus calling (removes low-quality bases from ends)
#[arg(long = "trim", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "trim", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub trim: bool,

/// Minimum consensus base quality (output consensus bases below this are masked to N)
Expand Down Expand Up @@ -214,7 +214,7 @@ impl Default for ReadGroupOptions {
#[derive(Debug, Clone, Args)]
pub struct OverlappingConsensusOptions {
/// Consensus call overlapping bases in read pairs before UMI consensus calling
#[arg(long = "consensus-call-overlapping-bases", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "consensus-call-overlapping-bases", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub consensus_call_overlapping_bases: bool,
}

Expand Down Expand Up @@ -315,7 +315,7 @@ pub struct SchedulerOptions {
///
/// Shows per-step timing, throughput, contention metrics, and
/// per-thread work distribution.
#[arg(long = "pipeline-stats", default_value_t = false, hide = true)]
#[arg(long = "pipeline-stats", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool, hide = true)]
pub pipeline_stats: bool,

/// Timeout in seconds for deadlock detection (default: 10, 0 = disabled).
Expand All @@ -329,7 +329,7 @@ pub struct SchedulerOptions {
///
/// Uses progressive doubling: 2x -> 4x -> unbind, with restoration
/// after 30s of sustained progress.
#[arg(long = "deadlock-recover", default_value_t = false, hide = true)]
#[arg(long = "deadlock-recover", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool, hide = true)]
pub deadlock_recover: bool,
}

Expand Down Expand Up @@ -451,7 +451,7 @@ pub struct QueueMemoryOptions {
/// When true, total memory = queue-memory * threads. For example,
/// --queue-memory 768 with --threads 16 allocates 12 GB total.
/// Set to false for a fixed total memory budget regardless of thread count.
#[arg(long = "queue-memory-per-thread", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "queue-memory-per-thread", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub queue_memory_per_thread: bool,

/// DEPRECATED: Use --queue-memory instead. Memory limit for pipeline queues in megabytes.
Expand Down Expand Up @@ -623,6 +623,16 @@ impl QueueMemoryOptions {
}
}

/// Parses a boolean value from a string, accepting: true/false, yes/no, y/n, t/f
/// (case-insensitive). Matches sopt/fgbio behavior.
pub(crate) fn parse_bool(s: &str) -> Result<bool, String> {
match s.to_ascii_lowercase().as_str() {
"true" | "t" | "yes" | "y" => Ok(true),
"false" | "f" | "no" | "n" => Ok(false),
_ => Err(format!("Invalid boolean value '{s}'. Expected: true|false|yes|no|y|n|t|f")),
}
}

/// Parses a memory size string into bytes.
///
/// Accepts both plain numbers (interpreted as MB) and human-readable formats like:
Expand Down Expand Up @@ -1208,4 +1218,76 @@ mod tests {
let cmd = TestBoolFlags::try_parse_from(args).expect("valid CLI args should parse");
assert_eq!(cmd.queue_memory.queue_memory_per_thread, expected);
}

#[rstest]
#[case("true", true)]
#[case("false", false)]
#[case("yes", true)]
#[case("no", false)]
#[case("t", true)]
#[case("f", false)]
#[case("y", true)]
#[case("n", false)]
#[case("True", true)]
#[case("TRUE", true)]
#[case("False", false)]
#[case("FALSE", false)]
#[case("Yes", true)]
#[case("YES", true)]
#[case("No", false)]
#[case("NO", false)]
#[case("T", true)]
#[case("F", false)]
#[case("Y", true)]
#[case("N", false)]
#[case("tRuE", true)]
#[case("fAlSe", false)]
#[case("yEs", true)]
fn test_parse_bool_valid(#[case] input: &str, #[case] expected: bool) {
assert_eq!(parse_bool(input).expect("should parse"), expected);
}

#[rstest]
#[case("")]
#[case("tru")]
#[case("fals")]
#[case("truee")]
#[case("noo")]
#[case("yess")]
#[case("maybe")]
#[case("0")]
#[case("1")]
#[case("on")]
#[case("off")]
#[case(" true")]
#[case("true ")]
fn test_parse_bool_invalid(#[case] input: &str) {
assert!(parse_bool(input).is_err(), "expected error for input: {input:?}");
}

#[rstest]
#[case(&["test", "--trim", "yes"], true)]
#[case(&["test", "--trim", "no"], false)]
#[case(&["test", "--trim", "y"], true)]
#[case(&["test", "--trim", "n"], false)]
#[case(&["test", "--trim", "t"], true)]
#[case(&["test", "--trim", "f"], false)]
#[case(&["test", "--trim", "YES"], true)]
#[case(&["test", "--trim", "NO"], false)]
#[case(&["test", "--trim=yes"], true)]
#[case(&["test", "--trim=no"], false)]
fn test_extended_bool_values_in_cli(#[case] args: &[&str], #[case] expected: bool) {
let cmd = TestBoolFlags::try_parse_from(args).expect("valid CLI args should parse");
assert_eq!(cmd.consensus.trim, expected);
}

#[rstest]
#[case(&["test", "--trim", "maybe"])]
#[case(&["test", "--trim", "0"])]
#[case(&["test", "--trim", "1"])]
#[case(&["test", "--trim", "on"])]
#[case(&["test", "--trim", "off"])]
fn test_extended_bool_values_in_cli_invalid(#[case] args: &[&str]) {
assert!(TestBoolFlags::try_parse_from(args).is_err());
}
}
5 changes: 3 additions & 2 deletions src/commands/compare/bams.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@ use std::path::{Path, PathBuf};
use std::thread;

use crate::commands::command::Command;
use crate::commands::common::parse_bool;

use super::raw_compare::{raw_compare_structured, raw_records_byte_equal};

Expand Down Expand Up @@ -164,14 +165,14 @@ pub struct CompareBams {
pub max_diffs: usize,

/// Quiet mode - only exit code indicates result (0=equal, 1=different)
#[arg(short = 'q', long = "quiet")]
#[arg(short = 'q', long = "quiet", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub quiet: bool,

/// Ignore record order when comparing in grouping mode.
/// Required for comparing output from consensus commands (simplex/duplex/codec)
/// when run with --threads, as parallel processing causes non-deterministic ordering.
/// Only valid with --mode grouping.
#[arg(long = "ignore-order", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "ignore-order", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub ignore_order: bool,

/// Initial buffer size for --ignore-order mode (number of records)
Expand Down
5 changes: 3 additions & 2 deletions src/commands/compare/metrics.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
//! This is useful for comparing metrics files produced by fgbio and fgumi,
//! which may have slightly different floating-point representations.

use crate::commands::common::parse_bool;
use anyhow::Result;
use clap::Parser;
use fgumi_lib::logging::OperationTimer;
Expand Down Expand Up @@ -69,11 +70,11 @@ pub struct CompareMetrics {
pub max_diffs: usize,

/// Quiet mode - only exit code indicates result (0=equal, 1=different)
#[arg(short = 'q', long = "quiet")]
#[arg(short = 'q', long = "quiet", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub quiet: bool,

/// Verbose mode - print success message when files match
#[arg(short = 'v', long = "verbose")]
#[arg(short = 'v', long = "verbose", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub verbose: bool,
}

Expand Down
6 changes: 3 additions & 3 deletions src/commands/correct.rs
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ use std::sync::atomic::{AtomicU64, Ordering};
use crate::commands::command::Command;
use crate::commands::common::{
BamIoOptions, CompressionOptions, QueueMemoryOptions, RejectsOptions, SchedulerOptions,
ThreadingOptions, build_pipeline_config,
ThreadingOptions, build_pipeline_config, parse_bool,
};

/// Result of matching an observed UMI to an expected UMI.
Expand Down Expand Up @@ -229,7 +229,7 @@ pub struct CorrectUmis {
pub umi_tag: String,

/// Don't store original UMIs in a separate tag.
#[arg(long)]
#[arg(long, default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub dont_store_original_umis: bool,

/// Size of the LRU cache for UMI matching.
Expand All @@ -241,7 +241,7 @@ pub struct CorrectUmis {
pub min_corrected: Option<f64>,

/// Reverse complement UMIs before matching.
#[arg(long)]
#[arg(long, default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub revcomp: bool,

/// Threading options for parallel processing.
Expand Down
8 changes: 4 additions & 4 deletions src/commands/dedup.rs
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@ use serde::{Deserialize, Serialize};
use crate::commands::command::Command;
use crate::commands::common::{
BamIoOptions, CompressionOptions, QueueMemoryOptions, SchedulerOptions, ThreadingOptions,
build_pipeline_config,
build_pipeline_config, parse_bool,
};
use fgumi_lib::sort::PA_TAG;
use fgumi_lib::sort::bam_fields;
Expand Down Expand Up @@ -1073,7 +1073,7 @@ pub struct MarkDuplicates {
pub family_size_histogram: Option<PathBuf>,

/// Remove duplicates instead of just marking them
#[arg(short = 'r', long = "remove-duplicates", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'r', long = "remove-duplicates", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub remove_duplicates: bool,

/// The tag containing the raw UMI sequence
Expand All @@ -1097,7 +1097,7 @@ pub struct MarkDuplicates {
pub min_map_q: Option<u8>,

/// Include reads flagged as not passing QC
#[arg(short = 'n', long = "include-non-pf-reads", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'n', long = "include-non-pf-reads", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub include_non_pf_reads: bool,

/// UMI grouping strategy
Expand Down Expand Up @@ -1126,7 +1126,7 @@ pub struct MarkDuplicates {

/// Skip UMI-based grouping; group by position only. Forces identity strategy
/// and ignores any existing UMI tags.
#[arg(long = "no-umi")]
#[arg(long = "no-umi", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub no_umi: bool,

/// Scheduler and pipeline options
Expand Down
4 changes: 2 additions & 2 deletions src/commands/downsample.rs
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ use std::io::Write;
use std::path::PathBuf;

use crate::commands::command::Command;
use crate::commands::common::{BamIoOptions, CompressionOptions};
use crate::commands::common::{BamIoOptions, CompressionOptions, parse_bool};

/// MI tag for molecular identifier
const MI_TAG: Tag = Tag::new(b'M', b'I');
Expand Down Expand Up @@ -76,7 +76,7 @@ pub struct Downsample {
pub seed: Option<u64>,

/// Validate that MI tags appear in consecutive groups (error if seen non-consecutively)
#[arg(long = "validate-mi-order", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "validate-mi-order", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub validate_mi_order: bool,

/// Output file for kept family size histogram
Expand Down
3 changes: 2 additions & 1 deletion src/commands/duplex_metrics.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
//! - Ideal duplex fraction calculation using proper binomial CDF
//! - Optional interval filtering (BED or Picard interval list format) to restrict analysis to specific regions

use crate::commands::common::parse_bool;
use anyhow::{Context, Result};
use clap::Parser;
use fgoxide::io::DelimFile;
Expand Down Expand Up @@ -100,7 +101,7 @@ pub struct DuplexMetrics {
pub min_ba_reads: usize,

/// Collect duplex UMI counts (memory intensive)
#[arg(long = "duplex-umi-counts", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "duplex-umi-counts", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub duplex_umi_counts: bool,

/// Optional intervals file to restrict analysis (BED or Picard interval list format)
Expand Down
3 changes: 2 additions & 1 deletion src/commands/fastq.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
//! This tool reads a BAM file and outputs interleaved FASTQ to stdout for piping to aligners.
//! Input should be queryname-sorted or template-coordinate sorted.

use crate::commands::common::parse_bool;
use anyhow::{Context, Result};
use clap::Parser;
use fgumi_lib::bam_io::create_bam_reader;
Expand Down Expand Up @@ -73,7 +74,7 @@ pub struct Fastq {
pub input: PathBuf,

/// Don't append /1 and /2 to read names.
#[arg(short = 'n', long = "no-read-suffix", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'n', long = "no-read-suffix", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub no_suffix: bool,

/// Exclude reads with any of these flags present [0x900 = secondary|supplementary].
Expand Down
8 changes: 4 additions & 4 deletions src/commands/filter.rs
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,7 @@ use std::time::Instant;
use crate::commands::command::Command;
use crate::commands::common::{
BamIoOptions, CompressionOptions, QueueMemoryOptions, SchedulerOptions, ThreadingOptions,
build_pipeline_config,
build_pipeline_config, parse_bool,
};

/// Filters and masks consensus reads based on various quality metrics.
Expand Down Expand Up @@ -147,15 +147,15 @@ pub struct Filter {
pub max_no_call_fraction: f64,

/// Reverse per-base tags for negative strand reads
#[arg(short = 'R', long = "reverse-per-base-tags", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 'R', long = "reverse-per-base-tags", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub reverse_per_base_tags: bool,

/// Threading options for parallel processing
#[command(flatten)]
pub threading: ThreadingOptions,

/// Filter templates together (all primary reads must pass)
#[arg(long = "filter-by-template", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(long = "filter-by-template", default_value = "true", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub filter_by_template: bool,

/// Optional output BAM file for rejected reads
Expand All @@ -167,7 +167,7 @@ pub struct Filter {
pub stats: Option<PathBuf>,

/// Require single-strand agreement for duplex consensus (mask bases where AB and BA disagree)
#[arg(short = 's', long = "require-single-strand-agreement", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set)]
#[arg(short = 's', long = "require-single-strand-agreement", default_value = "false", num_args = 0..=1, default_missing_value = "true", action = clap::ArgAction::Set, value_parser = parse_bool)]
pub require_single_strand_agreement: bool,

/// Compression options for output BAM.
Expand Down
Loading
Loading