feat: centralize genome metadata predicates in obikindex
Introduces a new predicate module in obikindex that implements genome metadata predicate parsing, evaluation, and group classification using three-valued logic. Extends the IndexMeta API with methods for single-predicate filtering and group quorum filtering. Updates obikmer command modules to delegate filter construction and matching to the centralized index API, removing local definitions and simplifying call sites.
This commit is contained in:
Generated
+1
@@ -1773,6 +1773,7 @@ dependencies = [
|
|||||||
"obiskbuilder",
|
"obiskbuilder",
|
||||||
"obiskio",
|
"obiskio",
|
||||||
"obisys",
|
"obisys",
|
||||||
|
"obitaxonomy",
|
||||||
"rayon",
|
"rayon",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ edition = "2024"
|
|||||||
[dependencies]
|
[dependencies]
|
||||||
obikseq = { path = "../obikseq" }
|
obikseq = { path = "../obikseq" }
|
||||||
obikpartitionner = { path = "../obikpartitionner" }
|
obikpartitionner = { path = "../obikpartitionner" }
|
||||||
|
obitaxonomy = { path = "../obitaxonomy" }
|
||||||
obiskio = { path = "../obiskio" }
|
obiskio = { path = "../obiskio" }
|
||||||
obisys = { path = "../obisys" }
|
obisys = { path = "../obisys" }
|
||||||
obicompactvec = { path = "../obicompactvec" }
|
obicompactvec = { path = "../obicompactvec" }
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
pub mod error;
|
pub mod error;
|
||||||
pub mod meta;
|
pub mod meta;
|
||||||
|
pub mod predicate;
|
||||||
pub mod state;
|
pub mod state;
|
||||||
mod cardcomp;
|
mod cardcomp;
|
||||||
mod distance;
|
mod distance;
|
||||||
@@ -18,6 +19,7 @@ pub use distance::{DistanceMetric, DistanceOutput};
|
|||||||
pub use index::KmerIndex;
|
pub use index::KmerIndex;
|
||||||
pub use merge::MergeMode;
|
pub use merge::MergeMode;
|
||||||
pub use meta::{validate_label, GenomeInfo, IndexConfig, IndexMeta, META_FILENAME};
|
pub use meta::{validate_label, GenomeInfo, IndexConfig, IndexMeta, META_FILENAME};
|
||||||
|
pub use predicate::{GroupFilterParams, MetaPred};
|
||||||
pub use state::{IndexState, SENTINEL_COUNTED, SENTINEL_INDEXED, SENTINEL_SCATTERED};
|
pub use state::{IndexState, SENTINEL_COUNTED, SENTINEL_INDEXED, SENTINEL_SCATTERED};
|
||||||
pub use stats::IndexBitsPerKmer;
|
pub use stats::IndexBitsPerKmer;
|
||||||
pub use siblings::{BasePairTally, CardinalityTally, RawSnpDistanceOutput, SiblingAnnexStats, SnpAlignment};
|
pub use siblings::{BasePairTally, CardinalityTally, RawSnpDistanceOutput, SiblingAnnexStats, SnpAlignment};
|
||||||
|
|||||||
@@ -0,0 +1,252 @@
|
|||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use obikpartitionner::GroupQuorumFilter;
|
||||||
|
use obitaxonomy::{TaxPath, TaxPattern};
|
||||||
|
|
||||||
|
use crate::meta::{GenomeInfo, IndexMeta};
|
||||||
|
|
||||||
|
// ── Operator ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
enum PredOp { Wildcard, Eq, Ne, Matches, NotMatches }
|
||||||
|
|
||||||
|
// ── MetaPred ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// A single predicate on genome metadata: `key OP val1|val2|…`
|
||||||
|
///
|
||||||
|
/// Operators: `=` (exact), `!=` (not equal), `~` (path ancestor), `!~` (not ancestor).
|
||||||
|
/// Multiple values separated by `|` are OR'd.
|
||||||
|
pub struct MetaPred {
|
||||||
|
key: String,
|
||||||
|
op: PredOp,
|
||||||
|
values: Vec<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MetaPred {
|
||||||
|
/// Parse a predicate string of the form `key=v1|v2`, `key!=v`, `key~path`, `key!~path`.
|
||||||
|
/// The special values `*` and `all` (case-insensitive) match every genome.
|
||||||
|
pub fn parse(s: &str) -> Result<Self, String> {
|
||||||
|
let t = s.trim();
|
||||||
|
if t == "*" || t.eq_ignore_ascii_case("all") {
|
||||||
|
return Ok(Self { key: String::new(), op: PredOp::Wildcard, values: vec![] });
|
||||||
|
}
|
||||||
|
|
||||||
|
let (op, key, rhs) =
|
||||||
|
if let Some(pos) = s.find("!=") {
|
||||||
|
(PredOp::Ne, &s[..pos], &s[pos+2..])
|
||||||
|
} else if let Some(pos) = s.find("!~") {
|
||||||
|
(PredOp::NotMatches, &s[..pos], &s[pos+2..])
|
||||||
|
} else if let Some(pos) = s.find('=') {
|
||||||
|
(PredOp::Eq, &s[..pos], &s[pos+1..])
|
||||||
|
} else if let Some(pos) = s.find('~') {
|
||||||
|
(PredOp::Matches, &s[..pos], &s[pos+1..])
|
||||||
|
} else {
|
||||||
|
return Err(format!("no operator found in predicate: {s}"));
|
||||||
|
};
|
||||||
|
|
||||||
|
let key = key.trim().to_string();
|
||||||
|
if key.is_empty() { return Err(format!("empty key in predicate: {s}")); }
|
||||||
|
|
||||||
|
let values: Vec<String> = rhs.split('|').map(|v| v.trim().to_string()).collect();
|
||||||
|
if values.iter().any(|v| v.is_empty()) {
|
||||||
|
return Err(format!("empty value in predicate: {s}"));
|
||||||
|
}
|
||||||
|
Ok(Self { key, op, values })
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate against one genome's metadata.
|
||||||
|
/// Returns `None` when the key is absent (NA propagation).
|
||||||
|
pub(crate) fn eval(&self, meta: &HashMap<String, String>) -> Option<bool> {
|
||||||
|
if matches!(self.op, PredOp::Wildcard) { return Some(true); }
|
||||||
|
let value = meta.get(&self.key)?;
|
||||||
|
Some(match self.op {
|
||||||
|
PredOp::Wildcard => unreachable!(),
|
||||||
|
PredOp::Eq => self.values.iter().any(|v| v == value),
|
||||||
|
PredOp::Ne => self.values.iter().all(|v| v != value),
|
||||||
|
PredOp::Matches => self.values.iter().any(|v| path_matches(value, v)),
|
||||||
|
PredOp::NotMatches => self.values.iter().all(|v| !path_matches(value, v)),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl GenomeInfo {
|
||||||
|
/// Evaluate a single metadata predicate against this genome.
|
||||||
|
/// Returns `None` when the predicate's key is absent (NA propagation).
|
||||||
|
pub fn matches(&self, pred: &MetaPred) -> Option<bool> {
|
||||||
|
pred.eval(&self.meta)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Path matching ─────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// True if the stored taxonomy `value` matches `pattern`.
|
||||||
|
///
|
||||||
|
/// `value` must be a valid `TaxPath` (starts with `taxonomy:/`).
|
||||||
|
/// `pattern` is a `TaxPattern` query (see `obitaxonomy::TaxPattern` for syntax).
|
||||||
|
/// Returns `false` if either fails to parse.
|
||||||
|
fn path_matches(value: &str, pattern: &str) -> bool {
|
||||||
|
let Ok(path) = TaxPath::parse(value) else { return false };
|
||||||
|
let Ok(pat) = TaxPattern::parse(pattern) else { return false };
|
||||||
|
pat.matches(&path)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Three-value group evaluation ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// AND of all predicates (ingroup semantics).
|
||||||
|
/// Short-circuits on `Some(false)`; propagates `None` if no predicate returns `false`.
|
||||||
|
fn eval_and(preds: &[MetaPred], meta: &HashMap<String, String>) -> Option<bool> {
|
||||||
|
let mut has_na = false;
|
||||||
|
for pred in preds {
|
||||||
|
match pred.eval(meta) {
|
||||||
|
Some(false) => return Some(false),
|
||||||
|
Some(true) => {}
|
||||||
|
None => has_na = true,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if has_na { None } else { Some(true) }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// OR of all predicates (outgroup semantics).
|
||||||
|
/// Short-circuits on `Some(true)`; propagates `None` if no predicate returns `true`.
|
||||||
|
fn eval_or(preds: &[MetaPred], meta: &HashMap<String, String>) -> Option<bool> {
|
||||||
|
let mut has_na = false;
|
||||||
|
for pred in preds {
|
||||||
|
match pred.eval(meta) {
|
||||||
|
Some(true) => return Some(true),
|
||||||
|
Some(false) => {}
|
||||||
|
None => has_na = true,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if has_na { None } else { Some(false) }
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Genome classification ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
enum Membership { Ingroup, Outgroup, Uncategorized }
|
||||||
|
|
||||||
|
fn classify(
|
||||||
|
genomes: &[GenomeInfo],
|
||||||
|
ingroup: &[MetaPred],
|
||||||
|
outgroup: &[MetaPred],
|
||||||
|
) -> Vec<Membership> {
|
||||||
|
genomes.iter().map(|g| {
|
||||||
|
let in_r = if ingroup.is_empty() { None } else { eval_and(ingroup, &g.meta) };
|
||||||
|
let out_r = if outgroup.is_empty() { None } else { eval_or(outgroup, &g.meta) };
|
||||||
|
|
||||||
|
// Ingroup wins over outgroup.
|
||||||
|
if in_r == Some(true) { return Membership::Ingroup; }
|
||||||
|
if out_r == Some(true) { return Membership::Outgroup; }
|
||||||
|
Membership::Uncategorized
|
||||||
|
}).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Group quorum filter construction ──────────────────────────────────────────
|
||||||
|
|
||||||
|
pub struct GroupFilterParams {
|
||||||
|
pub threshold: u32,
|
||||||
|
pub min_count: Option<isize>,
|
||||||
|
pub max_count: Option<isize>,
|
||||||
|
pub min_frac: Option<f64>,
|
||||||
|
pub max_frac: Option<f64>,
|
||||||
|
pub min_outgroup_count: Option<isize>,
|
||||||
|
pub max_outgroup_count: Option<isize>,
|
||||||
|
pub min_outgroup_frac: Option<f64>,
|
||||||
|
pub max_outgroup_frac: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl IndexMeta {
|
||||||
|
/// Returns indices of genomes matching `pred_str` (single predicate).
|
||||||
|
pub fn matching_genome_indices(&self, pred_str: &str) -> Result<Vec<usize>, String> {
|
||||||
|
let pred = MetaPred::parse(pred_str)?;
|
||||||
|
Ok(self.genomes.iter().enumerate()
|
||||||
|
.filter_map(|(i, g)| {
|
||||||
|
if g.matches(&pred) == Some(true) { Some(i) } else { std::option::Option::None }
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build a `GroupQuorumFilter` from parsed predicates, evaluated against `self.genomes`.
|
||||||
|
///
|
||||||
|
/// - No groups defined: `ingroup_idx` = all genomes (implicit ingroup).
|
||||||
|
/// - `ingroup` predicates only: outgroup indices are empty.
|
||||||
|
/// - `outgroup` predicates only: ingroup indices are empty.
|
||||||
|
/// - Both defined: ingroup wins on overlap; uncategorized genomes are ignored.
|
||||||
|
pub fn build_group_filter(
|
||||||
|
&self,
|
||||||
|
ingroup_preds: &[MetaPred],
|
||||||
|
outgroup_preds: &[MetaPred],
|
||||||
|
p: GroupFilterParams,
|
||||||
|
) -> Result<GroupQuorumFilter, String> {
|
||||||
|
let (ingroup_idx, outgroup_idx) = if ingroup_preds.is_empty() && outgroup_preds.is_empty() {
|
||||||
|
((0..self.genomes.len()).collect(), vec![])
|
||||||
|
} else {
|
||||||
|
let members = classify(&self.genomes, ingroup_preds, outgroup_preds);
|
||||||
|
let in_idx: Vec<usize> = members.iter().enumerate()
|
||||||
|
.filter(|(_, m)| matches!(m, Membership::Ingroup))
|
||||||
|
.map(|(i, _)| i).collect();
|
||||||
|
let out_idx: Vec<usize> = members.iter().enumerate()
|
||||||
|
.filter(|(_, m)| matches!(m, Membership::Outgroup))
|
||||||
|
.map(|(i, _)| i).collect();
|
||||||
|
(in_idx, out_idx)
|
||||||
|
};
|
||||||
|
|
||||||
|
let in_size = ingroup_idx.len();
|
||||||
|
let out_size = outgroup_idx.len();
|
||||||
|
|
||||||
|
let ingroup_quorum_explicit = p.min_count.is_some() || p.max_count.is_some()
|
||||||
|
|| p.min_frac.is_some() || p.max_frac.is_some();
|
||||||
|
let outgroup_quorum_explicit = p.min_outgroup_count.is_some() || p.max_outgroup_count.is_some()
|
||||||
|
|| p.min_outgroup_frac.is_some() || p.max_outgroup_frac.is_some();
|
||||||
|
|
||||||
|
let default_min_frac = if !ingroup_preds.is_empty() && !ingroup_quorum_explicit { 1.0 } else { 0.0 };
|
||||||
|
let default_max_outgroup_count = if !outgroup_preds.is_empty() && !outgroup_quorum_explicit { 0 } else { out_size };
|
||||||
|
|
||||||
|
// Resolve a signed count: negative means an offset from the group size
|
||||||
|
// (e.g. -1 = all but one), floored at 1 so the negative form always keeps
|
||||||
|
// constraining the group — even a singleton group, where n-1 would be 0
|
||||||
|
// and would otherwise drop the constraint entirely.
|
||||||
|
let resolve = |v: isize, size: usize| -> usize {
|
||||||
|
if v < 0 { (size as isize + v).max(1) as usize } else { v as usize }
|
||||||
|
};
|
||||||
|
|
||||||
|
let min_count = p.min_count.map(|v| resolve(v, in_size)).unwrap_or(0);
|
||||||
|
let max_count = p.max_count.map(|v| resolve(v, in_size)).unwrap_or(in_size);
|
||||||
|
let min_frac = p.min_frac.unwrap_or(default_min_frac);
|
||||||
|
let max_frac = p.max_frac.unwrap_or(1.0);
|
||||||
|
let min_outgroup_count = p.min_outgroup_count.map(|v| resolve(v, out_size)).unwrap_or(0);
|
||||||
|
let max_outgroup_count = p.max_outgroup_count.map(|v| resolve(v, out_size)).unwrap_or(default_max_outgroup_count);
|
||||||
|
let min_outgroup_frac = p.min_outgroup_frac.unwrap_or(0.0);
|
||||||
|
let max_outgroup_frac = p.max_outgroup_frac.unwrap_or(1.0);
|
||||||
|
|
||||||
|
for (v, lo, hi) in [
|
||||||
|
("--min-frac/--max-frac", min_frac, max_frac),
|
||||||
|
("--min-outgroup-frac/--max-outgroup-frac", min_outgroup_frac, max_outgroup_frac),
|
||||||
|
] {
|
||||||
|
if !(0.0..=1.0).contains(&lo) || !(0.0..=1.0).contains(&hi) {
|
||||||
|
return Err(format!("{v}: fraction values must be in [0.0, 1.0]"));
|
||||||
|
}
|
||||||
|
if lo > hi {
|
||||||
|
return Err(format!("{v}: min ({lo}) is greater than max ({hi})"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if min_count > max_count {
|
||||||
|
return Err(format!("--min-count/--max-count: min ({min_count}) is greater than max ({max_count})"));
|
||||||
|
}
|
||||||
|
if min_outgroup_count > max_outgroup_count {
|
||||||
|
return Err(format!("--min-outgroup-count/--max-outgroup-count: min ({min_outgroup_count}) is greater than max ({max_outgroup_count})"));
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(GroupQuorumFilter {
|
||||||
|
ingroup_idx,
|
||||||
|
outgroup_idx,
|
||||||
|
threshold: p.threshold,
|
||||||
|
min_count,
|
||||||
|
max_count,
|
||||||
|
min_frac,
|
||||||
|
max_frac,
|
||||||
|
min_outgroup_count,
|
||||||
|
max_outgroup_count,
|
||||||
|
min_outgroup_frac,
|
||||||
|
max_outgroup_frac,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -41,7 +41,7 @@ pub fn run(args: DumpArgs) {
|
|||||||
&idx.meta().genomes.len()
|
&idx.meta().genomes.len()
|
||||||
);
|
);
|
||||||
|
|
||||||
let filters = args.filter.build_filters(&idx.meta().genomes);
|
let filters = args.filter.build_filters(idx.meta());
|
||||||
let pb = progress_bar("dump", idx.n_partitions() as u64, "partitions");
|
let pb = progress_bar("dump", idx.n_partitions() as u64, "partitions");
|
||||||
|
|
||||||
let stdout = io::stdout();
|
let stdout = io::stdout();
|
||||||
|
|||||||
@@ -66,7 +66,7 @@ pub fn run(args: FilterCmdArgs) {
|
|||||||
&src.meta().genomes.len(), mode, args.source.display()
|
&src.meta().genomes.len(), mode, args.source.display()
|
||||||
);
|
);
|
||||||
|
|
||||||
let mut filters = args.filter.build_filters(&src.meta().genomes);
|
let mut filters = args.filter.build_filters(src.meta());
|
||||||
|
|
||||||
if let Some(v) = args.min_total_count {
|
if let Some(v) = args.min_total_count {
|
||||||
filters.push(Box::new(MinTotalCount { total: v }));
|
filters.push(Box::new(MinTotalCount { total: v }));
|
||||||
|
|||||||
@@ -1,144 +1,7 @@
|
|||||||
use std::collections::HashMap;
|
|
||||||
|
|
||||||
use clap::Args;
|
use clap::Args;
|
||||||
use obikindex::GenomeInfo;
|
use obikindex::{GroupFilterParams, IndexMeta, MetaPred};
|
||||||
use obikpartitionner::{GroupQuorumFilter, KmerFilter};
|
use obikpartitionner::KmerFilter;
|
||||||
use obitaxonomy::{TaxPath, TaxPattern};
|
|
||||||
|
|
||||||
// ── Operator ──────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
enum PredOp { Wildcard, Eq, Ne, Matches, NotMatches }
|
|
||||||
|
|
||||||
// ── MetaPred ──────────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// A single predicate on genome metadata: `key OP val1|val2|…`
|
|
||||||
///
|
|
||||||
/// Operators: `=` (exact), `!=` (not equal), `~` (path ancestor), `!~` (not ancestor).
|
|
||||||
/// Multiple values separated by `|` are OR'd.
|
|
||||||
pub struct MetaPred {
|
|
||||||
key: String,
|
|
||||||
op: PredOp,
|
|
||||||
values: Vec<String>,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl MetaPred {
|
|
||||||
/// Parse a predicate string of the form `key=v1|v2`, `key!=v`, `key~path`, `key!~path`.
|
|
||||||
/// The special values `*` and `all` (case-insensitive) match every genome.
|
|
||||||
pub fn parse(s: &str) -> Result<Self, String> {
|
|
||||||
let t = s.trim();
|
|
||||||
if t == "*" || t.eq_ignore_ascii_case("all") {
|
|
||||||
return Ok(Self { key: String::new(), op: PredOp::Wildcard, values: vec![] });
|
|
||||||
}
|
|
||||||
|
|
||||||
let (op, key, rhs) =
|
|
||||||
if let Some(pos) = s.find("!=") {
|
|
||||||
(PredOp::Ne, &s[..pos], &s[pos+2..])
|
|
||||||
} else if let Some(pos) = s.find("!~") {
|
|
||||||
(PredOp::NotMatches, &s[..pos], &s[pos+2..])
|
|
||||||
} else if let Some(pos) = s.find('=') {
|
|
||||||
(PredOp::Eq, &s[..pos], &s[pos+1..])
|
|
||||||
} else if let Some(pos) = s.find('~') {
|
|
||||||
(PredOp::Matches, &s[..pos], &s[pos+1..])
|
|
||||||
} else {
|
|
||||||
return Err(format!("no operator found in predicate: {s}"));
|
|
||||||
};
|
|
||||||
|
|
||||||
let key = key.trim().to_string();
|
|
||||||
if key.is_empty() { return Err(format!("empty key in predicate: {s}")); }
|
|
||||||
|
|
||||||
let values: Vec<String> = rhs.split('|').map(|v| v.trim().to_string()).collect();
|
|
||||||
if values.iter().any(|v| v.is_empty()) {
|
|
||||||
return Err(format!("empty value in predicate: {s}"));
|
|
||||||
}
|
|
||||||
Ok(Self { key, op, values })
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate against one genome's metadata.
|
|
||||||
/// Returns `None` when the key is absent (NA propagation).
|
|
||||||
fn eval(&self, meta: &HashMap<String, String>) -> Option<bool> {
|
|
||||||
if matches!(self.op, PredOp::Wildcard) { return Some(true); }
|
|
||||||
let value = meta.get(&self.key)?;
|
|
||||||
Some(match self.op {
|
|
||||||
PredOp::Wildcard => unreachable!(),
|
|
||||||
PredOp::Eq => self.values.iter().any(|v| v == value),
|
|
||||||
PredOp::Ne => self.values.iter().all(|v| v != value),
|
|
||||||
PredOp::Matches => self.values.iter().any(|v| path_matches(value, v)),
|
|
||||||
PredOp::NotMatches => self.values.iter().all(|v| !path_matches(value, v)),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Path matching ─────────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// True if the stored taxonomy `value` matches `pattern`.
|
|
||||||
///
|
|
||||||
/// `value` must be a valid `TaxPath` (starts with `taxonomy:/`).
|
|
||||||
/// `pattern` is a `TaxPattern` query (see `obitaxonomy::TaxPattern` for syntax).
|
|
||||||
/// Returns `false` if either fails to parse.
|
|
||||||
fn path_matches(value: &str, pattern: &str) -> bool {
|
|
||||||
let Ok(path) = TaxPath::parse(value) else { return false };
|
|
||||||
let Ok(pat) = TaxPattern::parse(pattern) else { return false };
|
|
||||||
pat.matches(&path)
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Three-value group evaluation ──────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// AND of all predicates (ingroup semantics).
|
|
||||||
/// Short-circuits on `Some(false)`; propagates `None` if no predicate returns `false`.
|
|
||||||
fn eval_and(preds: &[MetaPred], meta: &HashMap<String, String>) -> Option<bool> {
|
|
||||||
let mut has_na = false;
|
|
||||||
for pred in preds {
|
|
||||||
match pred.eval(meta) {
|
|
||||||
Some(false) => return Some(false),
|
|
||||||
Some(true) => {}
|
|
||||||
None => has_na = true,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if has_na { None } else { Some(true) }
|
|
||||||
}
|
|
||||||
|
|
||||||
/// OR of all predicates (outgroup semantics).
|
|
||||||
/// Short-circuits on `Some(true)`; propagates `None` if no predicate returns `true`.
|
|
||||||
fn eval_or(preds: &[MetaPred], meta: &HashMap<String, String>) -> Option<bool> {
|
|
||||||
let mut has_na = false;
|
|
||||||
for pred in preds {
|
|
||||||
match pred.eval(meta) {
|
|
||||||
Some(true) => return Some(true),
|
|
||||||
Some(false) => {}
|
|
||||||
None => has_na = true,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if has_na { None } else { Some(false) }
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Genome classification ─────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
enum Membership { Ingroup, Outgroup, Uncategorized }
|
|
||||||
|
|
||||||
fn classify(
|
|
||||||
genomes: &[GenomeInfo],
|
|
||||||
ingroup: &[MetaPred],
|
|
||||||
outgroup: &[MetaPred],
|
|
||||||
) -> Vec<Membership> {
|
|
||||||
genomes.iter().map(|g| {
|
|
||||||
let in_r = if ingroup.is_empty() { None } else { eval_and(ingroup, &g.meta) };
|
|
||||||
let out_r = if outgroup.is_empty() { None } else { eval_or(outgroup, &g.meta) };
|
|
||||||
|
|
||||||
// Ingroup wins over outgroup.
|
|
||||||
if in_r == Some(true) { return Membership::Ingroup; }
|
|
||||||
if out_r == Some(true) { return Membership::Outgroup; }
|
|
||||||
Membership::Uncategorized
|
|
||||||
}).collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
// ── Public constructor ────────────────────────────────────────────────────────
|
|
||||||
|
|
||||||
/// Build a `GroupQuorumFilter` from parsed predicates and genome metadata.
|
|
||||||
///
|
|
||||||
/// - No groups defined: `ingroup_idx` = all genomes (implicit ingroup).
|
|
||||||
/// - `ingroup` predicates only: outgroup indices are empty.
|
|
||||||
/// - `outgroup` predicates only: ingroup indices are empty.
|
|
||||||
/// - Both defined: ingroup wins on overlap; uncategorized genomes are ignored.
|
|
||||||
/// CLI args for ingroup/outgroup filtering — embeddable in any command via `#[command(flatten)]`.
|
/// CLI args for ingroup/outgroup filtering — embeddable in any command via `#[command(flatten)]`.
|
||||||
#[derive(Args)]
|
#[derive(Args)]
|
||||||
pub struct FilterArgs {
|
pub struct FilterArgs {
|
||||||
@@ -195,7 +58,7 @@ pub struct FilterArgs {
|
|||||||
|
|
||||||
impl FilterArgs {
|
impl FilterArgs {
|
||||||
/// Parse predicates and build a filter list ready to pass to `iter_partition_kmers`.
|
/// Parse predicates and build a filter list ready to pass to `iter_partition_kmers`.
|
||||||
pub fn build_filters(&self, genomes: &[GenomeInfo]) -> Vec<Box<dyn KmerFilter>> {
|
pub fn build_filters(&self, meta: &IndexMeta) -> Vec<Box<dyn KmerFilter>> {
|
||||||
let ingroup_preds: Vec<MetaPred> = self.ingroup.iter()
|
let ingroup_preds: Vec<MetaPred> = self.ingroup.iter()
|
||||||
.map(|s| MetaPred::parse(s).unwrap_or_else(|e| {
|
.map(|s| MetaPred::parse(s).unwrap_or_else(|e| {
|
||||||
eprintln!("error in --ingroup: {e}");
|
eprintln!("error in --ingroup: {e}");
|
||||||
@@ -208,8 +71,7 @@ impl FilterArgs {
|
|||||||
std::process::exit(1);
|
std::process::exit(1);
|
||||||
}))
|
}))
|
||||||
.collect();
|
.collect();
|
||||||
let filter = build_group_filter(
|
let filter = meta.build_group_filter(
|
||||||
genomes,
|
|
||||||
&ingroup_preds,
|
&ingroup_preds,
|
||||||
&outgroup_preds,
|
&outgroup_preds,
|
||||||
GroupFilterParams {
|
GroupFilterParams {
|
||||||
@@ -230,105 +92,3 @@ impl FilterArgs {
|
|||||||
vec![Box::new(filter)]
|
vec![Box::new(filter)]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns indices of genomes matching `pred_str` (single predicate).
|
|
||||||
pub fn matching_genome_indices(pred_str: &str, genomes: &[GenomeInfo]) -> Result<Vec<usize>, String> {
|
|
||||||
let pred = MetaPred::parse(pred_str)?;
|
|
||||||
Ok(genomes.iter().enumerate()
|
|
||||||
.filter_map(|(i, g)| {
|
|
||||||
if pred.eval(&g.meta) == Some(true) { Some(i) } else { std::option::Option::None }
|
|
||||||
})
|
|
||||||
.collect())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct GroupFilterParams {
|
|
||||||
pub threshold: u32,
|
|
||||||
pub min_count: Option<isize>,
|
|
||||||
pub max_count: Option<isize>,
|
|
||||||
pub min_frac: Option<f64>,
|
|
||||||
pub max_frac: Option<f64>,
|
|
||||||
pub min_outgroup_count: Option<isize>,
|
|
||||||
pub max_outgroup_count: Option<isize>,
|
|
||||||
pub min_outgroup_frac: Option<f64>,
|
|
||||||
pub max_outgroup_frac: Option<f64>,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn build_group_filter(
|
|
||||||
genomes: &[GenomeInfo],
|
|
||||||
ingroup_preds: &[MetaPred],
|
|
||||||
outgroup_preds: &[MetaPred],
|
|
||||||
p: GroupFilterParams,
|
|
||||||
) -> Result<GroupQuorumFilter, String> {
|
|
||||||
let (ingroup_idx, outgroup_idx) = if ingroup_preds.is_empty() && outgroup_preds.is_empty() {
|
|
||||||
((0..genomes.len()).collect(), vec![])
|
|
||||||
} else {
|
|
||||||
let members = classify(genomes, ingroup_preds, outgroup_preds);
|
|
||||||
let in_idx: Vec<usize> = members.iter().enumerate()
|
|
||||||
.filter(|(_, m)| matches!(m, Membership::Ingroup))
|
|
||||||
.map(|(i, _)| i).collect();
|
|
||||||
let out_idx: Vec<usize> = members.iter().enumerate()
|
|
||||||
.filter(|(_, m)| matches!(m, Membership::Outgroup))
|
|
||||||
.map(|(i, _)| i).collect();
|
|
||||||
(in_idx, out_idx)
|
|
||||||
};
|
|
||||||
|
|
||||||
let in_size = ingroup_idx.len();
|
|
||||||
let out_size = outgroup_idx.len();
|
|
||||||
|
|
||||||
let ingroup_quorum_explicit = p.min_count.is_some() || p.max_count.is_some()
|
|
||||||
|| p.min_frac.is_some() || p.max_frac.is_some();
|
|
||||||
let outgroup_quorum_explicit = p.min_outgroup_count.is_some() || p.max_outgroup_count.is_some()
|
|
||||||
|| p.min_outgroup_frac.is_some() || p.max_outgroup_frac.is_some();
|
|
||||||
|
|
||||||
let default_min_frac = if !ingroup_preds.is_empty() && !ingroup_quorum_explicit { 1.0 } else { 0.0 };
|
|
||||||
let default_max_outgroup_count = if !outgroup_preds.is_empty() && !outgroup_quorum_explicit { 0 } else { out_size };
|
|
||||||
|
|
||||||
// Resolve a signed count: negative means an offset from the group size
|
|
||||||
// (e.g. -1 = all but one), floored at 1 so the negative form always keeps
|
|
||||||
// constraining the group — even a singleton group, where n-1 would be 0
|
|
||||||
// and would otherwise drop the constraint entirely.
|
|
||||||
let resolve = |v: isize, size: usize| -> usize {
|
|
||||||
if v < 0 { (size as isize + v).max(1) as usize } else { v as usize }
|
|
||||||
};
|
|
||||||
|
|
||||||
let min_count = p.min_count.map(|v| resolve(v, in_size)).unwrap_or(0);
|
|
||||||
let max_count = p.max_count.map(|v| resolve(v, in_size)).unwrap_or(in_size);
|
|
||||||
let min_frac = p.min_frac.unwrap_or(default_min_frac);
|
|
||||||
let max_frac = p.max_frac.unwrap_or(1.0);
|
|
||||||
let min_outgroup_count = p.min_outgroup_count.map(|v| resolve(v, out_size)).unwrap_or(0);
|
|
||||||
let max_outgroup_count = p.max_outgroup_count.map(|v| resolve(v, out_size)).unwrap_or(default_max_outgroup_count);
|
|
||||||
let min_outgroup_frac = p.min_outgroup_frac.unwrap_or(0.0);
|
|
||||||
let max_outgroup_frac = p.max_outgroup_frac.unwrap_or(1.0);
|
|
||||||
|
|
||||||
for (v, lo, hi) in [
|
|
||||||
("--min-frac/--max-frac", min_frac, max_frac),
|
|
||||||
("--min-outgroup-frac/--max-outgroup-frac", min_outgroup_frac, max_outgroup_frac),
|
|
||||||
] {
|
|
||||||
if !(0.0..=1.0).contains(&lo) || !(0.0..=1.0).contains(&hi) {
|
|
||||||
return Err(format!("{v}: fraction values must be in [0.0, 1.0]"));
|
|
||||||
}
|
|
||||||
if lo > hi {
|
|
||||||
return Err(format!("{v}: min ({lo}) is greater than max ({hi})"));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if min_count > max_count {
|
|
||||||
return Err(format!("--min-count/--max-count: min ({min_count}) is greater than max ({max_count})"));
|
|
||||||
}
|
|
||||||
if min_outgroup_count > max_outgroup_count {
|
|
||||||
return Err(format!("--min-outgroup-count/--max-outgroup-count: min ({min_outgroup_count}) is greater than max ({max_outgroup_count})"));
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(GroupQuorumFilter {
|
|
||||||
ingroup_idx,
|
|
||||||
outgroup_idx,
|
|
||||||
threshold: p.threshold,
|
|
||||||
min_count,
|
|
||||||
max_count,
|
|
||||||
min_frac,
|
|
||||||
max_frac,
|
|
||||||
min_outgroup_count,
|
|
||||||
max_outgroup_count,
|
|
||||||
min_outgroup_frac,
|
|
||||||
max_outgroup_frac,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -2,13 +2,11 @@ use std::collections::{BTreeMap, HashMap};
|
|||||||
use std::path::PathBuf;
|
use std::path::PathBuf;
|
||||||
|
|
||||||
use clap::{Args, ValueEnum};
|
use clap::{Args, ValueEnum};
|
||||||
use obikindex::{GenomeInfo, KmerIndex};
|
use obikindex::{IndexMeta, KmerIndex};
|
||||||
use obikpartitionner::{AggOp, OutputCol};
|
use obikpartitionner::{AggOp, OutputCol};
|
||||||
use obisys::Reporter;
|
use obisys::Reporter;
|
||||||
use tracing::info;
|
use tracing::info;
|
||||||
|
|
||||||
use super::predicate::matching_genome_indices;
|
|
||||||
|
|
||||||
// ── CLI types ─────────────────────────────────────────────────────────────────
|
// ── CLI types ─────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, ValueEnum)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, ValueEnum)]
|
||||||
@@ -114,9 +112,11 @@ fn default_op(src_is_count: bool) -> AggOp {
|
|||||||
/// Returns `(specs, output_presence)`.
|
/// Returns `(specs, output_presence)`.
|
||||||
fn build_specs(
|
fn build_specs(
|
||||||
args: &SelectArgs,
|
args: &SelectArgs,
|
||||||
genomes: &[GenomeInfo],
|
meta: &IndexMeta,
|
||||||
src_is_count: bool,
|
src_is_count: bool,
|
||||||
) -> (Vec<OutputCol>, bool) {
|
) -> (Vec<OutputCol>, bool) {
|
||||||
|
let genomes = &meta.genomes;
|
||||||
|
|
||||||
// ── 1. Build group_indices: name → Vec<usize> ────────────────────────────
|
// ── 1. Build group_indices: name → Vec<usize> ────────────────────────────
|
||||||
// Also keep insertion order for the default `--select *` case.
|
// Also keep insertion order for the default `--select *` case.
|
||||||
let mut group_order: Vec<String> = Vec::new();
|
let mut group_order: Vec<String> = Vec::new();
|
||||||
@@ -137,7 +137,7 @@ fn build_specs(
|
|||||||
} else {
|
} else {
|
||||||
for raw in &args.group {
|
for raw in &args.group {
|
||||||
let (name, pred) = parse_name_value(raw, "--group");
|
let (name, pred) = parse_name_value(raw, "--group");
|
||||||
let idxs = matching_genome_indices(&pred, genomes).unwrap_or_else(|e| {
|
let idxs = meta.matching_genome_indices(&pred).unwrap_or_else(|e| {
|
||||||
eprintln!("error in --group {name}: {e}");
|
eprintln!("error in --group {name}: {e}");
|
||||||
std::process::exit(1);
|
std::process::exit(1);
|
||||||
});
|
});
|
||||||
@@ -231,7 +231,7 @@ pub fn run(args: SelectArgs) {
|
|||||||
});
|
});
|
||||||
|
|
||||||
let src_is_count = src.meta().config.with_counts;
|
let src_is_count = src.meta().config.with_counts;
|
||||||
let (specs, output_presence) = build_specs(&args, &src.meta().genomes.clone(), src_is_count);
|
let (specs, output_presence) = build_specs(&args, src.meta(), src_is_count);
|
||||||
|
|
||||||
info!(
|
info!(
|
||||||
"select: {} genome(s) → {} output column(s), output={}",
|
"select: {} genome(s) → {} output column(s), output={}",
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ pub fn run(args: UnitigArgs) {
|
|||||||
|
|
||||||
info!("unitig: building de Bruijn graph from {n} partition(s) (k={k})");
|
info!("unitig: building de Bruijn graph from {n} partition(s) (k={k})");
|
||||||
|
|
||||||
let filters = args.filter.build_filters(&idx.meta().genomes);
|
let filters = args.filter.build_filters(idx.meta());
|
||||||
let partition = idx.partition();
|
let partition = idx.partition();
|
||||||
let mut rep = Reporter::new();
|
let mut rep = Reporter::new();
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user